From 08c3e854403e184f1b5badac67389dacee09bfe4 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:11:04 -0700 Subject: [PATCH 01/17] test(e2e): stabilize terminal launch and rename menu fixtures (#18928) --- ...ackground-terminal-mount-authority.spec.ts | 47 ++++++++++++------- tests/e2e/tab-rename.spec.ts | 2 +- 2 files changed, 31 insertions(+), 18 deletions(-) diff --git a/tests/e2e/live-background-terminal-mount-authority.spec.ts b/tests/e2e/live-background-terminal-mount-authority.spec.ts index ea6ffd1871a..a785454f695 100644 --- a/tests/e2e/live-background-terminal-mount-authority.spec.ts +++ b/tests/e2e/live-background-terminal-mount-authority.spec.ts @@ -23,6 +23,10 @@ import type { } from '../../src/shared/runtime-types' import { PROTOCOL_VERSION } from '../../src/main/daemon/types' import { makePaneKey } from '../../src/shared/stable-pane-id' +import { + buildFakeAgentCommandOverride, + FAKE_AGENT_WINDOWS_SHELL +} from './helpers/fake-agent-command-override' type SpawnEvent = { args: string[]; pid: number } type TerminalIdentity = Pick< @@ -70,6 +74,10 @@ if (process.platform === 'win32') { chmodSync(executable, 0o755) } +const fakeCodexCommand = buildFakeAgentCommandOverride( + path.join(fakeCliDir, process.platform === 'win32' ? 'codex.cmd' : 'codex') +) + const test = base.extend({ launchEnv: [ { @@ -535,23 +543,28 @@ test('adopts runtime-owned agent and Setup PTYs on first mount', async ({ const repoId = added.result.repo.id await expect .poll(() => - orcaPage.evaluate(async (repoId) => { - const state = window.__store?.getState() - await state?.fetchRepos() - const repo = window.__store?.getState().repos.find((candidate) => candidate.id === repoId) - if (!repo) { - return false - } - await window.__store?.getState().updateRepo(repoId, { - hookSettings: { ...repo.hookSettings, setupAgentStartupPolicy: 'start-immediately' } - }) - await window.__store?.getState().updateSettings({ - disabledTuiAgents: [], - setupScriptLaunchMode: 'new-tab', - terminalHiddenViewParking: false - }) - return true - }, repoId) + orcaPage.evaluate( + async ({ repoId, command, windowsShell }) => { + const state = window.__store?.getState() + await state?.fetchRepos() + const repo = window.__store?.getState().repos.find((candidate) => candidate.id === repoId) + if (!repo) { + return false + } + await window.__store?.getState().updateRepo(repoId, { + hookSettings: { ...repo.hookSettings, setupAgentStartupPolicy: 'start-immediately' } + }) + await window.__store?.getState().updateSettings({ + agentCmdOverrides: { codex: command }, + terminalWindowsShell: windowsShell, + disabledTuiAgents: [], + setupScriptLaunchMode: 'new-tab', + terminalHiddenViewParking: false + }) + return true + }, + { repoId, command: fakeCodexCommand, windowsShell: FAKE_AGENT_WINDOWS_SHELL } + ) ) .toBe(true) diff --git a/tests/e2e/tab-rename.spec.ts b/tests/e2e/tab-rename.spec.ts index 6e7f0a7fdc1..30cdb9175a9 100644 --- a/tests/e2e/tab-rename.spec.ts +++ b/tests/e2e/tab-rename.spec.ts @@ -126,7 +126,7 @@ test.describe('Tab Rename (Inline)', () => { expect(originalTitle.length).toBeGreaterThan(0) await tabLocatorByTitle(orcaPage, originalTitle).click({ button: 'right' }) - await orcaPage.getByRole('menuitem', { name: 'Change Title', exact: true }).click() + await orcaPage.getByRole('menuitem', { name: /^Change Title(?:\s|$)/ }).click() const renameInput = orcaPage.getByRole('textbox', { name: `Rename tab ${originalTitle}`, From a730becd7a6274b61b141b205b0d94f27f3a5e0b Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:12:52 -0700 Subject: [PATCH 02/17] fix(automation): keep explicit background launches off screen (#18898) --- config/scripts/run-electron-vite-dev.mjs | 2 +- .../createMainWindow-startup-reveal.test.ts | 28 ++++++++++ src/main/window/focus-existing-window.test.ts | 27 +++++++++- src/main/window/focus-existing-window.ts | 8 ++- .../foreground-activation-policy.test.ts | 51 +++++++++++++------ .../window/foreground-activation-policy.ts | 22 ++++---- tests/AGENTS.md | 12 +++-- 7 files changed, 116 insertions(+), 34 deletions(-) diff --git a/config/scripts/run-electron-vite-dev.mjs b/config/scripts/run-electron-vite-dev.mjs index dfb0a0aceb7..c520083cb6a 100644 --- a/config/scripts/run-electron-vite-dev.mjs +++ b/config/scripts/run-electron-vite-dev.mjs @@ -616,7 +616,7 @@ if (!isHelpOrVersion && process.env.ORCA_DEV_INSTANCE_LABEL) { // Why: automation launches this app while someone is working; announce that the // window will come up without taking the foreground so the mode is visible in logs. if (!isHelpOrVersion && process.env.ORCA_BACKGROUND_LAUNCH === '1') { - console.error('[orca-dev] Background launch: window shows without stealing focus') + console.error('[orca-dev] Background launch: window stays off screen; automate through CDP') } let forwardedExtras = [] if (!userPassedPort && !isHelpOrVersion) { diff --git a/src/main/window/createMainWindow-startup-reveal.test.ts b/src/main/window/createMainWindow-startup-reveal.test.ts index 1200132d319..f103881ea83 100644 --- a/src/main/window/createMainWindow-startup-reveal.test.ts +++ b/src/main/window/createMainWindow-startup-reveal.test.ts @@ -78,6 +78,34 @@ describe('createMainWindow', () => { } } + it.each(['darwin', 'linux', 'win32'] as const)( + 'keeps explicit background startup hidden through ready/load/fallback on %s', + (platform) => { + vi.useFakeTimers() + vi.stubEnv('ORCA_BACKGROUND_LAUNCH', '1') + const { browserWindowInstance, windowHandlers } = createStartupRevealWindowFixture() + const showInactive = vi.fn() + Object.assign(browserWindowInstance, { showInactive }) + try { + withPlatform(platform, () => { + createMainWindow(createStartupRevealStore(true) as never, { revealOnDidFinishLoad: true }) + const revealAfterLoad = browserWindowInstance.webContents.on.mock.calls.find( + ([event]) => event === 'did-finish-load' + )?.[1] + expect(revealAfterLoad).toBeTypeOf('function') + revealAfterLoad?.() + windowHandlers['ready-to-show']() + vi.advanceTimersByTime(10_000) + expect(browserWindowInstance.show).not.toHaveBeenCalled() + expect(showInactive).not.toHaveBeenCalled() + expect(browserWindowInstance.maximize).not.toHaveBeenCalled() + }) + } finally { + vi.unstubAllEnvs() + } + } + ) + it('ignores duplicate ready-to-show events after startup maximize has already run', () => { const { browserWindowInstance, windowHandlers } = createStartupRevealWindowFixture() diff --git a/src/main/window/focus-existing-window.test.ts b/src/main/window/focus-existing-window.test.ts index babb9a15490..9f5dc522150 100644 --- a/src/main/window/focus-existing-window.test.ts +++ b/src/main/window/focus-existing-window.test.ts @@ -1,5 +1,5 @@ import type { App, BrowserWindow } from 'electron' -import { describe, expect, it, vi } from 'vitest' +import { afterEach, describe, expect, it, vi } from 'vitest' import { focusExistingMainWindow } from './focus-existing-window' type FakeWindowOptions = { @@ -78,7 +78,32 @@ function makeTimer(): { } } +afterEach(() => vi.unstubAllEnvs()) + describe('focusExistingMainWindow', () => { + it.each(['darwin', 'linux', 'win32'] as const)( + 'never restores or activates a background window on %s', + (platform) => { + vi.stubEnv('ORCA_BACKGROUND_LAUNCH', '1') + vi.stubEnv('ORCA_E2E_FOREGROUND', '1') + const app = makeFakeApp() + const window = makeFakeWindow({ minimized: true }) + const timer = makeTimer() + focusExistingMainWindow({ + app, + getWindow: () => window, + openWindow: vi.fn(), + platform, + setTimeout: timer.setTimeout + }) + expect(app.focus).not.toHaveBeenCalled() + for (const call of Object.values(window.calls)) { + expect(call).not.toHaveBeenCalled() + } + expect(timer.scheduledMs()).toEqual([]) + } + ) + it('aggressively foregrounds an existing Windows window on second launch', () => { const app = makeFakeApp() const window = makeFakeWindow() diff --git a/src/main/window/focus-existing-window.ts b/src/main/window/focus-existing-window.ts index 4cadb49a743..903e6a8b321 100644 --- a/src/main/window/focus-existing-window.ts +++ b/src/main/window/focus-existing-window.ts @@ -1,5 +1,9 @@ import type { App, BrowserWindow } from 'electron' -import { isBackgroundLaunch, showWindowWithoutStealingFocus } from './foreground-activation-policy' +import { + isBackgroundLaunch, + isWindowlessLaunch, + showWindowWithoutStealingFocus +} from './foreground-activation-policy' type FocusTimer = (callback: () => void, ms: number) => unknown @@ -34,7 +38,7 @@ function safelyFocusApp(app: Pick): void { } export function safelyRevealWindow(window: BrowserWindow): void { - if (window.isDestroyed()) { + if (window.isDestroyed() || isWindowlessLaunch()) { return } if (window.isMinimized()) { diff --git a/src/main/window/foreground-activation-policy.test.ts b/src/main/window/foreground-activation-policy.test.ts index 0a45f00387e..3b33c9be881 100644 --- a/src/main/window/foreground-activation-policy.test.ts +++ b/src/main/window/foreground-activation-policy.test.ts @@ -32,6 +32,10 @@ describe('isBackgroundLaunch', () => { expect(isBackgroundLaunch({})).toBe(false) }) + it('keeps an explicit background request despite inherited foreground flags', () => { + expect(isBackgroundLaunch({ ORCA_BACKGROUND_LAUNCH: '1', ORCA_E2E_FOREGROUND: '1' })).toBe(true) + }) + it('lets native-focus specs opt back into the foreground', () => { expect(isBackgroundLaunch({ ORCA_E2E_HEADFUL: '1', ORCA_E2E_FOREGROUND: '1' })).toBe(false) expect(isWindowlessLaunch({ ORCA_E2E_HEADLESS: '1', ORCA_E2E_FOREGROUND: '1' })).toBe(false) @@ -39,10 +43,17 @@ describe('isBackgroundLaunch', () => { }) describe('isWindowlessLaunch', () => { - it('is headless-only; a headful run still paints', () => { + it('keeps explicit background launches hidden while headful E2E can paint', () => { expect(isWindowlessLaunch({ ORCA_E2E_HEADLESS: '1' })).toBe(true) expect(isWindowlessLaunch({ ORCA_E2E_HEADLESS: '1', ORCA_E2E_HEADFUL: '1' })).toBe(false) - expect(isWindowlessLaunch({ ORCA_BACKGROUND_LAUNCH: '1' })).toBe(false) + expect(isWindowlessLaunch({ ORCA_BACKGROUND_LAUNCH: '1' })).toBe(true) + expect( + isWindowlessLaunch({ + ORCA_BACKGROUND_LAUNCH: '1', + ORCA_E2E_HEADFUL: '1', + ORCA_E2E_FOREGROUND: '1' + }) + ).toBe(true) }) }) @@ -54,9 +65,16 @@ describe('showWindowWithoutStealingFocus', () => { expect(window.showInactive).not.toHaveBeenCalled() }) - it('shows a background window without activating it', () => { + it('never reveals an explicitly background window', () => { const window = makeWindow() showWindowWithoutStealingFocus(window, { ORCA_BACKGROUND_LAUNCH: '1' }) + expect(window.showInactive).not.toHaveBeenCalled() + expect(window.show).not.toHaveBeenCalled() + }) + + it('still reveals explicitly headful E2E without activation', () => { + const window = makeWindow() + showWindowWithoutStealingFocus(window, { ORCA_E2E_HEADFUL: '1' }) expect(window.showInactive).toHaveBeenCalledOnce() expect(window.show).not.toHaveBeenCalled() }) @@ -83,18 +101,21 @@ describe('applyBackgroundActivationPolicy', () => { } } - it('drops the macOS Dock tile and menu bar for headless runs', () => { - const app = makeApp() - expect( - applyBackgroundActivationPolicy({ - app, - env: { ORCA_E2E_HEADLESS: '1' }, - platform: 'darwin' - }) - ).toBe(true) - expect(app.dock.hide).toHaveBeenCalledOnce() - expect(app.setActivationPolicy).toHaveBeenCalledWith('accessory') - }) + it.each(['ORCA_E2E_HEADLESS', 'ORCA_BACKGROUND_LAUNCH'])( + 'drops the macOS Dock tile and menu bar for %s', + (flag) => { + const app = makeApp() + expect( + applyBackgroundActivationPolicy({ + app, + env: { [flag]: '1' }, + platform: 'darwin' + }) + ).toBe(true) + expect(app.dock.hide).toHaveBeenCalledOnce() + expect(app.setActivationPolicy).toHaveBeenCalledWith('accessory') + } + ) it('leaves a headful or user launch with its normal Dock presence', () => { const headful = makeApp() diff --git a/src/main/window/foreground-activation-policy.ts b/src/main/window/foreground-activation-policy.ts index c2ee6b19e73..5d51f50e487 100644 --- a/src/main/window/foreground-activation-policy.ts +++ b/src/main/window/foreground-activation-policy.ts @@ -5,8 +5,8 @@ import { app as electronApp, type BrowserWindow } from 'electron' * validation). These runs may use the machine, but must never take the OS * foreground away from whatever the developer is doing. * - * ORCA_BACKGROUND_LAUNCH=1 opts a normal launch in; ORCA_E2E_FOREGROUND=1 opts - * back out for the few specs whose subject *is* native focus (IME, key events). + * ORCA_BACKGROUND_LAUNCH=1 keeps automation off screen. Native-focus specs + * can use ORCA_E2E_FOREGROUND=1 only without an explicit background request. */ type ActivationPolicyApp = { @@ -19,19 +19,21 @@ type PolicyEnv = Readonly> /** True when this process must not steal focus, raise windows, or activate the app. */ export function isBackgroundLaunch(env: PolicyEnv = process.env): boolean { + if (env.ORCA_BACKGROUND_LAUNCH === '1') { + return true + } if (env.ORCA_E2E_FOREGROUND === '1') { return false } - return ( - env.ORCA_BACKGROUND_LAUNCH === '1' || - env.ORCA_E2E_HEADLESS === '1' || - env.ORCA_E2E_HEADFUL === '1' - ) + return env.ORCA_E2E_HEADLESS === '1' || env.ORCA_E2E_HEADFUL === '1' } -/** True when no window should reach the screen at all (headless E2E; Playwright drives via CDP). */ +/** True when no window should reach the screen at all (background or headless E2E; Playwright drives via CDP). */ export function isWindowlessLaunch(env: PolicyEnv = process.env): boolean { - return isBackgroundLaunch(env) && env.ORCA_E2E_HEADLESS === '1' && env.ORCA_E2E_HEADFUL !== '1' + return ( + env.ORCA_BACKGROUND_LAUNCH === '1' || + (isBackgroundLaunch(env) && env.ORCA_E2E_HEADLESS === '1' && env.ORCA_E2E_HEADFUL !== '1') + ) } /** @@ -63,7 +65,7 @@ export function applyBackgroundActivationPolicy( /** * Reveal a window without taking the foreground: hidden entirely when windowless, - * `showInactive()` (visible, not raised over the active app) in background launches. + * `showInactive()` for explicitly headful E2E runs. */ export function showWindowWithoutStealingFocus( window: BrowserWindow, diff --git a/tests/AGENTS.md b/tests/AGENTS.md index f445415e7df..26987a87c25 100644 --- a/tests/AGENTS.md +++ b/tests/AGENTS.md @@ -6,16 +6,18 @@ take the foreground — no window raised over the editor, no focus stolen, no Do `src/main/window/foreground-activation-policy.ts` enforces this in the main process. It is on whenever `ORCA_E2E_HEADLESS=1`, `ORCA_E2E_HEADFUL=1`, or `ORCA_BACKGROUND_LAUNCH=1`: -- headless → the window never reaches the screen (Playwright drives it via CDP) -- headful / background → `showInactive()`, no `app.focus({ steal: true })`, no +- headless / explicit background → the window never reaches the screen (Playwright drives it via CDP) +- headful without explicit background → `showInactive()`, no `app.focus({ steal: true })`, no `moveTop()`/always-on-top reinforcement -- macOS headless → `accessory` activation policy, so no Dock tile and no menu-bar takeover +- macOS headless / explicit background → `accessory` activation policy, so no Dock tile and no menu-bar takeover Rules when adding tests or scripts: - Launch through `tests/e2e/helpers/orca-app.ts` (or `orca-restart.ts`) — they already set the env. - A raw `electron.launch()` outside those helpers must pass `ORCA_BACKGROUND_LAUNCH: '1'`. -- Call `showInactive()`, never `show()`, when an `app.evaluate()` block reveals a window. +- Do not reveal windows in explicit background or headless runs. Only an explicitly headful run + may call `showInactive()`; never call `show()` or `bringToFront()` in automated background checks. - Tag a spec `@headful` only when it needs real pixels; it still runs in the background. - `ORCA_E2E_FOREGROUND=1` is the only opt-out, for runs whose subject _is_ native focus (IME and - other OS-level key injection). Add a comment saying why. + other OS-level key injection). Clear `ORCA_BACKGROUND_LAUNCH` for that isolated run and add a + comment saying why; an explicit background request takes precedence. From abdee9ebd370d3e2a7eb968b642df6e625846b33 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:16:08 -0700 Subject: [PATCH 03/17] feat(automations): restore column sorting on the list (#18885) The flat-table redesign in #16532 dropped the sort UI, orphaning AutomationListSortHeader, nextAutomationListSort and the whole AutomationListViewItem layer. Wire them back to the rendered list. Name and Last run become interactive header cells again; the other six columns stay plain text. Sorting now spans local and external rows as one list, so the panel renders per-row components from a single sorted collection instead of two independent sections. Two model fixes fall out of that: - View items key on the host-qualified row key, not the bare automation ID. The old builder predated automation-list-row-identity, so under All hosts two authorities returning the same ID collapsed in the sort tie-break. - sortAutomationListViewItems takes the locale as a parameter instead of reading getIntlLocale(). A hidden global read is invisible to a dependency array, and the list result is memoized. Keyboard traversal and focus recovery now read the sorted order, so arrow navigation matches what is on screen. The dead unified filter is removed in favor of the live row/entry filters the page already used. --- .../automations/AutomationListExternalRow.tsx | 260 +++++++++++ .../AutomationListExternalRows.tsx | 286 +----------- .../automations/AutomationListLocalRow.tsx | 391 +++++++++++++++++ .../automations/AutomationListLocalRows.tsx | 406 +----------------- .../automations/AutomationListSortHeader.tsx | 51 +++ .../AutomationListTableHeader.test.tsx | 45 +- .../automations/AutomationListTableHeader.tsx | 101 +++-- .../automations/AutomationsListPanel.test.tsx | 43 +- .../automations/AutomationsListPanel.tsx | 87 ++-- ...utomationsPage.create-destination.test.tsx | 2 +- ...tionsPage.cross-authority-actions.test.tsx | 9 +- .../AutomationsPage.external-scope.test.tsx | 7 +- .../AutomationsPage.notice-recovery.test.tsx | 2 +- ...AutomationsPage.refresh-selection.test.tsx | 7 +- .../AutomationsPage.run-visibility.test.tsx | 4 +- .../automations/AutomationsPage.test.tsx | 8 +- .../automations/AutomationsPageListPanel.tsx | 8 +- .../automation-list-view-sort.test.ts | 83 ++-- .../automations/automation-list-view.test.ts | 207 ++++----- .../automations/automation-list-view.ts | 90 ++-- .../automations-page-listed-items.ts | 32 ++ .../automations-page-test-harness.tsx | 65 ++- .../use-automations-page-list-state.ts | 22 +- .../use-automations-page-local-state.ts | 9 +- .../pane-agent-identity-inventory.test.ts | 2 +- 25 files changed, 1241 insertions(+), 986 deletions(-) create mode 100644 src/renderer/src/components/automations/AutomationListExternalRow.tsx create mode 100644 src/renderer/src/components/automations/AutomationListLocalRow.tsx create mode 100644 src/renderer/src/components/automations/AutomationListSortHeader.tsx create mode 100644 src/renderer/src/components/automations/automations-page-listed-items.ts diff --git a/src/renderer/src/components/automations/AutomationListExternalRow.tsx b/src/renderer/src/components/automations/AutomationListExternalRow.tsx new file mode 100644 index 00000000000..b26173467ab --- /dev/null +++ b/src/renderer/src/components/automations/AutomationListExternalRow.tsx @@ -0,0 +1,260 @@ +import React from 'react' +import { MoreHorizontal, Pause, Pencil, Play, Trash2 } from 'lucide-react' +import { + ContextMenu, + ContextMenuContent, + ContextMenuItem, + ContextMenuSeparator, + ContextMenuTrigger +} from '@/components/ui/context-menu' +import { + DropdownMenu, + DropdownMenuContent, + DropdownMenuItem, + DropdownMenuSeparator, + DropdownMenuTrigger +} from '@/components/ui/dropdown-menu' +import { Button } from '@/components/ui/button' +import { cn } from '@/lib/utils' +import type { + ExternalAutomationAction, + ExternalAutomationJob, + ExternalAutomationManager +} from '../../../../shared/automations-types' +import type { SshConnectionState } from '../../../../shared/ssh-types' +import type { ExternalAutomationListEntry } from './external-automation-list-entries' +import type { ExternalAutomationScope } from './external-automation-scope-client' +import { + formatExternalDate, + getExternalProviderLabel, + getExternalTargetKindLabel +} from './external-automation-display' +import { getExternalAutomationScheduleDisplay } from './external-automation-schedule-display' +import { getExternalAutomationActionDisabledMessage } from './external-automation-source-availability' +import { AUTOMATIONS_TABLE_GRID_CLASS } from './automations-table-layout' +import { + LIST_TABLE_ROW_CLASS, + LIST_TABLE_ROW_SELECTED_CLASS, + LIST_TABLE_STICKY_ROW_CELL_CLASS +} from '@/lib/list-table-layout' +import { isPortaledRowMenuClick, isRowActivationKey } from '@/lib/list-row-interaction' +import { getExternalAutomationLastRunSnapshot } from './automation-list-last-run' +import { AutomationListLastRunCell } from './AutomationListLastRunCell' +import { AutomationListStatusCell } from './AutomationListStatusCell' +import { translate } from '@/i18n/i18n' + +export type AutomationListExternalRowProps = { + entry: ExternalAutomationListEntry + selectedExternalKey: string | null | undefined + relativeNow: number + sshConnectionStates: ReadonlyMap> + externalActionKey: string | null + onSelect: (entryKey: string) => void + onRequestAction: ( + manager: ExternalAutomationManager, + job: ExternalAutomationJob, + action: ExternalAutomationAction, + scope: ExternalAutomationScope + ) => void + onEdit: ( + manager: ExternalAutomationManager, + job: ExternalAutomationJob, + scope: ExternalAutomationScope + ) => void +} + +export function AutomationListExternalRow({ + entry, + selectedExternalKey, + relativeNow, + sshConnectionStates, + externalActionKey, + onSelect, + onRequestAction, + onEdit +}: AutomationListExternalRowProps): React.JSX.Element { + const providerLabel = getExternalProviderLabel(entry.manager) + const targetKindLabel = getExternalTargetKindLabel(entry.manager) + const isSelected = selectedExternalKey === entry.key + const sshStatus = + entry.manager.target.type === 'ssh' + ? sshConnectionStates.get(entry.manager.target.connectionId)?.status + : undefined + const disabledMessage = getExternalAutomationActionDisabledMessage({ + manager: entry.manager, + providerLabel, + targetKindLabel, + sshStatus, + actionInProgress: externalActionKey !== null + }) + const actionDisabled = disabledMessage !== null + const scheduleLabel = getExternalAutomationScheduleDisplay(entry.manager, entry.job).label + const hostLabel = entry.manager.targetLabel || entry.manager.label || 'Local' + const projectLabel = entry.job.workdir ?? providerLabel + const nextRunLabel = entry.job.enabled + ? formatExternalDate(entry.job.nextRunAt, relativeNow) + : translate('auto.components.automations.AutomationsPage.paused', 'Paused') + const lastRunSnapshot = getExternalAutomationLastRunSnapshot(entry.job) + + return ( + + +
{ + // Why: Radix portals menus out of the row DOM, but React still + // bubbles those clicks here — ignore so menu actions don't open detail. + if (isPortaledRowMenuClick(event)) { + return + } + onSelect(entry.key) + }} + onKeyDown={(event) => { + if (!isRowActivationKey(event)) { + return + } + event.preventDefault() + onSelect(entry.key) + }} + className={cn( + AUTOMATIONS_TABLE_GRID_CLASS, + LIST_TABLE_ROW_CLASS, + isSelected && LIST_TABLE_ROW_SELECTED_CLASS + )} + > + + {entry.job.name} + + + {scheduleLabel} + + + {projectLabel} + + + {hostLabel} + + + {nextRunLabel} + + + + + {providerLabel} + + + + + + + onRequestAction(entry.manager, entry.job, 'run', entry.scope)} + > + + + {disabledMessage ?? + translate('auto.components.automations.AutomationsPage.2faecab10b', 'Run Now')} + + + {entry.manager.provider === 'hermes' ? ( + onEdit(entry.manager, entry.job, entry.scope)} + > + + {translate('auto.components.automations.AutomationsPage.f4612e3f78', 'Edit')} + + ) : null} + + onRequestAction( + entry.manager, + entry.job, + entry.job.enabled ? 'pause' : 'resume', + entry.scope + ) + } + > + {entry.job.enabled ? : } + {entry.job.enabled + ? translate('auto.components.automations.AutomationsPage.b457436d6a', 'Pause') + : translate('auto.components.automations.AutomationsPage.376631ef2b', 'Resume')} + + + onRequestAction(entry.manager, entry.job, 'delete', entry.scope)} + > + + {translate('auto.components.automations.AutomationsPage.15e0bfb13b', 'Delete')} + + + +
+
+ + onRequestAction(entry.manager, entry.job, 'run', entry.scope)} + > + + + {disabledMessage ?? + translate('auto.components.automations.AutomationsPage.2faecab10b', 'Run Now')} + + + {entry.manager.provider === 'hermes' ? ( + onEdit(entry.manager, entry.job, entry.scope)} + > + + {translate('auto.components.automations.AutomationsPage.f4612e3f78', 'Edit')} + + ) : null} + + onRequestAction( + entry.manager, + entry.job, + entry.job.enabled ? 'pause' : 'resume', + entry.scope + ) + } + > + {entry.job.enabled ? : } + {entry.job.enabled + ? translate('auto.components.automations.AutomationsPage.b457436d6a', 'Pause') + : translate('auto.components.automations.AutomationsPage.376631ef2b', 'Resume')} + + + onRequestAction(entry.manager, entry.job, 'delete', entry.scope)} + > + + {translate('auto.components.automations.AutomationsPage.15e0bfb13b', 'Delete')} + + +
+ ) +} diff --git a/src/renderer/src/components/automations/AutomationListExternalRows.tsx b/src/renderer/src/components/automations/AutomationListExternalRows.tsx index 976a93b2433..3ed78fc2a69 100644 --- a/src/renderer/src/components/automations/AutomationListExternalRows.tsx +++ b/src/renderer/src/components/automations/AutomationListExternalRows.tsx @@ -1,285 +1,23 @@ import React from 'react' -import { MoreHorizontal, Pause, Pencil, Play, Trash2 } from 'lucide-react' -import { - ContextMenu, - ContextMenuContent, - ContextMenuItem, - ContextMenuSeparator, - ContextMenuTrigger -} from '@/components/ui/context-menu' -import { - DropdownMenu, - DropdownMenuContent, - DropdownMenuItem, - DropdownMenuSeparator, - DropdownMenuTrigger -} from '@/components/ui/dropdown-menu' -import { Button } from '@/components/ui/button' -import { cn } from '@/lib/utils' -import type { - ExternalAutomationAction, - ExternalAutomationJob, - ExternalAutomationManager -} from '../../../../shared/automations-types' -import type { SshConnectionState } from '../../../../shared/ssh-types' import type { ExternalAutomationListEntry } from './external-automation-list-entries' -import type { ExternalAutomationScope } from './external-automation-scope-client' import { - formatExternalDate, - getExternalProviderLabel, - getExternalTargetKindLabel -} from './external-automation-display' -import { getExternalAutomationScheduleDisplay } from './external-automation-schedule-display' -import { getExternalAutomationActionDisabledMessage } from './external-automation-source-availability' -import { AUTOMATIONS_TABLE_GRID_CLASS } from './automations-table-layout' -import { - LIST_TABLE_ROW_CLASS, - LIST_TABLE_ROW_SELECTED_CLASS, - LIST_TABLE_STICKY_ROW_CELL_CLASS -} from '@/lib/list-table-layout' -import { isPortaledRowMenuClick, isRowActivationKey } from '@/lib/list-row-interaction' -import { getExternalAutomationLastRunSnapshot } from './automation-list-last-run' -import { AutomationListLastRunCell } from './AutomationListLastRunCell' -import { AutomationListStatusCell } from './AutomationListStatusCell' -import { translate } from '@/i18n/i18n' + AutomationListExternalRow, + type AutomationListExternalRowProps +} from './AutomationListExternalRow' + +export type AutomationListExternalRowsProps = Omit & { + entries: readonly ExternalAutomationListEntry[] +} export function AutomationListExternalRows({ entries, - selectedExternalKey, - relativeNow, - sshConnectionStates, - externalActionKey, - onSelect, - onRequestAction, - onEdit -}: { - entries: readonly ExternalAutomationListEntry[] - selectedExternalKey: string | null | undefined - relativeNow: number - sshConnectionStates: ReadonlyMap> - externalActionKey: string | null - onSelect: (entryKey: string) => void - onRequestAction: ( - manager: ExternalAutomationManager, - job: ExternalAutomationJob, - action: ExternalAutomationAction, - scope: ExternalAutomationScope - ) => void - onEdit: ( - manager: ExternalAutomationManager, - job: ExternalAutomationJob, - scope: ExternalAutomationScope - ) => void -}): React.JSX.Element { + ...rowProps +}: AutomationListExternalRowsProps): React.JSX.Element { return ( <> - {entries.map((entry) => { - const providerLabel = getExternalProviderLabel(entry.manager) - const targetKindLabel = getExternalTargetKindLabel(entry.manager) - const isSelected = selectedExternalKey === entry.key - const sshStatus = - entry.manager.target.type === 'ssh' - ? sshConnectionStates.get(entry.manager.target.connectionId)?.status - : undefined - const disabledMessage = getExternalAutomationActionDisabledMessage({ - manager: entry.manager, - providerLabel, - targetKindLabel, - sshStatus, - actionInProgress: externalActionKey !== null - }) - const actionDisabled = disabledMessage !== null - const scheduleLabel = getExternalAutomationScheduleDisplay(entry.manager, entry.job).label - const hostLabel = entry.manager.targetLabel || entry.manager.label || 'Local' - const projectLabel = entry.job.workdir ?? providerLabel - const nextRunLabel = entry.job.enabled - ? formatExternalDate(entry.job.nextRunAt, relativeNow) - : translate('auto.components.automations.AutomationsPage.paused', 'Paused') - const lastRunSnapshot = getExternalAutomationLastRunSnapshot(entry.job) - - return ( - - -
{ - // Why: Radix portals menus out of the row DOM, but React still - // bubbles those clicks here — ignore so menu actions don't open detail. - if (isPortaledRowMenuClick(event)) { - return - } - onSelect(entry.key) - }} - onKeyDown={(event) => { - if (!isRowActivationKey(event)) { - return - } - event.preventDefault() - onSelect(entry.key) - }} - className={cn( - AUTOMATIONS_TABLE_GRID_CLASS, - LIST_TABLE_ROW_CLASS, - isSelected && LIST_TABLE_ROW_SELECTED_CLASS - )} - > - - {entry.job.name} - - - {scheduleLabel} - - - {projectLabel} - - - {hostLabel} - - - {nextRunLabel} - - - - - {providerLabel} - - - - - - - onRequestAction(entry.manager, entry.job, 'run', entry.scope)} - > - - - {disabledMessage ?? - translate( - 'auto.components.automations.AutomationsPage.2faecab10b', - 'Run Now' - )} - - - {entry.manager.provider === 'hermes' ? ( - onEdit(entry.manager, entry.job, entry.scope)} - > - - {translate( - 'auto.components.automations.AutomationsPage.f4612e3f78', - 'Edit' - )} - - ) : null} - - onRequestAction( - entry.manager, - entry.job, - entry.job.enabled ? 'pause' : 'resume', - entry.scope - ) - } - > - {entry.job.enabled ? ( - - ) : ( - - )} - {entry.job.enabled - ? translate( - 'auto.components.automations.AutomationsPage.b457436d6a', - 'Pause' - ) - : translate( - 'auto.components.automations.AutomationsPage.376631ef2b', - 'Resume' - )} - - - - onRequestAction(entry.manager, entry.job, 'delete', entry.scope) - } - > - - {translate( - 'auto.components.automations.AutomationsPage.15e0bfb13b', - 'Delete' - )} - - - -
-
- - onRequestAction(entry.manager, entry.job, 'run', entry.scope)} - > - - - {disabledMessage ?? - translate('auto.components.automations.AutomationsPage.2faecab10b', 'Run Now')} - - - {entry.manager.provider === 'hermes' ? ( - onEdit(entry.manager, entry.job, entry.scope)} - > - - {translate('auto.components.automations.AutomationsPage.f4612e3f78', 'Edit')} - - ) : null} - - onRequestAction( - entry.manager, - entry.job, - entry.job.enabled ? 'pause' : 'resume', - entry.scope - ) - } - > - {entry.job.enabled ? : } - {entry.job.enabled - ? translate('auto.components.automations.AutomationsPage.b457436d6a', 'Pause') - : translate('auto.components.automations.AutomationsPage.376631ef2b', 'Resume')} - - - onRequestAction(entry.manager, entry.job, 'delete', entry.scope)} - > - - {translate('auto.components.automations.AutomationsPage.15e0bfb13b', 'Delete')} - - -
- ) - })} + {entries.map((entry) => ( + + ))} ) } diff --git a/src/renderer/src/components/automations/AutomationListLocalRow.tsx b/src/renderer/src/components/automations/AutomationListLocalRow.tsx new file mode 100644 index 00000000000..a9c1a5bc8b6 --- /dev/null +++ b/src/renderer/src/components/automations/AutomationListLocalRow.tsx @@ -0,0 +1,391 @@ +import React from 'react' +import { MoreHorizontal, Pause, Pencil, Play, Trash2 } from 'lucide-react' +import { + ContextMenu, + ContextMenuContent, + ContextMenuItem, + ContextMenuSeparator, + ContextMenuTrigger +} from '@/components/ui/context-menu' +import { + DropdownMenu, + DropdownMenuContent, + DropdownMenuItem, + DropdownMenuSeparator, + DropdownMenuTrigger +} from '@/components/ui/dropdown-menu' +import { Button } from '@/components/ui/button' +import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' +import { AgentIcon } from '@/lib/agent-catalog' +import { cn } from '@/lib/utils' +import type { AutomationRun } from '../../../../shared/automations-types' +import { getAutomationRunRepoId } from '../../../../shared/automation-run-identity' +import { formatUiAutomationSchedule } from './automation-schedule-label' +import { + getExecutionHostLabel, + getLocalExecutionHostLabel, + getRepoExecutionHostId +} from '../../../../shared/execution-host' +import type { SshConnectionState } from '../../../../shared/ssh-types' +import type { ProjectHostSetup } from '../../../../shared/project-types' +import type { Repo } from '../../../../shared/repo-types' +import type { Worktree } from '../../../../shared/worktree/types' +import type { RuntimeStatus } from '../../../../shared/runtime-types' +import type { TaskSourceHostAvailability } from '../task-source-context-summary' +import type { AutomationRowAction } from './automation-captured-owner' +import type { AutomationHostTarget } from './automation-host-client' +import { + getAutomationRowLastRunSnapshot, + getLocalAutomationLastRunSnapshot +} from './automation-list-last-run' +import { AutomationListLastRunCell } from './AutomationListLastRunCell' +import { formatAutomationDateTimeWithRelative } from './automation-page-parts' +import { getAutomationTargetAvailability } from './automation-target-availability' +import { getAgentLabel } from './automation-draft-model' +import type { AutomationListRow } from './automation-list-row-identity' +import { + formatAutomationCost, + formatAutomationTokens, + type AutomationUsageSummary +} from './automation-usage-model' +import { AUTOMATIONS_TABLE_GRID_CLASS } from './automations-table-layout' +import { + LIST_TABLE_ROW_CLASS, + LIST_TABLE_ROW_SELECTED_CLASS, + LIST_TABLE_STICKY_ROW_CELL_CLASS +} from '@/lib/list-table-layout' +import { isPortaledRowMenuClick, isRowActivationKey } from '@/lib/list-row-interaction' +import { AutomationListStatusCell } from './AutomationListStatusCell' +import { translate } from '@/i18n/i18n' + +export type AutomationListLocalRowProps = { + row: AutomationListRow + selectedRowKey: string | null | undefined + isSelectedLocal: boolean + lastRunByAutomationId: ReadonlyMap + relativeNow: number + repoMap: ReadonlyMap + worktreeMap: ReadonlyMap + repoForRow?: (row: AutomationListRow) => Repo | undefined + worktreeForRow?: (row: AutomationListRow, repo: Repo | undefined) => Worktree | undefined + projectHostSetups: readonly ProjectHostSetup[] + sshConnectionStates: ReadonlyMap> + runtimeStatusByEnvironmentId: ReadonlyMap< + string, + { status: RuntimeStatus | null; checkedAt: number } + > + hostTargetFor: (row: AutomationListRow) => AutomationHostTarget | null + automationSourceHostAvailabilityByRowKey: ReadonlyMap + hostLabelById?: ReadonlyMap + isActionEnabled?: (row: AutomationListRow, action: AutomationRowAction) => boolean + onSelect: (rowKey: string) => void + onRunNow: (row: AutomationListRow) => void + onEdit: (row: AutomationListRow) => void + onToggle: (row: AutomationListRow) => void + onDelete: (row: AutomationListRow) => void +} + +const EMPTY_HOST_LABELS: ReadonlyMap = new Map() + +function automationUsageText(summary: AutomationUsageSummary | undefined): string { + if (!summary || summary.unavailableRuns > 0) { + return summary?.knownRuns + ? usageAmountText(summary) + : translate( + 'auto.components.automations.AutomationsPage.usageUnavailable', + 'Usage unavailable' + ) + } + return summary.knownRuns > 0 + ? usageAmountText(summary) + : translate('auto.components.automations.AutomationsPage.noRunUsageYet', 'No run usage yet') +} + +function usageAmountText(summary: AutomationUsageSummary): string { + return translate( + 'auto.components.automations.AutomationsPage.runUsageSummary', + '{{cost}} est. · {{tokens}} tokens', + { + cost: formatAutomationCost(summary.estimatedCostUsd), + tokens: formatAutomationTokens(summary.totalTokens) + } + ) +} + +export function AutomationListLocalRow({ + row, + selectedRowKey, + isSelectedLocal, + lastRunByAutomationId, + relativeNow, + repoMap, + worktreeMap, + repoForRow, + worktreeForRow, + projectHostSetups, + sshConnectionStates, + runtimeStatusByEnvironmentId, + hostTargetFor, + automationSourceHostAvailabilityByRowKey, + hostLabelById = EMPTY_HOST_LABELS, + isActionEnabled, + onSelect, + onRunNow, + onEdit, + onToggle, + onDelete +}: AutomationListLocalRowProps): React.JSX.Element { + const allows = (row: AutomationListRow, action: AutomationRowAction): boolean => + isActionEnabled?.(row, action) ?? true + const { automation } = row + const automationRepo = repoForRow?.(row) ?? repoMap.get(getAutomationRunRepoId(automation)) + const automationWorktree = automation.workspaceId + ? (worktreeForRow?.(row, automationRepo) ?? worktreeMap.get(automation.workspaceId)) + : null + const automationRunAvailability = getAutomationTargetAvailability({ + automation, + repo: automationRepo, + workspace: automationWorktree, + projectHostSetups, + sshConnectionStates, + runtimeStatusByEnvironmentId, + automationHostTarget: hostTargetFor(row), + sourceHostAvailability: automationSourceHostAvailabilityByRowKey.get(row.key) + }) + const projectLabel = + automationRepo?.displayName ?? + translate('auto.components.automations.AutomationsPage.13118faadf', 'Unknown project') + const scheduleLabel = formatUiAutomationSchedule(automation.rrule) + const nextRunLabel = automation.enabled + ? formatAutomationDateTimeWithRelative(automation.nextRunAt, relativeNow) + : translate('auto.components.automations.enablement.paused', 'Paused') + const isSelected = isSelectedLocal && selectedRowKey === row.key + const agentLabel = getAgentLabel(automation.agentId) + const hostId = + automation.runContext?.hostId ?? + (automationRepo ? getRepoExecutionHostId(automationRepo) : null) + const hostLabel = + row.hostLabel || + (hostId + ? (hostLabelById.get(hostId) ?? getExecutionHostLabel(hostId)) + : getLocalExecutionHostLabel()) + const agentTooltipLabel = `${agentLabel} · ${hostLabel} · ${automationUsageText(row.usageSummary ?? undefined)}` + const canRunNow = automationRunAvailability.canRunNow && allows(row, 'run') + const lastRun = lastRunByAutomationId.get(automation.id) + // Without a fetched run, the row's projected summary carries the newest + // retained run's status — the list never downloads run history for this. + const lastRunSnapshot = lastRun + ? getLocalAutomationLastRunSnapshot(automation, lastRun) + : getAutomationRowLastRunSnapshot(row) + + const actionItems = ( + <> + onRunNow(row)} + /> + } + label={translate('auto.components.automations.AutomationsPage.f4612e3f78', 'Edit')} + onSelect={() => onEdit(row)} + /> + : } + label={ + automation.enabled + ? translate('auto.components.automations.AutomationsPage.b457436d6a', 'Pause') + : translate('auto.components.automations.AutomationsPage.376631ef2b', 'Resume') + } + onSelect={() => onToggle(row)} + /> + + } + label={translate('auto.components.automations.AutomationsPage.15e0bfb13b', 'Delete')} + variant="destructive" + onSelect={() => onDelete(row)} + /> + + ) + + return ( + + +
{ + // Why: Radix portals menus out of the row DOM, but React still + // bubbles those clicks here — ignore so menu actions don't open detail. + if (isPortaledRowMenuClick(event)) { + return + } + onSelect(row.key) + }} + onKeyDown={(event) => { + if (!isRowActivationKey(event)) { + return + } + event.preventDefault() + onSelect(row.key) + }} + className={cn( + AUTOMATIONS_TABLE_GRID_CLASS, + LIST_TABLE_ROW_CLASS, + isSelected && LIST_TABLE_ROW_SELECTED_CLASS + )} + > + + {automation.name} + + + {scheduleLabel} + + + {projectLabel} + + + {hostLabel} + + + {nextRunLabel} + + + + + + + + + + + {agentTooltipLabel} + + + + + + + + { + if (canRunNow) { + onRunNow(row) + } + }} + > + + + {automationRunAvailability.canRunNow + ? translate('auto.components.automations.AutomationsPage.2faecab10b', 'Run Now') + : automationRunAvailability.message} + + + onEdit(row)}> + + {translate('auto.components.automations.AutomationsPage.f4612e3f78', 'Edit')} + + onToggle(row)}> + {automation.enabled ? ( + + ) : ( + + )} + {automation.enabled + ? translate('auto.components.automations.AutomationsPage.b457436d6a', 'Pause') + : translate('auto.components.automations.AutomationsPage.376631ef2b', 'Resume')} + + + onDelete(row)} + > + + {translate('auto.components.automations.AutomationsPage.15e0bfb13b', 'Delete')} + + + +
+
+ {actionItems} +
+ ) +} + +function MenuRunItem({ + disabled, + label, + onSelect +}: { + disabled: boolean + label: string + onSelect: () => void +}): React.JSX.Element { + return ( + { + if (disabled) { + event.preventDefault() + return + } + onSelect() + }} + > + + {label} + + ) +} + +function MenuItem({ + disabled, + icon, + label, + onSelect, + variant +}: { + disabled?: boolean + icon: React.ReactNode + label: string + onSelect: () => void + variant?: 'destructive' +}): React.JSX.Element { + return ( + + {icon} + {label} + + ) +} + +function MenuSeparator(): React.JSX.Element { + return +} diff --git a/src/renderer/src/components/automations/AutomationListLocalRows.tsx b/src/renderer/src/components/automations/AutomationListLocalRows.tsx index 292eb545b4d..3fa02cc1884 100644 --- a/src/renderer/src/components/automations/AutomationListLocalRows.tsx +++ b/src/renderer/src/components/automations/AutomationListLocalRows.tsx @@ -1,414 +1,20 @@ import React from 'react' -import { MoreHorizontal, Pause, Pencil, Play, Trash2 } from 'lucide-react' -import { - ContextMenu, - ContextMenuContent, - ContextMenuItem, - ContextMenuSeparator, - ContextMenuTrigger -} from '@/components/ui/context-menu' -import { - DropdownMenu, - DropdownMenuContent, - DropdownMenuItem, - DropdownMenuSeparator, - DropdownMenuTrigger -} from '@/components/ui/dropdown-menu' -import { Button } from '@/components/ui/button' -import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' -import { AgentIcon } from '@/lib/agent-catalog' -import { cn } from '@/lib/utils' -import type { AutomationRun } from '../../../../shared/automations-types' -import { getAutomationRunRepoId } from '../../../../shared/automation-run-identity' -import { formatUiAutomationSchedule } from './automation-schedule-label' -import { - getExecutionHostLabel, - getLocalExecutionHostLabel, - getRepoExecutionHostId -} from '../../../../shared/execution-host' -import type { SshConnectionState } from '../../../../shared/ssh-types' -import type { ProjectHostSetup } from '../../../../shared/project-types' -import type { Repo } from '../../../../shared/repo-types' -import type { Worktree } from '../../../../shared/worktree/types' -import type { RuntimeStatus } from '../../../../shared/runtime-types' -import type { TaskSourceHostAvailability } from '../task-source-context-summary' -import type { AutomationRowAction } from './automation-captured-owner' -import type { AutomationHostTarget } from './automation-host-client' -import { - getAutomationRowLastRunSnapshot, - getLocalAutomationLastRunSnapshot -} from './automation-list-last-run' -import { AutomationListLastRunCell } from './AutomationListLastRunCell' -import { formatAutomationDateTimeWithRelative } from './automation-page-parts' -import { getAutomationTargetAvailability } from './automation-target-availability' -import { getAgentLabel } from './automation-draft-model' import type { AutomationListRow } from './automation-list-row-identity' -import { - formatAutomationCost, - formatAutomationTokens, - type AutomationUsageSummary -} from './automation-usage-model' -import { AUTOMATIONS_TABLE_GRID_CLASS } from './automations-table-layout' -import { - LIST_TABLE_ROW_CLASS, - LIST_TABLE_ROW_SELECTED_CLASS, - LIST_TABLE_STICKY_ROW_CELL_CLASS -} from '@/lib/list-table-layout' -import { isPortaledRowMenuClick, isRowActivationKey } from '@/lib/list-row-interaction' -import { AutomationListStatusCell } from './AutomationListStatusCell' -import { translate } from '@/i18n/i18n' +import { AutomationListLocalRow, type AutomationListLocalRowProps } from './AutomationListLocalRow' -export type AutomationListLocalRowsProps = { +export type AutomationListLocalRowsProps = Omit & { rows: readonly AutomationListRow[] - selectedRowKey: string | null | undefined - isSelectedLocal: boolean - lastRunByAutomationId: ReadonlyMap - relativeNow: number - repoMap: ReadonlyMap - worktreeMap: ReadonlyMap - repoForRow?: (row: AutomationListRow) => Repo | undefined - worktreeForRow?: (row: AutomationListRow, repo: Repo | undefined) => Worktree | undefined - projectHostSetups: readonly ProjectHostSetup[] - sshConnectionStates: ReadonlyMap> - runtimeStatusByEnvironmentId: ReadonlyMap< - string, - { status: RuntimeStatus | null; checkedAt: number } - > - hostTargetFor: (row: AutomationListRow) => AutomationHostTarget | null - automationSourceHostAvailabilityByRowKey: ReadonlyMap - hostLabelById?: ReadonlyMap - isActionEnabled?: (row: AutomationListRow, action: AutomationRowAction) => boolean - onSelect: (rowKey: string) => void - onRunNow: (row: AutomationListRow) => void - onEdit: (row: AutomationListRow) => void - onToggle: (row: AutomationListRow) => void - onDelete: (row: AutomationListRow) => void -} - -const EMPTY_HOST_LABELS: ReadonlyMap = new Map() - -function automationUsageText(summary: AutomationUsageSummary | undefined): string { - if (!summary || summary.unavailableRuns > 0) { - return summary?.knownRuns - ? usageAmountText(summary) - : translate( - 'auto.components.automations.AutomationsPage.usageUnavailable', - 'Usage unavailable' - ) - } - return summary.knownRuns > 0 - ? usageAmountText(summary) - : translate('auto.components.automations.AutomationsPage.noRunUsageYet', 'No run usage yet') -} - -function usageAmountText(summary: AutomationUsageSummary): string { - return translate( - 'auto.components.automations.AutomationsPage.runUsageSummary', - '{{cost}} est. · {{tokens}} tokens', - { - cost: formatAutomationCost(summary.estimatedCostUsd), - tokens: formatAutomationTokens(summary.totalTokens) - } - ) } export function AutomationListLocalRows({ rows, - selectedRowKey, - isSelectedLocal, - lastRunByAutomationId, - relativeNow, - repoMap, - worktreeMap, - repoForRow, - worktreeForRow, - projectHostSetups, - sshConnectionStates, - runtimeStatusByEnvironmentId, - hostTargetFor, - automationSourceHostAvailabilityByRowKey, - hostLabelById = EMPTY_HOST_LABELS, - isActionEnabled, - onSelect, - onRunNow, - onEdit, - onToggle, - onDelete + ...rowProps }: AutomationListLocalRowsProps): React.JSX.Element { - const allows = (row: AutomationListRow, action: AutomationRowAction): boolean => - isActionEnabled?.(row, action) ?? true return ( <> - {rows.map((row) => { - const { automation } = row - const automationRepo = repoForRow?.(row) ?? repoMap.get(getAutomationRunRepoId(automation)) - const automationWorktree = automation.workspaceId - ? (worktreeForRow?.(row, automationRepo) ?? worktreeMap.get(automation.workspaceId)) - : null - const automationRunAvailability = getAutomationTargetAvailability({ - automation, - repo: automationRepo, - workspace: automationWorktree, - projectHostSetups, - sshConnectionStates, - runtimeStatusByEnvironmentId, - automationHostTarget: hostTargetFor(row), - sourceHostAvailability: automationSourceHostAvailabilityByRowKey.get(row.key) - }) - const projectLabel = - automationRepo?.displayName ?? - translate('auto.components.automations.AutomationsPage.13118faadf', 'Unknown project') - const scheduleLabel = formatUiAutomationSchedule(automation.rrule) - const nextRunLabel = automation.enabled - ? formatAutomationDateTimeWithRelative(automation.nextRunAt, relativeNow) - : translate('auto.components.automations.enablement.paused', 'Paused') - const isSelected = isSelectedLocal && selectedRowKey === row.key - const agentLabel = getAgentLabel(automation.agentId) - const hostId = - automation.runContext?.hostId ?? - (automationRepo ? getRepoExecutionHostId(automationRepo) : null) - const hostLabel = - row.hostLabel || - (hostId - ? (hostLabelById.get(hostId) ?? getExecutionHostLabel(hostId)) - : getLocalExecutionHostLabel()) - const agentTooltipLabel = `${agentLabel} · ${hostLabel} · ${automationUsageText(row.usageSummary ?? undefined)}` - const canRunNow = automationRunAvailability.canRunNow && allows(row, 'run') - const lastRun = lastRunByAutomationId.get(automation.id) - // Without a fetched run, the row's projected summary carries the newest - // retained run's status — the list never downloads run history for this. - const lastRunSnapshot = lastRun - ? getLocalAutomationLastRunSnapshot(automation, lastRun) - : getAutomationRowLastRunSnapshot(row) - - const actionItems = ( - <> - onRunNow(row)} - /> - } - label={translate('auto.components.automations.AutomationsPage.f4612e3f78', 'Edit')} - onSelect={() => onEdit(row)} - /> - : - } - label={ - automation.enabled - ? translate('auto.components.automations.AutomationsPage.b457436d6a', 'Pause') - : translate('auto.components.automations.AutomationsPage.376631ef2b', 'Resume') - } - onSelect={() => onToggle(row)} - /> - - } - label={translate('auto.components.automations.AutomationsPage.15e0bfb13b', 'Delete')} - variant="destructive" - onSelect={() => onDelete(row)} - /> - - ) - - return ( - - -
{ - // Why: Radix portals menus out of the row DOM, but React still - // bubbles those clicks here — ignore so menu actions don't open detail. - if (isPortaledRowMenuClick(event)) { - return - } - onSelect(row.key) - }} - onKeyDown={(event) => { - if (!isRowActivationKey(event)) { - return - } - event.preventDefault() - onSelect(row.key) - }} - className={cn( - AUTOMATIONS_TABLE_GRID_CLASS, - LIST_TABLE_ROW_CLASS, - isSelected && LIST_TABLE_ROW_SELECTED_CLASS - )} - > - - {automation.name} - - - {scheduleLabel} - - - {projectLabel} - - - {hostLabel} - - - {nextRunLabel} - - - - - - - - - - - {agentTooltipLabel} - - - - - - - - { - if (canRunNow) { - onRunNow(row) - } - }} - > - - - {automationRunAvailability.canRunNow - ? translate( - 'auto.components.automations.AutomationsPage.2faecab10b', - 'Run Now' - ) - : automationRunAvailability.message} - - - onEdit(row)}> - - {translate('auto.components.automations.AutomationsPage.f4612e3f78', 'Edit')} - - onToggle(row)} - > - {automation.enabled ? ( - - ) : ( - - )} - {automation.enabled - ? translate( - 'auto.components.automations.AutomationsPage.b457436d6a', - 'Pause' - ) - : translate( - 'auto.components.automations.AutomationsPage.376631ef2b', - 'Resume' - )} - - - onDelete(row)} - > - - {translate( - 'auto.components.automations.AutomationsPage.15e0bfb13b', - 'Delete' - )} - - - -
-
- {actionItems} -
- ) - })} + {rows.map((row) => ( + + ))} ) } - -function MenuRunItem({ - disabled, - label, - onSelect -}: { - disabled: boolean - label: string - onSelect: () => void -}): React.JSX.Element { - return ( - { - if (disabled) { - event.preventDefault() - return - } - onSelect() - }} - > - - {label} - - ) -} - -function MenuItem({ - disabled, - icon, - label, - onSelect, - variant -}: { - disabled?: boolean - icon: React.ReactNode - label: string - onSelect: () => void - variant?: 'destructive' -}): React.JSX.Element { - return ( - - {icon} - {label} - - ) -} - -function MenuSeparator(): React.JSX.Element { - return -} diff --git a/src/renderer/src/components/automations/AutomationListSortHeader.tsx b/src/renderer/src/components/automations/AutomationListSortHeader.tsx new file mode 100644 index 00000000000..2c24a344328 --- /dev/null +++ b/src/renderer/src/components/automations/AutomationListSortHeader.tsx @@ -0,0 +1,51 @@ +import React from 'react' +import { ArrowDown, ArrowUp } from 'lucide-react' +import { cn } from '@/lib/utils' +import { translate } from '@/i18n/i18n' +import type { AutomationListSort, AutomationListSortField } from './automation-list-view' + +export function AutomationListSortHeader({ + field, + label, + sort, + onSort +}: { + field: AutomationListSortField + label: string + sort: AutomationListSort | null + onSort: (field: AutomationListSortField) => void +}): React.JSX.Element { + const active = sort?.field === field + const direction = active ? sort.direction : null + // Why: one interpolated key per direction — word order and punctuation around + // the column name differ per language. + const sortedLabel = + direction === 'asc' + ? translate( + 'auto.components.automations.AutomationListSortHeader.sortedAscending', + '{{value0}}, sorted ascending', + { value0: label } + ) + : direction === 'desc' + ? translate( + 'auto.components.automations.AutomationListSortHeader.sortedDescending', + '{{value0}}, sorted descending', + { value0: label } + ) + : null + return ( + + ) +} diff --git a/src/renderer/src/components/automations/AutomationListTableHeader.test.tsx b/src/renderer/src/components/automations/AutomationListTableHeader.test.tsx index 5c5bbe8e568..638e96a23bc 100644 --- a/src/renderer/src/components/automations/AutomationListTableHeader.test.tsx +++ b/src/renderer/src/components/automations/AutomationListTableHeader.test.tsx @@ -1,7 +1,8 @@ // @vitest-environment happy-dom import { cleanup, render, screen } from '@testing-library/react' -import { afterEach, describe, expect, it } from 'vitest' +import { afterEach, describe, expect, it, vi } from 'vitest' +import userEvent from '@testing-library/user-event' import { AutomationListTableHeader } from './AutomationListTableHeader' import { LIST_TABLE_HEADER_CLASS, @@ -43,3 +44,45 @@ describe('AutomationListTableHeader', () => { expect(nameCell.className).toBe(LIST_TABLE_STICKY_HEADER_CELL_CLASS) }) }) + +describe('AutomationListTableHeader sorting', () => { + afterEach(cleanup) + + it('exposes only the orderable columns as buttons', () => { + render( {}} />) + + expect(screen.getAllByRole('button').map((button) => button.textContent)).toEqual([ + 'Name', + 'Last run' + ]) + }) + + it('reports the sorted column and direction in the accessible name', () => { + const { rerender } = render( + {}} /> + ) + expect(screen.getByRole('button', { name: 'Name, sorted ascending' })).toBeDefined() + expect(screen.getByRole('button', { name: 'Last run' })).toBeDefined() + + rerender( + {}} /> + ) + expect(screen.getByRole('button', { name: 'Last run, sorted descending' })).toBeDefined() + expect(screen.getByRole('button', { name: 'Name' })).toBeDefined() + }) + + it('requests a sort for the clicked column', async () => { + const onSort = vi.fn() + render() + + await userEvent.click(screen.getByRole('button', { name: 'Last run' })) + + expect(onSort.mock.calls).toEqual([['lastRun']]) + }) + + it('stays non-interactive when the list cannot be sorted', () => { + render() + + expect(screen.queryAllByRole('button')).toEqual([]) + }) +}) diff --git a/src/renderer/src/components/automations/AutomationListTableHeader.tsx b/src/renderer/src/components/automations/AutomationListTableHeader.tsx index dcbd107fcbc..605baf8a945 100644 --- a/src/renderer/src/components/automations/AutomationListTableHeader.tsx +++ b/src/renderer/src/components/automations/AutomationListTableHeader.tsx @@ -5,34 +5,85 @@ import { LIST_TABLE_HEADER_CLASS, LIST_TABLE_STICKY_HEADER_CELL_CLASS } from '@/lib/list-table-layout' +import { AutomationListSortHeader } from './AutomationListSortHeader' +import type { AutomationListSort, AutomationListSortField } from './automation-list-view' -export function AutomationListTableHeader(): React.JSX.Element { - const labels = [ - ['auto.components.automations.AutomationsPage.tableName', 'Name'], - ['auto.components.automations.AutomationDetail.18763ded26', 'Schedule'], - ['auto.components.automations.AutomationsPage.tableProject', 'Project'], - ['auto.components.automations.AutomationsPage.tableHost', 'Host'], - ['auto.components.automations.AutomationDetail.578ff46987', 'Next run'], - ['auto.components.automations.AutomationsPage.tableLastRun', 'Last run'], - ['auto.components.automations.AutomationsPage.tableStatus', 'Status'], - ['auto.components.automations.AutomationDetail.2df8970cd5', 'Agent'] - ] as const +type HeaderColumn = { + key: string + fallback: string + /** Absent for columns the list cannot order by. */ + sortField?: AutomationListSortField +} + +const COLUMNS: readonly HeaderColumn[] = [ + { + key: 'auto.components.automations.AutomationsPage.tableName', + fallback: 'Name', + sortField: 'name' + }, + { + key: 'auto.components.automations.AutomationDetail.18763ded26', + fallback: 'Schedule' + }, + { + key: 'auto.components.automations.AutomationsPage.tableProject', + fallback: 'Project' + }, + { + key: 'auto.components.automations.AutomationsPage.tableHost', + fallback: 'Host' + }, + { + key: 'auto.components.automations.AutomationDetail.578ff46987', + fallback: 'Next run' + }, + { + key: 'auto.components.automations.AutomationsPage.tableLastRun', + fallback: 'Last run', + sortField: 'lastRun' + }, + { + key: 'auto.components.automations.AutomationsPage.tableStatus', + fallback: 'Status' + }, + { + key: 'auto.components.automations.AutomationDetail.2df8970cd5', + fallback: 'Agent' + } +] + +export function AutomationListTableHeader({ + sort = null, + onSort +}: { + sort?: AutomationListSort | null + onSort?: (field: AutomationListSortField) => void +} = {}): React.JSX.Element { return (
- {labels.map(([key, fallback], index) => ( - - {translate(key, fallback)} - - ))} + {COLUMNS.map((column, index) => { + const label = translate(column.key, column.fallback) + const className = + index === 0 + ? LIST_TABLE_STICKY_HEADER_CELL_CLASS + : index === COLUMNS.length - 1 + ? 'text-center' + : undefined + return ( + + {column.sortField && onSort ? ( + + ) : ( + label + )} + + ) + })} {translate('auto.components.automations.AutomationsPage.tableActions', 'Actions')} diff --git a/src/renderer/src/components/automations/AutomationsListPanel.test.tsx b/src/renderer/src/components/automations/AutomationsListPanel.test.tsx index 2f772d2a7ed..f2362b83d0f 100644 --- a/src/renderer/src/components/automations/AutomationsListPanel.test.tsx +++ b/src/renderer/src/components/automations/AutomationsListPanel.test.tsx @@ -11,7 +11,12 @@ import { createRoot, type Root } from 'react-dom/client' import { afterEach, beforeEach, describe, expect, it } from 'vitest' import { TooltipProvider } from '@/components/ui/tooltip' import { AutomationsListPanel } from './AutomationsListPanel' -import { EMPTY_AUTOMATION_LIST_FILTER } from './automation-list-view' +import { + buildAutomationListViewItems, + EMPTY_AUTOMATION_LIST_FILTER, + type AutomationListSort, + type AutomationListSortField +} from './automation-list-view' import type { AutomationHostCatalogView } from './use-automation-host-catalog' import { makeAutomation, @@ -49,7 +54,13 @@ const HOST_CATALOG = { status: 'all', announceFallback: false }, - rows: { rows: [], automations: [], capturedOwners: new Map(), groups: [], answered: true }, + rows: { + rows: [], + automations: [], + capturedOwners: new Map(), + groups: [], + answered: true + }, loadCounts: { failedHostCount: 0, totalHostCount: 1 }, selectHost: () => undefined, recover: () => undefined, @@ -70,6 +81,8 @@ function renderPanel( selectExternalKey?: (key: string | null) => void externalEntries?: readonly ExternalAutomationListEntry[] setActivePaneTab?: (tab: AutomationPaneTab) => void + listSort?: AutomationListSort | null + onListSortChange?: (field: AutomationListSortField) => void } = {} ): void { const externalEntries = options.externalEntries ?? [] @@ -95,8 +108,12 @@ function renderPanel( externalManagersUncheckedNotice={uncheckedNotice} onSelectHost={() => undefined} onRecoverHost={() => undefined} - filteredRows={rows} - filteredExternalAutomationEntries={externalEntries} + sortedListItems={buildAutomationListViewItems({ + rows, + externalEntries + })} + listSort={options.listSort ?? null} + onListSortChange={options.onListSortChange ?? (() => undefined)} selectedRowKey={options.selectedRowKey ?? null} selectedExternalKey={options.selectedExternalKey ?? null} relativeNow={0} @@ -221,7 +238,11 @@ describe('AutomationsListPanel enter key navigation', () => { const input = searchField() expect(input).not.toBeNull() - const enter = new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }) + const enter = new KeyboardEvent('keydown', { + key: 'Enter', + bubbles: true, + cancelable: true + }) input?.dispatchEvent(enter) expect(enter.defaultPrevented).toBe(true) @@ -252,7 +273,11 @@ describe('AutomationsListPanel enter key navigation', () => { const input = searchField() expect(input).not.toBeNull() - const enter = new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }) + const enter = new KeyboardEvent('keydown', { + key: 'Enter', + bubbles: true, + cancelable: true + }) input?.dispatchEvent(enter) expect(enter.defaultPrevented).toBe(true) @@ -272,7 +297,11 @@ describe('AutomationsListPanel enter key navigation', () => { const input = searchField() expect(input).not.toBeNull() - const enter = new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }) + const enter = new KeyboardEvent('keydown', { + key: 'Enter', + bubbles: true, + cancelable: true + }) input?.dispatchEvent(enter) expect(detailOpened).toBe(false) diff --git a/src/renderer/src/components/automations/AutomationsListPanel.tsx b/src/renderer/src/components/automations/AutomationsListPanel.tsx index 5943096756a..783eee57da6 100644 --- a/src/renderer/src/components/automations/AutomationsListPanel.tsx +++ b/src/renderer/src/components/automations/AutomationsListPanel.tsx @@ -23,13 +23,19 @@ import { import type { AutomationListRow } from './automation-list-row-identity' import type { AutomationPaneTab } from './automation-page-state' import { AutomationListFilterPills } from './AutomationListFilterMenu' -import { isAutomationListFilterActive, type AutomationListFilter } from './automation-list-view' +import { + isAutomationListFilterActive, + type AutomationListFilter, + type AutomationListSort, + type AutomationListSortField, + type AutomationListViewItem +} from './automation-list-view' import { automationHostFilterStableKey } from '../../../../shared/automation-host-filter' import type { AutomationTemplate } from './automation-templates' import type { ExternalAutomationListEntry } from './external-automation-list-entries' import type { ExternalAutomationScope } from './external-automation-scope-client' -import { AutomationListLocalRows } from './AutomationListLocalRows' -import { AutomationListExternalRows } from './AutomationListExternalRows' +import { AutomationListLocalRow } from './AutomationListLocalRow' +import { AutomationListExternalRow } from './AutomationListExternalRow' import { AutomationHostFilterNotice, AutomationHostLoadSummary } from './AutomationHostFilterNotice' import { AutomationListEmptyView } from './AutomationListEmptyView' import { resolveAutomationListEmptyState } from './automation-list-empty-state' @@ -63,8 +69,10 @@ type AutomationsListPanelProps = { action: AutomationHostRecoveryAction, entry?: AutomationHostCatalogEntry | null ) => void - filteredRows: readonly AutomationListRow[] - filteredExternalAutomationEntries: readonly ExternalAutomationListEntry[] + /** Both collections as one list in render order; the sort spans local and external rows. */ + sortedListItems: readonly AutomationListViewItem[] + listSort: AutomationListSort | null + onListSortChange: (field: AutomationListSortField) => void selectedRowKey: string | null selectedExternalKey: string | null selectedExternal?: ExternalAutomationListEntry | null @@ -124,8 +132,9 @@ export function AutomationsListPanel(props: AutomationsListPanelProps): React.JS externalManagersUncheckedNotice, onSelectHost, onRecoverHost, - filteredRows, - filteredExternalAutomationEntries, + sortedListItems, + listSort, + onListSortChange, selectedRowKey, selectedExternalKey, relativeNow, @@ -161,18 +170,20 @@ export function AutomationsListPanel(props: AutomationsListPanelProps): React.JS // Hosts moved into the Filters menu, so its toolbar row is the focus fallback now. const toolbarRef = useRef(null) const pendingKeyboardScrollRef = useRef(false) - const rowKeys = React.useMemo(() => filteredRows.map((row) => row.key), [filteredRows]) - const visibleItems = React.useMemo( - () => [ - ...filteredRows.map((row) => ({ kind: 'local' as const, id: row.key })), - ...filteredExternalAutomationEntries.map((entry) => ({ - kind: 'external' as const, - id: entry.key - })) - ], - [filteredExternalAutomationEntries, filteredRows] + // Why: keyboard traversal and focus recovery read render order, which the sort owns. + const rowKeys = React.useMemo( + () => sortedListItems.filter((item) => item.kind === 'local').map((item) => item.id), + [sortedListItems] ) - useAutomationListFocusRecovery({ rowKeys, containerRef: listRef, fallbackRef: toolbarRef }) + const visibleItems = React.useMemo( + () => sortedListItems.map((item) => ({ kind: item.kind, id: item.id })), + [sortedListItems] + ) + useAutomationListFocusRecovery({ + rowKeys, + containerRef: listRef, + fallbackRef: toolbarRef + }) const handleSearchArrowNavigate = React.useCallback( (key: AutomationListArrowKey) => { const next = getAutomationListArrowNavigationTarget({ @@ -331,24 +342,30 @@ export function AutomationsListPanel(props: AutomationsListPanelProps): React.JS > {hasFilteredListItems ? (
- +
- - { - selectAutomationRow(null) - selectExternalKey(entryKey) - setActivePaneTab('overview') - onOpenDetail() - }} - onRequestAction={requestExternalAction} - onEdit={openEditExternalDialog} - /> + {sortedListItems.map((item) => + item.kind === 'local' ? ( + + ) : ( + { + selectAutomationRow(null) + selectExternalKey(entryKey) + setActivePaneTab('overview') + onOpenDetail() + }} + onRequestAction={requestExternalAction} + onEdit={openEditExternalDialog} + /> + ) + )}
) : ( diff --git a/src/renderer/src/components/automations/AutomationsPage.create-destination.test.tsx b/src/renderer/src/components/automations/AutomationsPage.create-destination.test.tsx index a0778029ef9..0635ecdb6ff 100644 --- a/src/renderer/src/components/automations/AutomationsPage.create-destination.test.tsx +++ b/src/renderer/src/components/automations/AutomationsPage.create-destination.test.tsx @@ -19,7 +19,6 @@ import { addRuntimeProject, api, installAutomationsPageHarness, - listedRow, mocks, renderPage, runtimeHost, @@ -30,6 +29,7 @@ import { scopedList, settleHostQueries } from './automations-page-test-harness' +import { listedRow } from './automations-page-listed-items' import { makeAutomation, REPO_ID, WORKSPACE_ID } from './automations-page-fixtures' import type { Repo } from '../../../../shared/repo-types' import type { ProjectHostSetup } from '../../../../shared/project-types' diff --git a/src/renderer/src/components/automations/AutomationsPage.cross-authority-actions.test.tsx b/src/renderer/src/components/automations/AutomationsPage.cross-authority-actions.test.tsx index 618c092b45a..8193502cb36 100644 --- a/src/renderer/src/components/automations/AutomationsPage.cross-authority-actions.test.tsx +++ b/src/renderer/src/components/automations/AutomationsPage.cross-authority-actions.test.tsx @@ -22,6 +22,7 @@ import { SELF_PRECONDITION, settleHostQueries } from './automations-page-test-harness' +import { listedRows } from './automations-page-listed-items' import { makeAutomation } from './automations-page-fixtures' installAutomationsPageHarness() @@ -36,9 +37,7 @@ async function collidingHosts(): Promise { } function selectDesktopRow(): string { - const row = mocks.listPanel?.filteredRows.find( - (candidate) => candidate.automation.name === 'Desktop nightly' - ) + const row = listedRows().find((candidate) => candidate.automation.name === 'Desktop nightly') expect(row).toBeDefined() return row?.key ?? '' } @@ -58,9 +57,7 @@ describe('AutomationsPage row actions under a colliding automation id', () => { await renderPage() await settleHostQueries() - const remote = mocks.listPanel?.filteredRows.find( - (candidate) => candidate.automation.name === 'Remote nightly' - ) + const remote = listedRows().find((candidate) => candidate.automation.name === 'Remote nightly') await act(async () => { mocks.listPanel?.selectAutomationRow(remote?.key ?? '') }) diff --git a/src/renderer/src/components/automations/AutomationsPage.external-scope.test.tsx b/src/renderer/src/components/automations/AutomationsPage.external-scope.test.tsx index cd966dcbcd7..b4ea3413cb3 100644 --- a/src/renderer/src/components/automations/AutomationsPage.external-scope.test.tsx +++ b/src/renderer/src/components/automations/AutomationsPage.external-scope.test.tsx @@ -20,6 +20,7 @@ import { RUNTIME_SELF_FILTER, settleHostQueries } from './automations-page-test-harness' +import { listedExternalEntries } from './automations-page-listed-items' import { makeExternalManager } from './automations-page-fixtures' installAutomationsPageHarness() @@ -117,7 +118,7 @@ describe('AutomationsPage external manager probes', () => { await renderPage() await settleHostQueries() - expect(mocks.listPanel?.filteredExternalAutomationEntries).toEqual([]) + expect(listedExternalEntries()).toEqual([]) }) it('drops the previous host rows when the selection moves, not when the new probe lands', async () => { @@ -127,7 +128,7 @@ describe('AutomationsPage external manager probes', () => { const { rerender } = await renderPage() await settleHostQueries() - expect(mocks.listPanel?.filteredExternalAutomationEntries.length).toBeGreaterThan(0) + expect(listedExternalEntries().length).toBeGreaterThan(0) // The new host never answers, so anything still listed belongs to the old one. api.automations.listExternalManagerForOwner.mockImplementation( @@ -137,7 +138,7 @@ describe('AutomationsPage external manager probes', () => { await rerender() await settleHostQueries() - expect(mocks.listPanel?.filteredExternalAutomationEntries).toEqual([]) + expect(listedExternalEntries()).toEqual([]) }) it('reports a host it could not check rather than showing it as clean', async () => { diff --git a/src/renderer/src/components/automations/AutomationsPage.notice-recovery.test.tsx b/src/renderer/src/components/automations/AutomationsPage.notice-recovery.test.tsx index d99cfb91dac..69c27640be1 100644 --- a/src/renderer/src/components/automations/AutomationsPage.notice-recovery.test.tsx +++ b/src/renderer/src/components/automations/AutomationsPage.notice-recovery.test.tsx @@ -15,7 +15,6 @@ import { addRuntimeProject, api, installAutomationsPageHarness, - listedRow, mocks, renderPage, runtimeHost, @@ -26,6 +25,7 @@ import { scopedList, settleHostQueries } from './automations-page-test-harness' +import { listedRow } from './automations-page-listed-items' import { makeAutomation } from './automations-page-fixtures' installAutomationsPageHarness() diff --git a/src/renderer/src/components/automations/AutomationsPage.refresh-selection.test.tsx b/src/renderer/src/components/automations/AutomationsPage.refresh-selection.test.tsx index 523cc50df7d..f33c4d09d6c 100644 --- a/src/renderer/src/components/automations/AutomationsPage.refresh-selection.test.tsx +++ b/src/renderer/src/components/automations/AutomationsPage.refresh-selection.test.tsx @@ -22,6 +22,7 @@ import { SELF_PRECONDITION, settleHostQueries } from './automations-page-test-harness' +import { listedRows } from './automations-page-listed-items' import { makeAutomation, makeRun } from './automations-page-fixtures' installAutomationsPageHarness() @@ -69,7 +70,7 @@ describe('AutomationsPage refresh', () => { await renderPage() - expect(mocks.listPanel?.filteredRows[0]?.usageSummary).toEqual(usageSummary) + expect(listedRows()[0]?.usageSummary).toEqual(usageSummary) }) it('does not re-list through the active runtime just because one is selected', async () => { @@ -231,9 +232,7 @@ describe('AutomationsPage multi-host selection', () => { ) ).toEqual(['Desktop nightly', 'Remote nightly']) - const remote = mocks.listPanel?.filteredRows.find( - (row) => row.automation.name === 'Remote nightly' - ) + const remote = listedRows().find((row) => row.automation.name === 'Remote nightly') await act(async () => { mocks.listPanel?.selectAutomationRow(remote?.key ?? '') }) diff --git a/src/renderer/src/components/automations/AutomationsPage.run-visibility.test.tsx b/src/renderer/src/components/automations/AutomationsPage.run-visibility.test.tsx index f7ad0be65a7..5e605a4487a 100644 --- a/src/renderer/src/components/automations/AutomationsPage.run-visibility.test.tsx +++ b/src/renderer/src/components/automations/AutomationsPage.run-visibility.test.tsx @@ -16,12 +16,12 @@ import type { Automation } from '../../../../shared/automations-types' import { api, installAutomationsPageHarness, - listedRow, mocks, renderPage, scopedList, settleHostQueries } from './automations-page-test-harness' +import { listedRow, listedRows } from './automations-page-listed-items' import { makeAutomation } from './automations-page-fixtures' installAutomationsPageHarness() @@ -42,7 +42,7 @@ function desktopStoreHolds(automations: Automation[]): void { /** The next-run column reads this; the mocked list panel renders only names. */ function listedNextRunAt(): number | null | undefined { - return mocks.listPanel?.filteredRows[0]?.automation.nextRunAt + return listedRows()[0]?.automation.nextRunAt } describe('AutomationsPage run visibility', () => { diff --git a/src/renderer/src/components/automations/AutomationsPage.test.tsx b/src/renderer/src/components/automations/AutomationsPage.test.tsx index 768d3250d80..a62a433a4d1 100644 --- a/src/renderer/src/components/automations/AutomationsPage.test.tsx +++ b/src/renderer/src/components/automations/AutomationsPage.test.tsx @@ -24,13 +24,13 @@ import { api, DESKTOP_SELF_OWNER, installAutomationsPageHarness, - listedRow, mocks, renderPage, rows, scopedList, SELF_PRECONDITION } from './automations-page-test-harness' +import { listedRow, listedExternalEntries } from './automations-page-listed-items' import { makeAutomation, makeExternalManager, @@ -147,7 +147,7 @@ describe('AutomationsPage list rendering', () => { api.automations.updateExternalForOwner.mockResolvedValue(undefined) await renderPage() - const entry = mocks.listPanel?.filteredExternalAutomationEntries[0] + const entry = listedExternalEntries()[0] if (!entry) { throw new Error('no external entry to edit') } @@ -177,7 +177,7 @@ describe('AutomationsPage list rendering', () => { api.automations.runExternalActionForOwner.mockResolvedValue(undefined) await renderPage() - const entry = mocks.listPanel?.filteredExternalAutomationEntries[0] + const entry = listedExternalEntries()[0] if (!entry) { throw new Error('no external entry to act on') } @@ -217,7 +217,7 @@ describe('AutomationsPage list rendering', () => { api.automations.listExternalRunsForOwner.mockResolvedValue({ runs: [], total: 0 }) const { container } = await renderPage() - const entry = mocks.listPanel?.filteredExternalAutomationEntries[0] + const entry = listedExternalEntries()[0] if (!entry) { throw new Error('no external entry to read runs for') } diff --git a/src/renderer/src/components/automations/AutomationsPageListPanel.tsx b/src/renderer/src/components/automations/AutomationsPageListPanel.tsx index 25c7ff7b88c..7adea56c608 100644 --- a/src/renderer/src/components/automations/AutomationsPageListPanel.tsx +++ b/src/renderer/src/components/automations/AutomationsPageListPanel.tsx @@ -1,6 +1,7 @@ import React from 'react' import type { AutomationsPageController } from './use-automations-page-controller' import { AutomationsListPanel } from './AutomationsListPanel' +import { nextAutomationListSort } from './automation-list-view' export function AutomationsPageListPanel({ controller, @@ -45,8 +46,6 @@ export function AutomationsPageListPanel({ hasListItems, hasFilteredListItems, isListSearchQueryTooLarge, - filteredRows, - filteredExternalAutomationEntries, selectedRow, selectedExternal, searchCounts @@ -79,8 +78,9 @@ export function AutomationsPageListPanel({ void pageRefresh.refresh() } }} - filteredRows={filteredRows} - filteredExternalAutomationEntries={filteredExternalAutomationEntries} + sortedListItems={list.sortedListItems} + listSort={local.listSort} + onListSortChange={(field) => local.setListSort(nextAutomationListSort(local.listSort, field))} selectedRowKey={selectedRow?.key ?? null} selectedExternalKey={local.selectedExternalKey} selectedExternal={selectedExternal} diff --git a/src/renderer/src/components/automations/automation-list-view-sort.test.ts b/src/renderer/src/components/automations/automation-list-view-sort.test.ts index d3dfbe63133..8fbef8a0590 100644 --- a/src/renderer/src/components/automations/automation-list-view-sort.test.ts +++ b/src/renderer/src/components/automations/automation-list-view-sort.test.ts @@ -5,32 +5,34 @@ import { type AutomationListSort, type AutomationListViewItem } from './automation-list-view' +import { unscopedAutomationListRows } from './automation-list-row-identity' import { makeAutomation } from './automations-page-fixtures' -const locale = vi.hoisted(() => ({ value: 'en' })) -vi.mock('@/i18n/i18n', () => ({ getIntlLocale: () => locale.value })) - afterEach(() => { vi.restoreAllMocks() - locale.value = 'en' }) -function rows(count = 512): AutomationListViewItem[] { +function items(count = 512): AutomationListViewItem[] { const names = ['Alpha', 'álpha', 'Ångström', 'Zebra', 'Örebro', 'I', 'ı', 'İ', 'job 10', 'job 2'] return buildAutomationListViewItems({ - automations: Array.from({ length: count }, (_, index) => - makeAutomation({ id: `job-${index}`, name: names[(index * 7) % names.length] }) + rows: unscopedAutomationListRows( + Array.from({ length: count }, (_, index) => + makeAutomation({ + id: `job-${index}`, + name: names[(index * 7) % names.length] + }) + ) ), - externalEntries: [], - runs: [] + externalEntries: [] }) } -function previousOrder(items: AutomationListViewItem[], sort: AutomationListSort) { +/** The pre-collator comparator, resolving options on every comparison. */ +function previousOrder(list: AutomationListViewItem[], sort: AutomationListSort, locale: string) { function compare(left: AutomationListViewItem, right: AutomationListViewItem) { const value = sort.field === 'name' - ? left.name.localeCompare(right.name, locale.value, { sensitivity: 'base' }) + ? left.name.localeCompare(right.name, locale, { sensitivity: 'base' }) : (left.lastRunAt ?? 0) - (right.lastRunAt ?? 0) return value !== 0 ? sort.direction === 'asc' @@ -38,37 +40,35 @@ function previousOrder(items: AutomationListViewItem[], sort: AutomationListSort : -value : left.id.localeCompare(right.id) } - return [...items].sort(compare) + return [...list].sort(compare) } describe('automation list collation', () => { it.each(['en', 'sv', 'tr', 'ja'])( 'preserves %s ordering, tie-breaks and input identity', - (language) => { - locale.value = language - const items = rows() - const original = [...items] + (locale) => { + const list = items() + const original = [...list] for (const direction of ['asc', 'desc'] as const) { const sort = { field: 'name', direction } as const - const expected = previousOrder(items, sort) - const result = sortAutomationListViewItems(items, sort) + const expected = previousOrder(list, sort, locale) + const result = sortAutomationListViewItems(list, sort, locale) expect(result).toEqual(expected) expect(result.every((row, index) => row === expected[index])).toBe(true) } - expect(items).toEqual(original) + expect(list).toEqual(original) } ) - it('resolves collation once per name sort and responds to locale changes', () => { - const items = rows() + it('resolves collation once per name sort and follows the locale it is given', () => { + const list = items() const OriginalCollator = Intl.Collator const construct = vi.spyOn(Intl, 'Collator').mockImplementation(function (locales, options) { return new OriginalCollator(locales, options) }) const compare = vi.spyOn(String.prototype, 'localeCompare') - sortAutomationListViewItems(items, { field: 'name', direction: 'asc' }) - locale.value = 'sv' - sortAutomationListViewItems(items, { field: 'name', direction: 'desc' }) + sortAutomationListViewItems(list, { field: 'name', direction: 'asc' }, 'en') + sortAutomationListViewItems(list, { field: 'name', direction: 'desc' }, 'sv') expect(construct.mock.calls).toEqual([ ['en', { sensitivity: 'base' }], ['sv', { sensitivity: 'base' }] @@ -76,16 +76,39 @@ describe('automation list collation', () => { expect(compare.mock.calls.filter((args) => args.length >= 3)).toHaveLength(0) }) + it('orders by row key, not the bare automation ID, so hosts cannot collapse', () => { + const duplicate = makeAutomation({ id: 'shared', name: 'Same' }) + const list = buildAutomationListViewItems({ + rows: [ + { + key: 'row|host-b|shared', + automation: duplicate, + hostLabel: 'b', + usageSummary: null + }, + { + key: 'row|host-a|shared', + automation: duplicate, + hostLabel: 'a', + usageSummary: null + } + ], + externalEntries: [] + }) + const sorted = sortAutomationListViewItems(list, { field: 'name', direction: 'asc' }, 'en') + expect(sorted.map((item) => item.id)).toEqual(['row|host-a|shared', 'row|host-b|shared']) + }) + it('does not construct collation for unsorted, time-sorted or trivial lists', () => { - const items = rows() + const list = items() const construct = vi.spyOn(Intl, 'Collator') - expect(sortAutomationListViewItems(items, null)).toEqual(items) + expect(sortAutomationListViewItems(list, null, 'en')).toEqual(list) const sort = { field: 'lastRun', direction: 'desc' } as const - expect(sortAutomationListViewItems(items, sort)).toEqual(previousOrder(items, sort)) - expect(sortAutomationListViewItems([], { field: 'name', direction: 'asc' })).toEqual([]) + expect(sortAutomationListViewItems(list, sort, 'en')).toEqual(previousOrder(list, sort, 'en')) + expect(sortAutomationListViewItems([], { field: 'name', direction: 'asc' }, 'en')).toEqual([]) expect( - sortAutomationListViewItems(items.slice(0, 1), { field: 'name', direction: 'asc' }) - ).toEqual(items.slice(0, 1)) + sortAutomationListViewItems(list.slice(0, 1), { field: 'name', direction: 'asc' }, 'en') + ).toEqual(list.slice(0, 1)) expect(construct).not.toHaveBeenCalled() }) }) diff --git a/src/renderer/src/components/automations/automation-list-view.test.ts b/src/renderer/src/components/automations/automation-list-view.test.ts index 188c94f3db7..169fbdd360a 100644 --- a/src/renderer/src/components/automations/automation-list-view.test.ts +++ b/src/renderer/src/components/automations/automation-list-view.test.ts @@ -1,7 +1,6 @@ import { describe, expect, it } from 'vitest' import type { Automation, - AutomationRun, AutomationRunStatus, ExternalAutomationJob, ExternalAutomationManager @@ -48,31 +47,6 @@ function makeAutomation(overrides: Partial = {}): Automation { } } -function makeRun(overrides: Partial = {}): AutomationRun { - return { - id: 'run-1', - automationId: 'automation-1', - title: 'Zebra job', - scheduledFor: 10, - status: 'completed', - trigger: 'scheduled', - workspaceId: 'worktree-1', - sessionKind: 'terminal', - chatSessionId: null, - terminalSessionId: null, - terminalPaneKey: null, - terminalPtyId: null, - outputSnapshot: null, - precheckResult: null, - usage: null, - error: null, - startedAt: 20, - dispatchedAt: 30, - createdAt: 10, - ...overrides - } -} - function makeExternalEntry( overrides: Partial = {} ): ExternalAutomationListEntry { @@ -118,22 +92,69 @@ function makeExternalEntry( } } +/** A catalog row with an optional projected last-run status, keyed like a real host row. */ +function makeCatalogRow( + id: string, + overrides: Partial = {}, + lastRunStatus?: AutomationRunStatus +): AutomationListRow { + return { + key: `row|host|${id}`, + automation: makeAutomation({ id, ...overrides }), + hostLabel: 'This computer', + usageSummary: lastRunStatus + ? { + knownRuns: 1, + unavailableRuns: 0, + inputTokens: 0, + outputTokens: 0, + cacheTokens: 0, + reasoningOutputTokens: 0, + totalTokens: 0, + estimatedCostUsd: null, + lastRunStatus, + lastRunAt: 111 + } + : null + } +} + +const rowKey = (id: string): string => `row|host|${id}` + describe('automation-list-view', () => { it('counts and detects active filters', () => { - expect(isAutomationListFilterActive({ status: 'all', lastRun: 'all', agentIds: [] })).toBe( - false - ) - expect(isAutomationListFilterActive({ status: 'paused', lastRun: 'all', agentIds: [] })).toBe( - true - ) - expect(countAutomationListFilters({ status: 'paused', lastRun: 'failed', agentIds: [] })).toBe( - 2 - ) + expect( + isAutomationListFilterActive({ + status: 'all', + lastRun: 'all', + agentIds: [] + }) + ).toBe(false) + expect( + isAutomationListFilterActive({ + status: 'paused', + lastRun: 'all', + agentIds: [] + }) + ).toBe(true) + expect( + countAutomationListFilters({ + status: 'paused', + lastRun: 'failed', + agentIds: [] + }) + ).toBe(2) }) it('toggles sort direction and defaults last run to newest first', () => { - expect(nextAutomationListSort(null, 'name')).toEqual({ field: 'name', direction: 'asc' }) - expect(nextAutomationListSort(null, 'lastRun')).toEqual({ field: 'lastRun', direction: 'desc' }) + expect(nextAutomationListSort(null, 'name')).toEqual({ + field: 'name', + direction: 'asc' + }) + expect(nextAutomationListSort(null, 'lastRun')).toEqual({ + field: 'lastRun', + direction: 'desc' + }) expect(nextAutomationListSort({ field: 'name', direction: 'asc' }, 'name')).toEqual({ field: 'name', direction: 'desc' @@ -146,82 +167,62 @@ describe('automation-list-view', () => { it('filters by enabled state and last-run outcome', () => { const items = applyAutomationListView({ - automations: [ - makeAutomation({ id: 'paused', name: 'Paused', enabled: false }), - makeAutomation({ id: 'ok', name: 'Healthy' }) + rows: [ + makeCatalogRow('paused', { name: 'Paused', enabled: false }, 'completed'), + makeCatalogRow('ok', { name: 'Healthy' }, 'dispatch_failed') ], externalEntries: [makeExternalEntry()], - runs: [ - makeRun({ automationId: 'paused', status: 'completed' }), - makeRun({ automationId: 'ok', status: 'dispatch_failed' }) - ], filter: { status: 'enabled', lastRun: 'failed', agentIds: [] }, - sort: null + sort: null, + locale: 'en' }) - expect(items.map((item) => item.id)).toEqual(['ok', 'manager-1:job-1']) + expect(items.map((item) => item.id)).toEqual([rowKey('ok'), 'manager-1:job-1']) }) it('filters local rows by multiple agents and leaves external rows out of agent scopes', () => { const items = applyAutomationListView({ - automations: [ - makeAutomation({ id: 'codex-job', agentId: 'codex' }), - makeAutomation({ id: 'claude-job', agentId: 'claude' }) + rows: [ + makeCatalogRow('codex-job', { agentId: 'codex' }), + makeCatalogRow('claude-job', { agentId: 'claude' }) ], externalEntries: [makeExternalEntry()], - runs: [], filter: { status: 'all', lastRun: 'all', agentIds: ['codex', 'claude'] }, - sort: null + sort: null, + locale: 'en' }) - expect(items.map((item) => item.id)).toEqual(['codex-job', 'claude-job']) + expect(items.map((item) => item.id)).toEqual([rowKey('codex-job'), rowKey('claude-job')]) }) it('counts an agent filter alongside status and last-run filters', () => { - expect(isAutomationListFilterActive({ status: 'all', lastRun: 'all', agentIds: [] })).toBe( - false - ) expect( - countAutomationListFilters({ status: 'paused', lastRun: 'failed', agentIds: ['codex'] }) + isAutomationListFilterActive({ + status: 'all', + lastRun: 'all', + agentIds: [] + }) + ).toBe(false) + expect( + countAutomationListFilters({ + status: 'paused', + lastRun: 'failed', + agentIds: ['codex'] + }) ).toBe(3) }) it('sorts by name across local and external rows', () => { const items = applyAutomationListView({ - automations: [makeAutomation({ name: 'Zebra job' })], + rows: [makeCatalogRow('zebra', { name: 'Zebra job' })], externalEntries: [makeExternalEntry({ name: 'Alpha digest' })], - runs: [], filter: { status: 'all', lastRun: 'all', agentIds: [] }, - sort: { field: 'name', direction: 'asc' } + sort: { field: 'name', direction: 'asc' }, + locale: 'en' }) expect(items.map((item) => item.name)).toEqual(['Alpha digest', 'Zebra job']) }) it('filters catalog rows by status, agent, and the projected last-run status', () => { - function makeCatalogRow( - id: string, - overrides: Partial, - lastRunStatus?: AutomationRunStatus - ): AutomationListRow { - return { - key: `row|host|${id}`, - automation: makeAutomation({ id, ...overrides }), - hostLabel: 'This computer', - usageSummary: lastRunStatus - ? { - knownRuns: 1, - unavailableRuns: 0, - inputTokens: 0, - outputTokens: 0, - cacheTokens: 0, - reasoningOutputTokens: 0, - totalTokens: 0, - estimatedCostUsd: null, - lastRunStatus, - lastRunAt: 111 - } - : null - } - } const rows = [ makeCatalogRow('paused-codex', { enabled: false, agentId: 'codex' }), makeCatalogRow('failed-claude', { agentId: 'claude' }, 'dispatch_failed'), @@ -229,9 +230,10 @@ describe('automation-list-view', () => { makeCatalogRow('never-codex', { agentId: 'codex' }) ] const ids = (filter: Partial) => - filterAutomationListRows(rows, { ...EMPTY_AUTOMATION_LIST_FILTER, ...filter }).map( - (row) => row.automation.id - ) + filterAutomationListRows(rows, { + ...EMPTY_AUTOMATION_LIST_FILTER, + ...filter + }).map((row) => row.automation.id) expect(ids({ status: 'paused' })).toEqual(['paused-codex']) expect(ids({ agentIds: ['claude'] })).toEqual(['failed-claude']) @@ -249,7 +251,10 @@ describe('automation-list-view', () => { catalogRef: targetId === null ? null - : { authority: { kind: 'desktop' }, selector: { kind: 'ssh', targetId } }, + : { + authority: { kind: 'desktop' }, + selector: { kind: 'ssh', targetId } + }, hostLabel: targetId ?? '', usageSummary: null }) @@ -257,9 +262,10 @@ describe('automation-list-view', () => { const keyOf = (row: AutomationListRow): string => row.catalogRef ? hostStableKey(row.catalogRef) : '' const ids = (hostStableKeys: readonly string[]) => - filterAutomationListRows(rows, { ...EMPTY_AUTOMATION_LIST_FILTER, hostStableKeys }).map( - (row) => row.automation.id - ) + filterAutomationListRows(rows, { + ...EMPTY_AUTOMATION_LIST_FILTER, + hostStableKeys + }).map((row) => row.automation.id) // Multi-select is any-of; a pre-catalog row names no host and is excluded. expect(ids([keyOf(rows[0]), keyOf(rows[1])])).toEqual(['on-a', 'on-b']) @@ -290,15 +296,22 @@ describe('automation-list-view', () => { it('sorts by last run newest first and keeps never-run rows last', () => { const items = applyAutomationListView({ - automations: [ - makeAutomation({ id: 'old', name: 'Old' }), - makeAutomation({ id: 'never', name: 'Never' }) + rows: [ + makeCatalogRow('old', { + name: 'Old', + lastRunAt: Date.parse('2026-08-11T09:00:00Z') + }), + makeCatalogRow('never', { name: 'Never' }) ], externalEntries: [makeExternalEntry({ lastRunAt: '2026-08-12T09:00:00Z' })], - runs: [makeRun({ automationId: 'old', dispatchedAt: Date.parse('2026-08-11T09:00:00Z') })], filter: { status: 'all', lastRun: 'all', agentIds: [] }, - sort: { field: 'lastRun', direction: 'desc' } + sort: { field: 'lastRun', direction: 'desc' }, + locale: 'en' }) - expect(items.map((item) => item.id)).toEqual(['manager-1:job-1', 'old', 'never']) + expect(items.map((item) => item.id)).toEqual([ + 'manager-1:job-1', + rowKey('old'), + rowKey('never') + ]) }) }) diff --git a/src/renderer/src/components/automations/automation-list-view.ts b/src/renderer/src/components/automations/automation-list-view.ts index 459d4b0f588..cedd394ed23 100644 --- a/src/renderer/src/components/automations/automation-list-view.ts +++ b/src/renderer/src/components/automations/automation-list-view.ts @@ -1,5 +1,3 @@ -import { getIntlLocale } from '@/i18n/i18n' -import type { Automation, AutomationRun } from '../../../../shared/automations-types' import type { TuiAgent } from '../../../../shared/tui-agent' import { hostStableKey } from '../../../../shared/automation-owner-key' import type { AutomationListRow } from './automation-list-row-identity' @@ -7,8 +5,6 @@ import type { ExternalAutomationListEntry } from './external-automation-list-ent import { getAutomationRowLastRunSnapshot, getExternalAutomationLastRunSnapshot, - getLocalAutomationLastRunSnapshot, - indexLatestAutomationRuns, type AutomationLastRunSnapshot } from './automation-list-last-run' @@ -22,6 +18,13 @@ export type AutomationListSort = { direction: AutomationListSortDirection } +/** + * A row and an external job flattened to what the shared list renders and sorts. + * + * `id` is the row's own key, never the bare automation ID: under All hosts two + * authorities can return the same ID, and the sort tie-break decides render + * order, so a bare ID would collapse them. See `automation-list-row-identity`. + */ export type AutomationListViewItem = | { kind: 'local' @@ -31,7 +34,7 @@ export type AutomationListViewItem = lastRunAt: number | null lastRun: AutomationLastRunSnapshot agentId: TuiAgent - automation: Automation + row: AutomationListRow } | { kind: 'external' @@ -117,30 +120,26 @@ function matchesLastRunFilter( return snapshot.tone === filter } +/** Flattens the two rendered collections into one sortable list, preserving row identity. */ export function buildAutomationListViewItems({ - automations, - externalEntries, - runs + rows, + externalEntries }: { - automations: readonly Automation[] + rows: readonly AutomationListRow[] externalEntries: readonly ExternalAutomationListEntry[] - runs: readonly AutomationRun[] }): AutomationListViewItem[] { - const lastRunByAutomationId = indexLatestAutomationRuns(runs) - const locals: AutomationListViewItem[] = automations.map((automation) => { - const lastRun = getLocalAutomationLastRunSnapshot( - automation, - lastRunByAutomationId.get(automation.id) - ) + const locals: AutomationListViewItem[] = rows.map((row) => { + // Why: the same snapshot the row cell renders, so the sort matches the column. + const lastRun = getAutomationRowLastRunSnapshot(row) return { kind: 'local', - id: automation.id, - name: automation.name, - enabled: automation.enabled, + id: row.key, + name: row.automation.name, + enabled: row.automation.enabled, lastRunAt: lastRun.at, lastRun, - agentId: automation.agentId, - automation + agentId: row.automation.agentId, + row } }) const externals: AutomationListViewItem[] = externalEntries.map((entry) => { @@ -217,34 +216,21 @@ export function filterExternalAutomationListEntries( ) } -export function filterAutomationListViewItems( - items: readonly AutomationListViewItem[], - filter: AutomationListFilter -): AutomationListViewItem[] { - if (!isAutomationListFilterActive(filter)) { - return [...items] - } - return items.filter( - (item) => - matchesStatusFilter(item.enabled, filter.status) && - matchesLastRunFilter(item.lastRun, filter.lastRun) && - (filter.agentIds.length === 0 || - (item.agentId !== null && filter.agentIds.includes(item.agentId))) - ) -} - +/** + * `locale` is a parameter, not a `getIntlLocale()` read, so callers memoizing this + * can declare it — a hidden read is invisible to a dependency array. + */ export function sortAutomationListViewItems( items: readonly AutomationListViewItem[], - sort: AutomationListSort | null + sort: AutomationListSort | null, + locale: string ): AutomationListViewItem[] { if (!sort || items.length < 2) { return [...items] } const next = [...items] const compareNames = - sort.field === 'name' - ? new Intl.Collator(getIntlLocale(), { sensitivity: 'base' }).compare - : null + sort.field === 'name' ? new Intl.Collator(locale, { sensitivity: 'base' }).compare : null next.sort((left, right) => { const compared = compareNames ? compareNames(left.name, right.name) @@ -257,24 +243,26 @@ export function sortAutomationListViewItems( return next } +/** The rendered list: filter each collection with its own rules, then sort as one. */ export function applyAutomationListView({ - automations, + rows, externalEntries, - runs, filter, - sort + sort, + locale }: { - automations: readonly Automation[] + rows: readonly AutomationListRow[] externalEntries: readonly ExternalAutomationListEntry[] - runs: readonly AutomationRun[] filter: AutomationListFilter sort: AutomationListSort | null + locale: string }): AutomationListViewItem[] { return sortAutomationListViewItems( - filterAutomationListViewItems( - buildAutomationListViewItems({ automations, externalEntries, runs }), - filter - ), - sort + buildAutomationListViewItems({ + rows: filterAutomationListRows(rows, filter), + externalEntries: filterExternalAutomationListEntries(externalEntries, filter) + }), + sort, + locale ) } diff --git a/src/renderer/src/components/automations/automations-page-listed-items.ts b/src/renderer/src/components/automations/automations-page-listed-items.ts new file mode 100644 index 00000000000..d87ae62b4a8 --- /dev/null +++ b/src/renderer/src/components/automations/automations-page-listed-items.ts @@ -0,0 +1,32 @@ +/** + * What the page actually listed, read back from the mocked list panel. + * + * Tests act through the same authority-qualified keys and render order the + * user's click carries, rather than synthesizing either. + */ + +import type { AutomationListRow } from './automation-list-row-identity' +import type { ExternalAutomationListEntry } from './external-automation-list-entries' +import { mocks } from './automations-page-test-harness' + +function listedItems() { + return mocks.listPanel?.sortedListItems ?? [] +} + +/** Local rows the page listed, in render order. */ +export function listedRows(): readonly AutomationListRow[] { + return listedItems().flatMap((item) => (item.kind === 'local' ? [item.row] : [])) +} + +/** External entries the page listed, in render order. */ +export function listedExternalEntries(): readonly ExternalAutomationListEntry[] { + return listedItems().flatMap((item) => (item.kind === 'external' ? [item.entry] : [])) +} + +export function listedRow(automationId: string): AutomationListRow { + const row = listedRows().find((entry) => entry.automation.id === automationId) + if (!row) { + throw new Error(`no listed row for ${automationId}`) + } + return row +} diff --git a/src/renderer/src/components/automations/automations-page-test-harness.tsx b/src/renderer/src/components/automations/automations-page-test-harness.tsx index d9fd1088bf7..e34ae0640bc 100644 --- a/src/renderer/src/components/automations/automations-page-test-harness.tsx +++ b/src/renderer/src/components/automations/automations-page-test-harness.tsx @@ -27,6 +27,7 @@ import type { AutomationHostCatalogView } from './use-automation-host-catalog' import type { AutomationCreateDestinationControl } from './use-automation-create-destination' import type { ExternalAutomationListEntry } from './external-automation-list-entries' import type { AutomationListRow } from './automation-list-row-identity' +import type { AutomationListViewItem } from './automation-list-view' import { resetAutomationCapabilityProbes } from './automation-scoped-list-client' import { addRuntimeProject as addRuntimeProjectFixture, @@ -39,7 +40,7 @@ export const RUNTIME_REPO_ID = RUNTIME_REPO_ID_FIXTURE export const RUNTIME_WORKSPACE_ID = RUNTIME_WORKSPACE_ID_FIXTURE export type ListPanelProps = { - filteredExternalAutomationEntries: ExternalAutomationListEntry[] + sortedListItems: readonly AutomationListViewItem[] selectedExternal: ExternalAutomationListEntry | null openEditExternalDialog: ( manager: ExternalAutomationListEntry['manager'], @@ -55,7 +56,6 @@ export type ListPanelProps = { ) => void hasListItems: boolean hasFilteredListItems: boolean - filteredRows: readonly AutomationListRow[] selectedRowKey: string | null selectedExternalKey: string | null hostCatalog: AutomationHostCatalogView @@ -211,30 +211,31 @@ vi.mock('./AutomationsListPanel', () => ({ return (
- ))} - {props.filteredExternalAutomationEntries.map((entry) => ( - - ))} + {props.sortedListItems.map((item) => + item.kind === 'local' ? ( + + ) : ( + + ) + )} {props.hasListItems ? null :
}
) @@ -407,18 +408,6 @@ export async function refreshOnFocus(): Promise { }) } -/** - * The row the page actually listed for an ID, so tests act through the same - * authority-qualified key the user's click carries rather than a synthesized one. - */ -export function listedRow(automationId: string): AutomationListRow { - const row = mocks.listPanel?.filteredRows.find((entry) => entry.automation.id === automationId) - if (!row) { - throw new Error(`no listed row for ${automationId}`) - } - return row -} - export function rows(container: HTMLElement, testId: string): string[] { return [...container.querySelectorAll(`[data-testid="${testId}"]`)].map( (node) => node.textContent ?? '' diff --git a/src/renderer/src/components/automations/use-automations-page-list-state.ts b/src/renderer/src/components/automations/use-automations-page-list-state.ts index 65cce90ffc2..b13c8a689ec 100644 --- a/src/renderer/src/components/automations/use-automations-page-list-state.ts +++ b/src/renderer/src/components/automations/use-automations-page-list-state.ts @@ -4,9 +4,12 @@ import { buildExternalAutomationListEntries } from './external-automation-list-e import { externalAutomationScopeEntries } from './external-automation-scope-gating' import { externalAutomationUncheckedNotice } from './external-automation-unchecked-hosts' import { + buildAutomationListViewItems, filterAutomationListRows, - filterExternalAutomationListEntries + filterExternalAutomationListEntries, + sortAutomationListViewItems } from './automation-list-view' +import { getIntlLocale } from '@/i18n/i18n' import { unscopedAutomationListRows } from './automation-list-row-identity' import { useAutomationHostCatalog } from './use-automation-host-catalog' import { useAutomationListSearch } from './use-automation-list-search' @@ -28,6 +31,7 @@ export function useAutomationsPageListState({ failedAuthorityKeys, listSearchQuery, listFilter, + listSort, selectedRowKey, selectedExternalKey, selectedAutomationRuns, @@ -129,6 +133,21 @@ export function useAutomationsPageListState({ () => externalAutomationUncheckedNotice(scopedExternal.failures, hostCatalog.entries), [hostCatalog.entries, scopedExternal.failures] ) + // Why: a language switch changes collation without touching rows, so the locale + // has to reach the memo as a value. + const sortLocale = getIntlLocale() + const sortedListItems = useMemo( + () => + sortAutomationListViewItems( + buildAutomationListViewItems({ + rows: filteredRows, + externalEntries: filteredExternalAutomationEntries + }), + listSort, + sortLocale + ), + [filteredExternalAutomationEntries, filteredRows, listSort, sortLocale] + ) return { hostCatalog, @@ -146,6 +165,7 @@ export function useAutomationsPageListState({ isListSearchQueryTooLarge, filteredRows, filteredExternalAutomationEntries, + sortedListItems, hasListItems, hasFilteredListItems, searchCounts, diff --git a/src/renderer/src/components/automations/use-automations-page-local-state.ts b/src/renderer/src/components/automations/use-automations-page-local-state.ts index e92f666cb22..7a097b144c3 100644 --- a/src/renderer/src/components/automations/use-automations-page-local-state.ts +++ b/src/renderer/src/components/automations/use-automations-page-local-state.ts @@ -12,7 +12,11 @@ import type { AutomationActionNotice } from './automation-row-action-dispatch' import type { AutomationHostCatalogEntry } from './automation-host-catalog-types' import type { AutomationCreateDestination } from './automation-create-destination' import type { AutomationListRow } from './automation-list-row-identity' -import { EMPTY_AUTOMATION_LIST_FILTER, type AutomationListFilter } from './automation-list-view' +import { + EMPTY_AUTOMATION_LIST_FILTER, + type AutomationListFilter, + type AutomationListSort +} from './automation-list-view' import type { AutomationPaneTab, AutomationRunPageOrigin, @@ -54,6 +58,7 @@ export function useAutomationsPageLocalState(store: AutomationsPageStoreState) { const [isSaving, setIsSaving] = useState(false) const [listSearchQuery, setListSearchQuery] = useState('') const [listFilter, setListFilter] = useState(EMPTY_AUTOMATION_LIST_FILTER) + const [listSort, setListSort] = useState(null) const [createOpen, setCreateOpen] = useState(false) const [createTarget, setCreateTarget] = useState('orca') const [editingAutomationId, setEditingAutomationId] = useState(null) @@ -178,6 +183,8 @@ export function useAutomationsPageLocalState(store: AutomationsPageStoreState) { setListSearchQuery, listFilter, setListFilter, + listSort, + setListSort, createOpen, setCreateOpen, createTarget, diff --git a/src/shared/pane-agent-identity-inventory.test.ts b/src/shared/pane-agent-identity-inventory.test.ts index ee868bfcc16..d493dec1aef 100644 --- a/src/shared/pane-agent-identity-inventory.test.ts +++ b/src/shared/pane-agent-identity-inventory.test.ts @@ -56,7 +56,7 @@ const INVENTORY: readonly InventoryGroup[] = [ 'src/renderer/src/components/agent-session-continuation/AgentSessionContinuationDialog.tsx', 2 ], - ['src/renderer/src/components/automations/AutomationListLocalRows.tsx', 2], + ['src/renderer/src/components/automations/AutomationListLocalRow.tsx', 2], 'src/renderer/src/components/automations/automation-draft-model.ts', ['src/renderer/src/components/automations/automation-list-search-rows.ts', 2], ['src/renderer/src/components/dashboard-popout/AgentMapSnapshotWorkspaceMenu.tsx', 2], From 55dcc5ceeeac04dc515ce7d14c084fdee5b82c69 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:26:36 -0700 Subject: [PATCH 04/17] test: pin terminal Codex home to an explicit managed account (#18935) --- tests/e2e/terminal-codex-home.spec.ts | 52 +++++++++++++++++++++------ 1 file changed, 41 insertions(+), 11 deletions(-) diff --git a/tests/e2e/terminal-codex-home.spec.ts b/tests/e2e/terminal-codex-home.spec.ts index 1f85a4f4c9b..3364152d38c 100644 --- a/tests/e2e/terminal-codex-home.spec.ts +++ b/tests/e2e/terminal-codex-home.spec.ts @@ -1,3 +1,5 @@ +import { mkdirSync, writeFileSync } from 'node:fs' +import path from 'node:path' import { test, expect } from './helpers/orca-app' import { execInTerminal, @@ -27,7 +29,42 @@ test.describe('Terminal Codex runtime home', () => { await ensureTerminalVisible(orcaPage) }) - test('terminal process receives the Orca-managed Codex home', async ({ orcaPage }) => { + test('terminal process receives the selected account Codex home', async ({ + electronApp, + orcaPage + }) => { + const userData = await electronApp.evaluate(({ app }) => app.getPath('userData')) + const accountId = 'e2e-terminal-home' + const managedHomePath = path.join(userData, 'codex-accounts', accountId, 'home') + mkdirSync(managedHomePath, { recursive: true }) + writeFileSync(path.join(managedHomePath, '.orca-managed-home'), `${accountId}\n`) + writeFileSync( + path.join(managedHomePath, 'auth.json'), + JSON.stringify({ OPENAI_API_KEY: 'e2e-placeholder' }) + ) + await orcaPage.evaluate( + async ({ accountId, managedHomePath }) => { + const state = window.__store!.getState() + await state.updateSettings({ + codexManagedAccounts: [ + { + id: accountId, + email: 'terminal-home@example.invalid', + managedHomePath, + createdAt: 1, + updatedAt: 1, + lastAuthenticatedAt: 1 + } + ], + activeCodexManagedAccountId: accountId, + activeCodexManagedAccountIdsByRuntime: { host: accountId, wsl: {} } + }) + const tab = state.createTab(state.activeWorktreeId!) + state.setActiveTab(tab.id) + state.setActiveTabType('terminal') + }, + { accountId, managedHomePath } + ) await waitForActiveTerminalManager(orcaPage) const ptyId = await waitForActivePanePtyId(orcaPage) const marker = `__ORCA_CODEX_HOME_E2E_${Date.now()}__` @@ -43,17 +80,10 @@ test.describe('Terminal Codex runtime home', () => { .poll( async () => { probe = readCodexHomeProbe(await getTerminalContent(orcaPage), marker) - return Boolean( - probe?.codexHome && - probe.orcaCodexHome && - probe.codexHome === probe.orcaCodexHome && - /[\\/]codex-runtime-home[\\/]home$/.test(probe.codexHome) - ) + return probe }, - { timeout: 15_000, message: 'Terminal did not expose Orca-managed Codex home env' } + { timeout: 15_000, message: 'Terminal did not expose the selected Codex account home' } ) - .toBe(true) - - expect(probe?.codexHome).toBe(probe?.orcaCodexHome) + .toEqual({ codexHome: managedHomePath, orcaCodexHome: managedHomePath }) }) }) From 59756b8a1cec8b266a1461056f18c768454bfcb0 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:31:11 -0700 Subject: [PATCH 05/17] test: deliver real terminal input and preserve setup reports (#18939) --- .../e2e/terminal-scroll-intent-follow.spec.ts | 44 +++++++++++-------- .../terminal-send-agent-prompt-submit.spec.ts | 1 + tests/tools/repro-terminal-send-submit.mjs | 4 +- 3 files changed, 30 insertions(+), 19 deletions(-) diff --git a/tests/e2e/terminal-scroll-intent-follow.spec.ts b/tests/e2e/terminal-scroll-intent-follow.spec.ts index c4d8b171fed..ae2dfa15466 100644 --- a/tests/e2e/terminal-scroll-intent-follow.spec.ts +++ b/tests/e2e/terminal-scroll-intent-follow.spec.ts @@ -168,6 +168,7 @@ async function injectQueuedWriteThenType(page: Page, paneKey: string): Promise { const injectionTarget = window as Window & { __terminalPtyDataInjection?: { inject: (paneKey: string, data: string) => boolean } + __releaseScrollIntentTestWrite?: () => void } const state = window.__store?.getState() const worktreeId = state?.activeWorktreeId @@ -184,38 +185,44 @@ async function injectQueuedWriteThenType(page: Page, paneKey: string): Promise void } | null } = { write: null } + const heldWrites: { data: string; callback?: () => void }[] = [] terminal.write = ((data: string, callback?: () => void) => { - holder.write = { data, callback } + heldWrites.push({ data, callback }) }) as typeof terminal.write + injectionTarget.__releaseScrollIntentTestWrite = () => { + terminal.write = originalWrite + delete injectionTarget.__releaseScrollIntentTestWrite + for (const held of heldWrites) { + originalWrite.call(terminal, held.data, held.callback) + } + } try { const payload = '\x1b[?2026h\r\x1b[2KWorking in-flight\x1b[?2026l' if (!injectionTarget.__terminalPtyDataInjection?.inject(targetPaneKey, payload)) { throw new Error('PTY injector unavailable') } + if (heldWrites.length === 0) { + throw new Error('Foreground terminal write was not captured') + } const textarea = pane.container.querySelector('.xterm-helper-textarea') if (!textarea) { throw new Error('xterm helper textarea unavailable') } textarea.focus() - const event = new KeyboardEvent('keydown', { - bubbles: true, - cancelable: true, - key: 'x', - code: 'KeyX' - }) - Object.defineProperty(event, 'keyCode', { configurable: true, value: 88 }) - Object.defineProperty(event, 'which', { configurable: true, value: 88 }) - textarea.dispatchEvent(event) - } finally { - terminal.write = originalWrite + } catch (error) { + injectionTarget.__releaseScrollIntentTestWrite() + throw error } - const heldWrite = holder.write - if (!heldWrite) { - throw new Error('Foreground terminal write was not captured') - } - originalWrite.call(terminal, heldWrite.data, heldWrite.callback) }, paneKey) + try { + await page.keyboard.press('x') + } finally { + await page.evaluate(() => { + ;( + window as Window & { __releaseScrollIntentTestWrite?: () => void } + ).__releaseScrollIntentTestWrite?.() + }) + } } async function startStreamingFixturePhase1(page: Page): Promise { @@ -308,5 +315,6 @@ test.describe('terminal scroll intent keeps following output', () => { { timeout: 5_000, intervals: [25] } ) .toBe(0) + await waitForMarkerAtBottom(orcaPage, 'STREAM_PHASE2_DONE') }) }) diff --git a/tests/e2e/terminal-send-agent-prompt-submit.spec.ts b/tests/e2e/terminal-send-agent-prompt-submit.spec.ts index 3567cb1d8a2..c1a602dd9d1 100644 --- a/tests/e2e/terminal-send-agent-prompt-submit.spec.ts +++ b/tests/e2e/terminal-send-agent-prompt-submit.spec.ts @@ -58,6 +58,7 @@ async function createFakeCodexTerminal( if (!worktree) { throw new Error(`runtime did not register ${testRepoPath}`) } + rmSync(fixtureReport, { force: true }) const created = await client.call<{ terminal: { handle: string } }>('terminal.create', { worktree: `id:${worktree.id}`, command: [fakeCodexCommand, ...args].join(' '), diff --git a/tests/tools/repro-terminal-send-submit.mjs b/tests/tools/repro-terminal-send-submit.mjs index 7e1c0152012..41d2464cbbc 100644 --- a/tests/tools/repro-terminal-send-submit.mjs +++ b/tests/tools/repro-terminal-send-submit.mjs @@ -186,7 +186,9 @@ async function parentMain() { const expectBlocked = hasFlag('expect-blocked') const providedHandle = argValue('terminal') await mkdir(tempDir, { recursive: true }) - await rm(reportPath, { force: true }) + if (!providedHandle) { + await rm(reportPath, { force: true }) + } let handle = providedHandle if (!handle) { From ab8e10e298df8be5b3294553d593a1420349c662 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:43:41 -0700 Subject: [PATCH 06/17] test: isolate skill cloud fixture ports across workers (#18942) --- .../e2e/helpers/remote-skill-cloud-fixture.ts | 41 ++++++++++++------- .../remote-skill-cloud-fixture.unit.test.ts | 41 +++++++++++++++++++ tests/e2e/paired-skill-installation.spec.ts | 8 ++-- tests/e2e/ssh-skill-installation.spec.ts | 21 ++++++---- 4 files changed, 85 insertions(+), 26 deletions(-) create mode 100644 tests/e2e/helpers/remote-skill-cloud-fixture.unit.test.ts diff --git a/tests/e2e/helpers/remote-skill-cloud-fixture.ts b/tests/e2e/helpers/remote-skill-cloud-fixture.ts index f1d28d926b7..8e1650947dd 100644 --- a/tests/e2e/helpers/remote-skill-cloud-fixture.ts +++ b/tests/e2e/helpers/remote-skill-cloud-fixture.ts @@ -8,13 +8,12 @@ import { } from '../../../src/main/skills/skill-package-creation' import { SKILL_PACKAGE_CONTENT_TYPE } from '../../../src/shared/skill-package-manifest' -export const REMOTE_SKILL_CLOUD_PORT = Number(process.env.ORCA_E2E_SKILL_CLOUD_PORT ?? '43961') -export const REMOTE_SKILL_CLOUD_ORIGIN = `http://127.0.0.1:${REMOTE_SKILL_CLOUD_PORT}` export const REMOTE_SKILL_PACKAGE_ID = 'package_remote_e2e' export const REMOTE_SKILL_VERSION_ID = 'version_remote_e2e' export const REMOTE_SKILL_NAME = 'remote-e2e-skill' export type RemoteSkillCloudFixture = { + origin: string archive: CreatedSkillPackage bytes: Buffer requests: { method: string; path: string; body: unknown }[] @@ -39,19 +38,30 @@ export async function startRemoteSkillCloudFixture(): Promise { - void handleRemoteSkillCloudRequest({ request, response, archive, bytes, requests }).catch( - (error) => { - response.writeHead(500, { 'content-type': 'application/json' }) - response.end(JSON.stringify({ code: 'fixture_failed', message: String(error) })) - } - ) + void handleRemoteSkillCloudRequest({ + request, + response, + archive, + bytes, + requests, + origin + }).catch((error) => { + response.writeHead(500, { 'content-type': 'application/json' }) + response.end(JSON.stringify({ code: 'fixture_failed', message: String(error) })) + }) }) await new Promise((resolve, reject) => { server.once('error', reject) - server.listen(REMOTE_SKILL_CLOUD_PORT, '127.0.0.1', resolve) + server.listen(Number(process.env.ORCA_E2E_SKILL_CLOUD_PORT ?? 0), '127.0.0.1', resolve) }) - return { archive, bytes, requests, root, server } + const address = server.address() + if (!address || typeof address === 'string') { + throw new Error('Skill fixture has no TCP address') + } + origin = `http://127.0.0.1:${address.port}` + return { archive, bytes, requests, root, server, origin } } export async function stopRemoteSkillCloudFixture(fixture: RemoteSkillCloudFixture): Promise { @@ -60,13 +70,14 @@ export async function stopRemoteSkillCloudFixture(fixture: RemoteSkillCloudFixtu } async function handleRemoteSkillCloudRequest(input: { + origin: string request: IncomingMessage response: ServerResponse archive: CreatedSkillPackage bytes: Buffer requests: RemoteSkillCloudFixture['requests'] }): Promise { - const path = new URL(input.request.url ?? '/', REMOTE_SKILL_CLOUD_ORIGIN).pathname + const path = new URL(input.request.url ?? '/', input.origin).pathname if (input.request.method === 'GET' && path === '/package.tar.gz') { input.requests.push({ method: 'GET', path, body: null }) input.response.writeHead(200, { @@ -84,17 +95,19 @@ async function handleRemoteSkillCloudRequest(input: { const body = JSON.parse(await readRequestBody(input.request)) as unknown input.requests.push({ method: 'POST', path, body }) input.response.writeHead(200, { 'content-type': 'application/json' }) - input.response.end(JSON.stringify(downloadGrant(input.archive, input.bytes.length))) + input.response.end( + JSON.stringify(downloadGrant(input.archive, input.bytes.length, input.origin)) + ) return } input.response.writeHead(404, { 'content-type': 'application/json' }) input.response.end(JSON.stringify({ code: 'not_found', message: 'Not found' })) } -function downloadGrant(archive: CreatedSkillPackage, compressedBytes: number) { +function downloadGrant(archive: CreatedSkillPackage, compressedBytes: number, origin: string) { return { grant: { - url: `${REMOTE_SKILL_CLOUD_ORIGIN}/package.tar.gz`, + url: `${origin}/package.tar.gz`, expiresAt: '2099-01-01T00:00:00.000Z' }, version: { diff --git a/tests/e2e/helpers/remote-skill-cloud-fixture.unit.test.ts b/tests/e2e/helpers/remote-skill-cloud-fixture.unit.test.ts new file mode 100644 index 00000000000..70291cf0e8f --- /dev/null +++ b/tests/e2e/helpers/remote-skill-cloud-fixture.unit.test.ts @@ -0,0 +1,41 @@ +import { expect, it, vi } from 'vitest' +import { + REMOTE_SKILL_PACKAGE_ID, + REMOTE_SKILL_VERSION_ID, + startRemoteSkillCloudFixture, + stopRemoteSkillCloudFixture +} from './remote-skill-cloud-fixture' + +it('serves concurrent skill fixtures from independent bound origins', async () => { + vi.stubEnv('ORCA_E2E_SKILL_CLOUD_PORT', undefined) + const results = await Promise.allSettled([ + startRemoteSkillCloudFixture(), + startRemoteSkillCloudFixture() + ]) + const fixtures = results.flatMap((result) => + result.status === 'fulfilled' ? [result.value] : [] + ) + try { + expect(results.every((result) => result.status === 'fulfilled')).toBe(true) + expect(new Set(fixtures.map((fixture) => fixture.origin)).size).toBe(2) + for (const fixture of fixtures) { + const response = await fetch( + `${fixture.origin}/v1/skill-packages/${REMOTE_SKILL_PACKAGE_ID}/versions/${REMOTE_SKILL_VERSION_ID}/download-grants`, + { + method: 'POST', + body: '{}', + headers: { 'content-type': 'application/json' } + } + ) + expect(response.status).toBe(200) + const result = (await response.json()) as { grant: { url: string } } + expect(result.grant.url).toBe(`${fixture.origin}/package.tar.gz`) + const archive = await fetch(result.grant.url) + expect(Buffer.from(await archive.arrayBuffer())).toEqual(fixture.bytes) + expect(fixture.requests).toHaveLength(2) + } + } finally { + await Promise.all(fixtures.map(stopRemoteSkillCloudFixture)) + vi.unstubAllEnvs() + } +}) diff --git a/tests/e2e/paired-skill-installation.spec.ts b/tests/e2e/paired-skill-installation.spec.ts index ac31a81e2f1..0268dbb84a3 100644 --- a/tests/e2e/paired-skill-installation.spec.ts +++ b/tests/e2e/paired-skill-installation.spec.ts @@ -14,7 +14,6 @@ import { type HeadlessPairedRuntimeHost } from './helpers/headless-paired-runtime-host' import { - REMOTE_SKILL_CLOUD_ORIGIN, REMOTE_SKILL_NAME, REMOTE_SKILL_PACKAGE_ID, REMOTE_SKILL_VERSION_ID, @@ -119,13 +118,14 @@ test('installs on a headless serve runtime through the same contract', async ({ }) function cloudClientEnvironment(): Record { + const { origin } = requireCloudFixture() return { - ORCA_ARTIFACTS_API_URL: REMOTE_SKILL_CLOUD_ORIGIN, - ORCA_CLOUD_API_URL: REMOTE_SKILL_CLOUD_ORIGIN, + ORCA_ARTIFACTS_API_URL: origin, + ORCA_CLOUD_API_URL: origin, ORCA_CLOUD_CLIENT_ID: 'skills-e2e-client', ORCA_CLOUD_DEV_AUTH: '1', ORCA_CLOUD_ALLOW_PLAINTEXT_SESSION: '1', - ORCA_SKILL_PACKAGE_DOWNLOAD_ORIGINS: REMOTE_SKILL_CLOUD_ORIGIN + ORCA_SKILL_PACKAGE_DOWNLOAD_ORIGINS: origin } } diff --git a/tests/e2e/ssh-skill-installation.spec.ts b/tests/e2e/ssh-skill-installation.spec.ts index a102478fb62..794883b2cd1 100644 --- a/tests/e2e/ssh-skill-installation.spec.ts +++ b/tests/e2e/ssh-skill-installation.spec.ts @@ -10,7 +10,6 @@ import { import { connectDockerSshRelayTarget } from './helpers/docker-ssh-relay-connection' import { waitForActiveWorktree, waitForSessionReady } from './helpers/store' import { - REMOTE_SKILL_CLOUD_ORIGIN, REMOTE_SKILL_NAME, REMOTE_SKILL_PACKAGE_ID, REMOTE_SKILL_VERSION_ID, @@ -25,13 +24,19 @@ const REMOTE_FOLDER = '/tmp/orca-skill-folder-workspace' let cloud: RemoteSkillCloudFixture | null = null test.use({ - orcaAppExtraEnv: { - ORCA_ARTIFACTS_API_URL: REMOTE_SKILL_CLOUD_ORIGIN, - ORCA_CLOUD_API_URL: REMOTE_SKILL_CLOUD_ORIGIN, - ORCA_CLOUD_CLIENT_ID: 'skills-e2e-client', - ORCA_CLOUD_DEV_AUTH: '1', - ORCA_CLOUD_ALLOW_PLAINTEXT_SESSION: '1', - ORCA_SKILL_PACKAGE_DOWNLOAD_ORIGINS: REMOTE_SKILL_CLOUD_ORIGIN + // oxlint-disable-next-line no-empty-pattern -- The server starts in beforeAll before this test fixture runs. + orcaAppExtraEnv: async ({}, provideEnv) => { + if (!cloud) { + throw new Error('Skill cloud fixture unavailable') + } + await provideEnv({ + ORCA_ARTIFACTS_API_URL: cloud.origin, + ORCA_CLOUD_API_URL: cloud.origin, + ORCA_CLOUD_CLIENT_ID: 'skills-e2e-client', + ORCA_CLOUD_DEV_AUTH: '1', + ORCA_CLOUD_ALLOW_PLAINTEXT_SESSION: '1', + ORCA_SKILL_PACKAGE_DOWNLOAD_ORIGINS: cloud.origin + }) } }) From 22a7bfd3804717898e82b30ddbf880c5511b8c5f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:45:48 -0700 Subject: [PATCH 07/17] test: align Source Control AI fixtures with current settings (#18941) --- tests/e2e/helpers/source-control-ai-generation.ts | 15 +++++++++++++-- tests/e2e/helpers/source-control-ai-generators.ts | 6 +++--- 2 files changed, 16 insertions(+), 5 deletions(-) diff --git a/tests/e2e/helpers/source-control-ai-generation.ts b/tests/e2e/helpers/source-control-ai-generation.ts index f6be16d7342..c93a20e37a1 100644 --- a/tests/e2e/helpers/source-control-ai-generation.ts +++ b/tests/e2e/helpers/source-control-ai-generation.ts @@ -67,7 +67,7 @@ export async function seedCreatePrComposer(page: Page): Promise<{ prWorktreePath: string primaryBranch: string }> { - return page.evaluate(async () => { + const seeded = await page.evaluate(async () => { const store = window.__store ?? (() => { @@ -101,6 +101,7 @@ export async function seedCreatePrComposer(page: Page): Promise<{ const eligibility = { provider: 'github' as const, review: null, + reviewLookupOutcome: 'not_found' as const, canCreate: true, blockedReason: null, nextAction: null, @@ -121,7 +122,7 @@ export async function seedCreatePrComposer(page: Page): Promise<{ ...current.remoteStatusesByWorktree, [prWorktree.id]: { hasUpstream: true, - upstreamName: `origin/${branch}`, + upstreamName: primaryBranch, ahead: 0, behind: 0 } @@ -130,6 +131,10 @@ export async function seedCreatePrComposer(page: Page): Promise<{ args.branch === branch ? eligibility : { ...eligibility, canCreate: false }, fetchHostedReviewForBranch: async () => null, fetchPRForBranch: async () => null, + enqueueGitHubPRRefresh: () => undefined, + // Ignore provider work queued before this generation-only fixture was installed. + getEffectiveGitHubPRRefreshState: () => undefined, + prRefreshStates: {}, fetchUpstreamStatus: async () => undefined, setUpstreamStatus: () => undefined })) @@ -141,6 +146,12 @@ export async function seedCreatePrComposer(page: Page): Promise<{ primaryBranch } }) + // Checks reads fresh Git state instead of the seeded store cache. + execFileSync('git', ['branch', '--set-upstream-to', seeded.primaryBranch], { + cwd: seeded.prWorktreePath, + stdio: 'pipe' + }) + return seeded } export async function seedCommitMessageComposer(page: Page): Promise<{ diff --git a/tests/e2e/helpers/source-control-ai-generators.ts b/tests/e2e/helpers/source-control-ai-generators.ts index be3f1b43247..8c09bb8b556 100644 --- a/tests/e2e/helpers/source-control-ai-generators.ts +++ b/tests/e2e/helpers/source-control-ai-generators.ts @@ -14,13 +14,13 @@ async function setCustomGenerator(page: Page, scriptPath: string): Promise } await store.getState().updateSettings({ activeRuntimeEnvironmentId: null, - commitMessageAi: { - ...currentSettings.commitMessageAi, + sourceControlAi: { enabled: true, agentId: 'custom' as const, selectedModelByAgent: {}, selectedThinkingByModel: {}, - customPrompt: '', + instructionsByOperation: {}, + actions: {}, customAgentCommand: `node ${JSON.stringify(scriptPath)}` } }) From 7bb54cc2f73c08a3df026c28766afd48b0e24471 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:56:57 -0700 Subject: [PATCH 08/17] ci: reduce runner overhead and disposable package compression (#18948) * ci: reduce PR runner overhead and package compression time * ci: validate mobile when its dependency action changes --- .github/workflows/mobile.yml | 27 +--- .github/workflows/pr.yml | 108 ++++++-------- .github/workflows/skill-update-roundtrip.yml | 4 + config/scripts/pr-code-change-scope.test.mjs | 7 +- config/scripts/pr-e2e-gate-contract.test.mjs | 30 ++-- docs/reference/ci-runner-efficiency.md | 99 +++++++++++++ docs/reference/windows-signing-runner-time.md | 137 ++++++++++++++++++ 7 files changed, 311 insertions(+), 101 deletions(-) create mode 100644 docs/reference/ci-runner-efficiency.md create mode 100644 docs/reference/windows-signing-runner-time.md diff --git a/.github/workflows/mobile.yml b/.github/workflows/mobile.yml index 59f6cf20bf4..6dbfc02aa3c 100644 --- a/.github/workflows/mobile.yml +++ b/.github/workflows/mobile.yml @@ -15,8 +15,13 @@ on: # Why: this job holds the only checks that load the Fastfile, so edits to # it or to the release workflow it guards must re-run them. - '.github/workflows/mobile.yml' + - '.github/actions/install-node-dependencies/**' - '.github/workflows/mobile-ios-release.yml' +concurrency: + group: mobile-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + jobs: verify: runs-on: ubuntu-latest @@ -35,10 +40,7 @@ jobs: - name: Checkout uses: actions/checkout@v6 - - name: Setup Node.js - uses: actions/setup-node@v6 - with: - node-version-file: package.json + - uses: ./.github/actions/install-node-dependencies # bundler-cache installs mobile/Gemfile.lock, so this job is also what # proves the pinned fastlane the release workflow depends on still @@ -50,23 +52,6 @@ jobs: bundler-cache: true working-directory: mobile - - name: Setup pnpm - uses: pnpm/setup@v2 - with: - install: false - - # Why: the mobile typecheck imports shared types from ../src/shared, and - # some of those files import runtime deps (tweetnacl, ws) resolved from - # the repo-root node_modules. Without a root install, tsc fails with - # "Cannot find module 'tweetnacl'/'ws'". Mobile is a separate pnpm project - # (not in the root workspace), so this is a distinct install. - # --ignore-scripts skips the root postinstall (Electron native-module - # rebuild) which is irrelevant to a type-only check and would only add - # time and failure surface on this ubuntu mobile runner. - - name: Install root dependencies - working-directory: . - run: pnpm install --frozen-lockfile --ignore-scripts - - name: Install dependencies run: pnpm install --frozen-lockfile diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index a749214e232..ba2eaf83192 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -41,6 +41,10 @@ jobs: managed_hook_node18: ${{ steps.filter.outputs.managed_hook_node18 }} package: ${{ steps.filter.outputs.package }} package_windows: ${{ steps.filter.outputs.package_windows }} + e2e_should_run: ${{ steps.e2e_filter.outputs.should_run }} + test_files: ${{ steps.e2e_filter.outputs.test_files }} + ssh_source_changed: ${{ steps.e2e_filter.outputs.ssh_source_changed }} + native_ime_source_changed: ${{ steps.e2e_filter.outputs.native_ime_source_changed }} steps: - name: Checkout uses: actions/checkout@v6 @@ -66,6 +70,37 @@ jobs: printf '%s\n' "$CHANGED" printf '%s\n' "$CHANGED" | node config/scripts/pr-code-change-scope.mjs | tee -a "$GITHUB_OUTPUT" + # Reuse the path-detector checkout instead of queuing another runner. + - name: Filter changed E2E specs + id: e2e_filter + if: github.event.pull_request.draft != true && steps.filter.outputs.should_run == 'true' + run: | + set -euo pipefail + BASE="${{ github.event.pull_request.base.sha }}" + HEAD="${{ github.event.pull_request.head.sha }}" + CHANGED="$(git diff --name-only --diff-filter=AMCR --merge-base "$BASE" "$HEAD")" + # Source routes are executable contracts so a test can prove exact + # authorities, exclusions, and sentinels without evaluating workflow shell. + TEST_FILES_JSON="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs)" + echo "test_files=$TEST_FILES_JSON" >> "$GITHUB_OUTPUT" + # Why a separate signal: the Docker-SSH lane must trigger on SSH source, not on a + # spec name surviving in a route's list. Same routes, so the two cannot drift. + SSH_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --ssh-source)" + echo "ssh_source_changed=$SSH_SOURCE_CHANGED" >> "$GITHUB_OUTPUT" + echo "SSH source changed: $SSH_SOURCE_CHANGED" + # Why its own signal: the real-IME lane is a whole ibus session, not a spec, so it must + # trigger on IME source rather than on a spec name in some route's list. + NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)" + echo "native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED" >> "$GITHUB_OUTPUT" + echo "Native IME source changed: $NATIVE_IME_SOURCE_CHANGED" + if [ "$TEST_FILES_JSON" != '[]' ]; then + echo "should_run=true" >> "$GITHUB_OUTPUT" + echo "Changed E2E specs: $TEST_FILES_JSON" + else + echo "should_run=false" >> "$GITHUB_OUTPUT" + echo "No changed E2E specs" + fi + static_analysis: name: static analysis needs: [code_paths] @@ -712,7 +747,11 @@ jobs: - name: Package unpacked app env: ORCA_REUSE_PREPARED_NATIVE_RUNTIME: '1' - run: pnpm exec electron-builder --config config/electron-builder.config.cjs --linux AppImage deb rpm --x64 --publish never + # PR artifacts are only inspected locally; gzip avoids release-size xz compression. + run: >- + pnpm exec electron-builder --config config/electron-builder.config.cjs + --linux AppImage deb rpm --x64 --publish never + --config.deb.compression=gz --config.rpm.compression=gzip - name: Verify root-package marker payloads run: | @@ -861,65 +900,10 @@ jobs: - name: Smoke packaged CLI run: node config/scripts/smoke-packaged-cli.mjs --app-dir=dist/win-unpacked - # Why: PR E2E is advisory and only validates changed specs; scheduled and - # release runs retain full-suite coverage. - e2e-paths: - name: detect changed e2e specs - needs: [code_paths] - runs-on: ubuntu-latest - if: github.event.pull_request.draft != true && needs.code_paths.outputs.should_run == 'true' - # Why: detector only needs to read the checkout; do not inherit repo defaults. - permissions: - contents: read - outputs: - should_run: ${{ steps.filter.outputs.should_run }} - test_files: ${{ steps.filter.outputs.test_files }} - ssh_source_changed: ${{ steps.filter.outputs.ssh_source_changed }} - native_ime_source_changed: ${{ steps.filter.outputs.native_ime_source_changed }} - steps: - - name: Checkout - uses: actions/checkout@v6 - with: - # Why blob:none: full history is needed for the merge-base diff, but historical - # file contents are not. Blobs are ~89% of this repo's pack, and Git fetches the - # few this job actually reads on demand. - fetch-depth: 0 - filter: blob:none - persist-credentials: false - - - name: Filter changed E2E specs - id: filter - run: | - set -euo pipefail - BASE="${{ github.event.pull_request.base.sha }}" - HEAD="${{ github.event.pull_request.head.sha }}" - CHANGED="$(git diff --name-only --diff-filter=AMCR --merge-base "$BASE" "$HEAD")" - # Source routes are executable contracts so a test can prove exact - # authorities, exclusions, and sentinels without evaluating workflow shell. - TEST_FILES_JSON="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs)" - echo "test_files=$TEST_FILES_JSON" >> "$GITHUB_OUTPUT" - # Why a separate signal: the Docker-SSH lane must trigger on SSH source, not on a - # spec name surviving in a route's list. Same routes, so the two cannot drift. - SSH_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --ssh-source)" - echo "ssh_source_changed=$SSH_SOURCE_CHANGED" >> "$GITHUB_OUTPUT" - echo "SSH source changed: $SSH_SOURCE_CHANGED" - # Why its own signal: the real-IME lane is a whole ibus session, not a spec, so it must - # trigger on IME source rather than on a spec name in some route's list. - NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)" - echo "native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED" >> "$GITHUB_OUTPUT" - echo "Native IME source changed: $NATIVE_IME_SOURCE_CHANGED" - if [ "$TEST_FILES_JSON" != '[]' ]; then - echo "should_run=true" >> "$GITHUB_OUTPUT" - echo "Changed E2E specs: $TEST_FILES_JSON" - else - echo "should_run=false" >> "$GITHUB_OUTPUT" - echo "No changed E2E specs" - fi - e2e: name: e2e - needs: e2e-paths - if: needs.e2e-paths.outputs.should_run == 'true' + needs: code_paths + if: needs.code_paths.outputs.e2e_should_run == 'true' # Why: reusable e2e.yml only checkouts, builds, and uploads artifacts. permissions: contents: read @@ -928,8 +912,8 @@ jobs: # The synthetic pull-request merge ref can disappear while this reusable # workflow is queued. The head SHA is immutable and works for every PR. ref: ${{ github.event.pull_request.head.sha }} - test_files: ${{ needs.e2e-paths.outputs.test_files }} - ssh_source_changed: ${{ needs.e2e-paths.outputs.ssh_source_changed }} + test_files: ${{ needs.code_paths.outputs.test_files }} + ssh_source_changed: ${{ needs.code_paths.outputs.ssh_source_changed }} # Why this is not in verify's needs: it is the first PR-gate run of a harness whose reliability # is only known from nightly main runs (20/20 green, 2026-08-09..2026-08-29, p50 3m25s). It @@ -939,8 +923,8 @@ jobs: # require `success || skipped` outside the strict loop — see the note on `e2e`. terminal_ime_native: name: real IME - needs: e2e-paths - if: needs.e2e-paths.outputs.native_ime_source_changed == 'true' + needs: code_paths + if: needs.code_paths.outputs.native_ime_source_changed == 'true' # Why: the reusable workflow only checks out, builds, and uploads artifacts. permissions: contents: read diff --git a/.github/workflows/skill-update-roundtrip.yml b/.github/workflows/skill-update-roundtrip.yml index 71fcf264f69..239f1b2f27c 100644 --- a/.github/workflows/skill-update-roundtrip.yml +++ b/.github/workflows/skill-update-roundtrip.yml @@ -22,6 +22,10 @@ on: - main paths: *skill-roundtrip-paths +concurrency: + group: skill-roundtrip-${{ github.event_name }}-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + jobs: roundtrip: strategy: diff --git a/config/scripts/pr-code-change-scope.test.mjs b/config/scripts/pr-code-change-scope.test.mjs index 4642372135c..f31822e5b93 100644 --- a/config/scripts/pr-code-change-scope.test.mjs +++ b/config/scripts/pr-code-change-scope.test.mjs @@ -414,10 +414,11 @@ describe('PR Checks skip wiring', () => { }) it('skips e2e detection on docs-only PRs without dropping the draft gate', () => { - expect(prWorkflow.jobs['e2e-paths'].needs).toEqual(['code_paths']) - expect(prWorkflow.jobs['e2e-paths'].if).toBe( - "github.event.pull_request.draft != true && needs.code_paths.outputs.should_run == 'true'" + const filter = prWorkflow.jobs.code_paths.steps.find((step) => step.id === 'e2e_filter') + expect(filter.if).toBe( + "github.event.pull_request.draft != true && steps.filter.outputs.should_run == 'true'" ) + expect(prWorkflow.jobs['e2e-paths']).toBeUndefined() }) it('lets verify pass skipped jobs the classifier turned off', () => { diff --git a/config/scripts/pr-e2e-gate-contract.test.mjs b/config/scripts/pr-e2e-gate-contract.test.mjs index ceac6b8cc6e..67e271868df 100644 --- a/config/scripts/pr-e2e-gate-contract.test.mjs +++ b/config/scripts/pr-e2e-gate-contract.test.mjs @@ -39,7 +39,7 @@ const nativeImeSpec = readFileSync( 'utf8' ) -const filterStep = prWorkflow.jobs['e2e-paths'].steps.find( +const filterStep = prWorkflow.jobs.code_paths.steps.find( (step) => step.name === 'Filter changed E2E specs' ) const rollbackStep = prWorkflow.jobs.static_analysis.steps.find( @@ -106,16 +106,16 @@ describe('PR E2E gate contract', () => { // Why: without this the job could lose its filter and run on every PR — the // cost the path filter exists to avoid — while the gate assertions above // stay green. - expect(prWorkflow.jobs.e2e.needs).toBe('e2e-paths') - expect(prWorkflow.jobs.e2e.if).toBe("needs.e2e-paths.outputs.should_run == 'true'") - expect(prWorkflow.jobs['e2e-paths'].outputs.should_run).toBe( - '${{ steps.filter.outputs.should_run }}' + expect(prWorkflow.jobs.e2e.needs).toBe('code_paths') + expect(prWorkflow.jobs.e2e.if).toBe("needs.code_paths.outputs.e2e_should_run == 'true'") + expect(prWorkflow.jobs.code_paths.outputs.e2e_should_run).toBe( + '${{ steps.e2e_filter.outputs.should_run }}' ) - expect(prWorkflow.jobs['e2e-paths'].outputs.test_files).toBe( - '${{ steps.filter.outputs.test_files }}' + expect(prWorkflow.jobs.code_paths.outputs.test_files).toBe( + '${{ steps.e2e_filter.outputs.test_files }}' ) expect(prWorkflow.jobs.e2e.with.ref).toBe('${{ github.event.pull_request.head.sha }}') - expect(prWorkflow.jobs.e2e.with.test_files).toBe('${{ needs.e2e-paths.outputs.test_files }}') + expect(prWorkflow.jobs.e2e.with.test_files).toBe('${{ needs.code_paths.outputs.test_files }}') }) it('enforces every job verify depends on', () => { @@ -360,11 +360,11 @@ describe('PR E2E gate contract', () => { expect(sshLaneCondition).toContain("inputs.ssh_source_changed == 'true' ||") expect(e2eWorkflow.on.workflow_call.inputs.ssh_source_changed.type).toBe('string') - expect(prWorkflow.jobs['e2e-paths'].outputs.ssh_source_changed).toBe( - '${{ steps.filter.outputs.ssh_source_changed }}' + expect(prWorkflow.jobs.code_paths.outputs.ssh_source_changed).toBe( + '${{ steps.e2e_filter.outputs.ssh_source_changed }}' ) expect(prWorkflow.jobs.e2e.with.ssh_source_changed).toBe( - '${{ needs.e2e-paths.outputs.ssh_source_changed }}' + '${{ needs.code_paths.outputs.ssh_source_changed }}' ) expect(filterStep.run).toContain('pr-e2e-source-routing.mjs --ssh-source') expect(filterStep.run).toContain('ssh_source_changed=$SSH_SOURCE_CHANGED') @@ -565,12 +565,12 @@ describe('PR E2E gate contract', () => { expect(prWorkflow.jobs.terminal_ime_native.uses).toBe( './.github/workflows/terminal-ime-e2e.yml' ) - expect(prWorkflow.jobs.terminal_ime_native.needs).toBe('e2e-paths') + expect(prWorkflow.jobs.terminal_ime_native.needs).toBe('code_paths') expect(prWorkflow.jobs.terminal_ime_native.if).toBe( - "needs.e2e-paths.outputs.native_ime_source_changed == 'true'" + "needs.code_paths.outputs.native_ime_source_changed == 'true'" ) - expect(prWorkflow.jobs['e2e-paths'].outputs.native_ime_source_changed).toBe( - '${{ steps.filter.outputs.native_ime_source_changed }}' + expect(prWorkflow.jobs.code_paths.outputs.native_ime_source_changed).toBe( + '${{ steps.e2e_filter.outputs.native_ime_source_changed }}' ) expect(filterStep.run).toContain('pr-e2e-source-routing.mjs --native-ime-source') expect(filterStep.run).toContain('native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED') diff --git a/docs/reference/ci-runner-efficiency.md b/docs/reference/ci-runner-efficiency.md new file mode 100644 index 00000000000..e569f749102 --- /dev/null +++ b/docs/reference/ci-runner-efficiency.md @@ -0,0 +1,99 @@ +# CI efficiency and runner capacity + +Audit date: September 5, 2026. No paid capacity or provider configuration changed. + +## Measurements and changes + +Three recent successful PR runs used 54.6–64.9 aggregate runner minutes: +[33998366568](https://github.com/stablyai/orca/actions/runs/33998366568), +[33998220287](https://github.com/stablyai/orca/actions/runs/33998220287), and +[33998181502](https://github.com/stablyai/orca/actions/runs/33998181502). +These are sums of active job durations, excluding skipped jobs; they are not +billing minutes or queue time. This small sample is not a historical average. + +- Consolidate E2E routing into the existing code-path detector. The removed + detector occupied 20–22 seconds and required another runner allocation and + full-history checkout per nondraft code PR. The same routing commands remain, + including SSH and native IME selection; actual E2E results remain advisory. + A routing-script error now fails the required code-path detector. +- Use gzip for PR-only Debian/RPM artifacts. The two sampled Linux packaging + jobs took 8m10s and 8m19s overall; one spent 3m47s in electron-builder. Its + default Debian/RPM compression is xz. PR artifacts are inspected on the same + runner, so their download size offers no benefit. Keep all AppImage, Debian, + RPM, payload, launcher, and shutdown checks. Release compression is unchanged. + Compression savings need a hosted run; do not equate the full packaging step + with removable compression time. +- Cancel superseded Mobile Checks and Skill update round-trip PR runs. The + skill matrix has 13 jobs. Preserve non-cancelling main/merge-group skill runs, + with separate concurrency groups per event. +- Reuse the existing script-free root dependency action in Mobile Checks, + including the pnpm cache keyed by both root and mobile lockfiles. The root + install remains necessary because mobile types import root dependencies. + +The repository already has eight unit shards, path-scoped platform checks, +native caches, one shared E2E build, PR cancellation, incremental TypeScript +caching, and changed-spec E2E routing. Increasing shards would increase setup +work and simultaneous runner demand. Do not adjust the count without comparing +critical-path time and aggregate job time on the same commit. + +## Runner recommendations + +The repository is **public**, verified using the GitHub API. Standard +GitHub-hosted Linux, Windows, and macOS runners have free compute minutes for +public repositories. Queue pressure and third-party provider allowances still +matter; artifact storage and larger runners have separate billing rules. +See [GitHub Actions billing](https://docs.github.com/en/billing/concepts/product-billing/github-actions). + +1. Keep standard GitHub-hosted runners as the default. Ask GitHub Support for a + higher concurrent-job limit before paying for more capacity. The documented + standard limits depend on the account plan (Free: 20 total/5 macOS; Team: + 60/5; Enterprise: 500/50), and increases are subject to approval. The actual + account entitlement was not verified. See [limits](https://docs.github.com/en/actions/reference/limits). +2. Reserve existing Blacksmith allowance for macOS if that is the priority. + Blacksmith documents 3,000 free x64 2-vCPU-equivalent minutes per organization; + a 6-vCPU Mac minute consumes 20 equivalents, or 150 actual Mac minutes if + it uses the entire free pool. Cloud workflows also use Blacksmith Linux. + Moving Linux to hosted GitHub saves shared allowance, but does not necessarily + free Mac hardware capacity. Account-specific contracts and usage were not + inspected. See [Blacksmith runners](https://docs.blacksmith.sh/blacksmith-runners/overview). +3. Treat Ubicloud as an optional small Linux overflow trial. Its documented + $2.50 monthly credit buys 1,250 premium 2-vCPU minutes at $0.002/minute, or + 2,000 standard 2-vCPU minutes at $0.00125/minute. New accounts default to + premium and require a credit card. No enforceable hard spending cap was + verified, so changing runner labels cannot guarantee the no-spend constraint. + One PR's roughly 55–65 runner minutes also makes clear how small this pool + is relative to repository activity (hardware speeds differ). + See [pricing](https://ubicloud.com/docs/about/pricing) and + [setup](https://ubicloud.com/docs/github-actions-integration/quickstart). + +## Machines that also run coding agents + +Do not register the credentialed host directly as a public-PR runner. A PR can +execute arbitrary build/test code, and a persistent host lets it access local +credentials or affect subsequent jobs. Docker alone is not adequate isolation +when it exposes the host home, Docker socket, SSH agent, or office network. + +A possible no-new-hardware experiment is a disposable VM per job, preferably on +a dedicated spare machine, with a just-in-time single-job runner, no shared +home/keychain/SSH agent or host mounts, restricted network access, and CPU/RAM +limits that leave room for coding agents. Destroy the VM after every job; +ephemeral runner registration by itself does not clean the machine. Start with +trusted branch/manual workloads and keep public fork PRs on hosted runners. +Provisioning and ongoing patching are real operational costs even when the +machine is already owned. See GitHub's +[self-hosted runner security guidance](https://docs.github.com/en/actions/security-for-github-actions/security-guides/security-hardening-for-github-actions). + +## Release waits + +The latest successful sampled Windows release used 13m59s of a 21m56s job in +signing wait/download steps. The same release held an Ubuntu job for 11m38s +polling the isolated Mac build. These are stronger occupancy opportunities than +small checkout savings, especially when approval takes hours. + +[Windows signing without occupying a runner](windows-signing-runner-time.md) +describes a staged, same-run design, required protected environments, and +rehearsal criteria. No callback integration or protected Windows signing +environments currently exist. An environment-gated design adds a GitHub +approval after each SignPath approval and changes the current automatic inner +signing timeout fallback; those are explicit release-policy decisions, so this +PR leaves production signing behavior unchanged. diff --git a/docs/reference/windows-signing-runner-time.md b/docs/reference/windows-signing-runner-time.md new file mode 100644 index 00000000000..fb02ba6bdca --- /dev/null +++ b/docs/reference/windows-signing-runner-time.md @@ -0,0 +1,137 @@ +# Windows signing without occupying a runner during approval + +Status: implementation proposal; production signing behavior is unchanged. + +## Measured cost + +In [release run 33821033674](https://github.com/stablyai/orca/actions/runs/33821033674) +(September 4, 2026), the Windows job took 21m56s. The inner-binary download step +took 13m19s and the installer download step took 40s: 13m59s, or 64% of the job, +was spent in the signing download/wait steps. These durations include the +download itself, so they are an upper bound on removable idle time, not a +prediction of net savings after transferring state between jobs. + +`release-cut.yml` submits both requests with `wait-for-completion: false`, but +then invokes `Get-SignedArtifact` on the same Windows runner with one-hour and +four-hour completion timeouts. The six-hour job timeout accommodates both +waits. Changing the submission flag again, polling less often, or running the +wait inside a container does not release the runner slot. + +This is runner occupancy, not a billing estimate. Standard GitHub-hosted +runners in a public repository may be free; removing the waits still releases +concurrency for other work. Check actual billing before assigning dollar savings. + +The same release also occupied an Ubuntu runner for 11m38s while +`run-release-mac-build-workflow.mjs` waited on the isolated macOS workflow. +That is a separate orchestration optimization. Windows development-channel +builds deliberately ship unsigned and have no SignPath wait to remove. + +## Proposed execution graph + +Keep all Windows stages in the original `release-cut.yml` run to preserve the +current SignPath GitHub artifact provenance boundary: + +1. `build-windows` builds and uploads the unpacked app, original installer, + updater metadata, and inner-signing manifest. It submits the inner request, + sends the existing notification, exposes the request ID, and finishes. +2. `package-windows` depends on that job and uses a protected environment named + `windows-inner-signing`. Its runner is allocated only after GitHub approval. + It restores the exact build, downloads the signed binaries with a short, + bounded completion wait, applies the existing signature restoration and + signed `elevate.exe` cache replacement, builds the NSIS installer, uploads it, + submits the second signing request, notifies approvers, and finishes. +3. `finalize-windows` depends on packaging and uses a second protected environment + named `windows-installer-signing`. After approval it downloads the signed + installer, regenerates its blockmap and `latest.yml`, runs existing outer and + inner signature checks, uploads evidence, and uploads the assets to the draft. +4. `publish-release` depends on finalization as well as the existing Linux, macOS, + and blocking release gates. It remains the only job that publishes the draft. + +The approver signs in SignPath, waits for that request to finish, and then +approves the corresponding pending GitHub job. Each notification should link +to both places and explain the order. GitHub approval is an extra action; +approving in SignPath alone does not release an environment gate. + +## Required configuration + +The repository environments were inspected through the GitHub API on +September 5, 2026. Neither Windows environment exists. `adhoc-mac-build` has no +protection rules; it cannot be reused as an approval gate. No SignPath callback +handler was found in the repository's workflows, scripts, application, or cloud +code. + +Before enabling the graph: + +1. Create both environments in repository Settings → Environments. +2. Add the release approvers as required reviewers for each environment. Decide + whether a release initiator may approve their own job, and configure that + consistently with the existing SignPath policy. +3. Restrict deployment branches to the trusted refs used to dispatch release + workflows, and check that the release workflow's ref passes the restriction. + The workflow ref and the checked-out release tag are different concepts. +4. Read back both environments through the API and verify that + `required_reviewers` rules exist before changing the release graph. Merely + referring to a new environment name in YAML can create an unprotected + environment and silently leave the wait on the runner. +5. Add a preflight assertion for those rules so accidental removal fails before + any signing request is submitted. Verify the API access required for this + assertion using the release workflow's token; do not assume an administrator's + local `gh` access proves workflow-token access. + +An automatic alternative requires a SignPath completion callback and an +authenticated integration that releases the corresponding deployment gate. +Confirm the Foundation plan supports the necessary callback before choosing +that architecture. Do not introduce a long-running GitHub polling job as the +callback substitute: it would continue occupying a slot. + +## State and failure contracts + +- Use artifacts from this exact run and attempt, with a manifest containing the + tag, tag commit SHA, workflow SHA, request IDs, artifact IDs, and SHA-256 hashes. + Artifact names alone are insufficient. Preserve the original unsigned + installer for the existing inner-signing fallback. +- Restore `dist/win-unpacked`, the staging list, the installer, and updater + metadata as one checkpoint. Use an archive to preserve the tree. Do not ship + a fresh rebuild of the app after approving a different binary tree. +- Each new Windows runner needs the pinned Node/pnpm toolchain, build + dependencies, SignPath module, and electron-builder tool cache. The second + runner must populate the NSIS cache before replacing `elevate.exe`; the old + code assumes the first installer build already populated that cache. +- Retain checkout-from-tag behavior and the existing support for release tags + that predate the composite action. Explicitly restore new orchestration code + from the workflow SHA when necessary. +- Preserve the rule that rerunning a workflow never submits a new signing + request. A resume must consume the recorded request and artifacts. Test failed + stage reruns, whole-workflow reruns, and missing/expired checkpoints separately. +- Keep installer signature checks blocking. Keep inner verification evidence + and its current warning-only policy unless changed in a separate decision. +- Resolve the current one-hour inner-signing fallback deliberately: an + environment approval can remain pending longer than one hour and rejection + skips dependent jobs. It cannot reproduce the existing automatic timeout + fallback by itself. A first migration should explicitly document the new + manual release/cancellation behavior; silently treating rejected approval as + permission to ship is not acceptable. +- Keep the release-wide concurrency lock while the graph waits, preventing + another release from overtaking this draft. This saves worker occupancy, but + does not shorten the serialized release queue's human approval time. + +## Validation before production + +First adapt `windows-signing-rehearsal.yml` to exercise the same staged code +using the auto-approved test-signing policy. Then run a manual rehearsal with +the protected environments and confirm that pending approval has no allocated +Windows runner. Verify signed bytes through the existing extraction-based +installer checks, not only the outer installer signature. + +Cover approval before SignPath completion, rejected approval, missing signed +files, changed checkpoint hashes, lost checkpoints, expired artifacts, failed +packaging, and stage reruns without duplicate submissions. Confirm no release +becomes public until all platform and signature gates pass. Compare transferred +artifact/setup time with the original 13m59s wait sample to measure net savings. + +A separate `workflow_dispatch` continuation can avoid environment provisioning, +but changes this design substantially: the original release run finishes, +workflow-level concurrency no longer protects the pending draft, and SignPath +must accept artifacts assembled from a prior run. That option needs a durable +release state machine and provenance validation before production use; it is +not a drop-in replacement for the two download steps. From 71f2c5d3f9bd29d13c93c43b6a09105648001cea Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 17:15:57 -0700 Subject: [PATCH 09/17] test: keep artifact share fixtures unexpired across calendar dates (#18955) --- src/main/artifacts/artifact-cloud-recovery.test.ts | 2 +- src/main/artifacts/artifact-cloud-service-races.test.ts | 2 +- src/main/artifacts/artifact-cloud-service.test.ts | 5 ++++- 3 files changed, 6 insertions(+), 3 deletions(-) diff --git a/src/main/artifacts/artifact-cloud-recovery.test.ts b/src/main/artifacts/artifact-cloud-recovery.test.ts index f57b53c2b02..5690a37b94c 100644 --- a/src/main/artifacts/artifact-cloud-recovery.test.ts +++ b/src/main/artifacts/artifact-cloud-recovery.test.ts @@ -336,7 +336,7 @@ function createResponseBody(slug: string): object { renderedContentType: 'text/html', createdAt: '2026-08-06T00:00:00.000Z', updatedAt: '2026-08-06T00:00:00.000Z', - expiresAt: '2026-09-06T00:00:00.000Z', + expiresAt: new Date(Date.now() + 30 * 24 * 60 * 60 * 1000).toISOString(), byteSize: 17, deletedAt: null }, diff --git a/src/main/artifacts/artifact-cloud-service-races.test.ts b/src/main/artifacts/artifact-cloud-service-races.test.ts index c31c2a23e3e..8606dce4bec 100644 --- a/src/main/artifacts/artifact-cloud-service-races.test.ts +++ b/src/main/artifacts/artifact-cloud-service-races.test.ts @@ -33,7 +33,7 @@ function createResponse(slug: string): Response { renderedContentType: 'text/html', createdAt: '2026-08-06T00:00:00.000Z', updatedAt: '2026-08-06T00:00:00.000Z', - expiresAt: '2026-09-06T00:00:00.000Z', + expiresAt: new Date(Date.now() + 30 * 24 * 60 * 60 * 1000).toISOString(), byteSize: 12, deletedAt: null }, diff --git a/src/main/artifacts/artifact-cloud-service.test.ts b/src/main/artifacts/artifact-cloud-service.test.ts index 75da3922fc8..8a02478feb2 100644 --- a/src/main/artifacts/artifact-cloud-service.test.ts +++ b/src/main/artifacts/artifact-cloud-service.test.ts @@ -43,7 +43,10 @@ const cloudB: OrcaProfileCloudSummary = { linkedAt: 2 } -function createResponse(slug = 'artifact-a', expiresAt = '2026-09-06T00:00:00.000Z'): Response { +function createResponse( + slug = 'artifact-a', + expiresAt = new Date(Date.now() + 30 * 24 * 60 * 60 * 1000).toISOString() +): Response { return new Response( JSON.stringify({ artifact: { From 3bb038a1851922b75ab15e7e4b1631e11a36f32f Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:20:59 -0400 Subject: [PATCH 10/17] docs(relay): 2026-09 reconnect findings, improvement checklist, roadmap, and Roll 2 plan (#18958) Operator record for the 2026-09-04 relay reconnect incident and the Roll 1 same-cap cell image roll (complete 2026-09-05, selector gen 148), plus the follow-up checklist, roadmap, and the Roll 2 implementation plan. Docs only; split out of #18565 so the record merges independently of the code. --- .../relay-improvement-checklist-2026-09.md | 189 ++++ .../docs/relay-improvement-roadmap-2026-09.md | 67 ++ .../docs/relay-reconnect-2026-09-findings.md | 991 ++++++++++++++++++ cloud/docs/relay-roll2-plan-2026-09.md | 154 +++ 4 files changed, 1401 insertions(+) create mode 100644 cloud/docs/relay-improvement-checklist-2026-09.md create mode 100644 cloud/docs/relay-improvement-roadmap-2026-09.md create mode 100644 cloud/docs/relay-reconnect-2026-09-findings.md create mode 100644 cloud/docs/relay-roll2-plan-2026-09.md diff --git a/cloud/docs/relay-improvement-checklist-2026-09.md b/cloud/docs/relay-improvement-checklist-2026-09.md new file mode 100644 index 00000000000..91f1cc742ef --- /dev/null +++ b/cloud/docs/relay-improvement-checklist-2026-09.md @@ -0,0 +1,189 @@ +# Relay improvement: implementation checklist, lanes, and disruption + +Companion to [`relay-improvement-roadmap-2026-09.md`](./relay-improvement-roadmap-2026-09.md) (item numbers +match). This file answers three questions per item: what are the concrete steps, what can run in parallel, +and will a user notice. + +## Status as of 2026-09-04 22:30Z + +Three buckets. "Merged" means the code is on `main` and nothing in production has changed yet. "Deployed" means users are already getting it. "Awaiting owner" means I will not touch production without a go. + +**Deployed to production** +- Auth instance cap 20 + dead-family audit fix (orca-cloud #474) as revision `orca-cloud-auth-00031-tox`. +- Dynamic NAT ports in both regions (stablyai/orca #18693). Zero drops and zero proxy dial errors since. +- Nine alert policies with log metrics: 4 auth (#475), 3 relay Cloud SQL/NAT (#18693), 1 cell process-exit (#18717), all on the relay Slack channel. + +**Merged, ships with the next relay cell image roll (Roll 1 carries `519f4914`; Roll 2 needs a fresh image build)** +- Per-cell inventory locks, delta counters, pool `statement_timeout` (#18722). Roll 2. +- Cells dial Cloud SQL with `--private-ip` when configured (#18720). Inert until 2.1 applies. +- Phone shows a clear "sign in on the desktop again" state when the desktop is signed out (#18698). + +**Merged, ships with the next auth deploy** +- Refresh rotation grace window (orca-cloud #478). Startup adds one nullable column (brief exclusive lock on `refresh_tokens`). +- Pruning job code (orca-cloud #476) is in the image; the job itself is Terraform-disabled until 1.2. + +**Merged, ships with the next desktop release** +- Never replay a refresh token after a timeout; ±10 % jitter on relay lease renewal (#18719). +- Renderer learns when a cloud session is revoked (#18694). + +**Merged, not applied** +- Incident dashboard (#18717) blocked behind the runtime-metric label drift (5.x first item). +- Monitor probe fix (#18723) is live in the workflow; the same-cap roll gate has not yet produced a green dry-run since. + +**Awaiting owner go (production mutations)** +1. Roll 1 cell image roll (1.1): dry-run gate, then c8 canary, then batches. +2. Auth deploy carrying #478 (3.1): quiet minute for the column add. +3. orca-cloud #477 private IP (2.1): merge arms an instance restart and a one-way door. Recommendation: hold. +4. Runtime-metric `region` label drift (5.x): intentional replacement of 21 metrics, or drop the label. +5. Enable pruning (1.2): first budget 20k rows; needs a Terraform apply. +6. Paging channel for auth alerts (5.2): needs the destination from you. + +**Open code follow-ups (no gate, nobody assigned)** +- Monitor summary Markdown does not render `tolerated: true` continuity events (added by #18798); the state artifact has them, the checkpoint table does not. +- Relay container boot races the `cloud-sql-proxy` sidecar: c13's fresh container exited twice (`applyPostgresSchema` connection timeout, 2 s each) before the proxy was listening. Make schema apply wait for the proxy or order the containers. +- `cloud-deploy-relay-production-capacity-job.yml` (~line 416) has the same wave-0 single-shot preflight carve-out that #18778 removes from the same-cap job; its single-evidence path never retries freshness-only failures. +- `cloud/package.json` `test` names every dev-script test file explicitly; an unregistered `*.test.mjs` is silently never run in CI (found by #18769). Needs a glob or a ratchet that fails on an unlisted test file. +- Same-cap job's verify step uses bare `curl --fail-with-body` against the just-rolled cell; one 503 at the LB warm-up edge failed c8 canary #2 (run 33935407461) after the transition verifier had already passed. Needs a bounded retry, same rule as #18723/#18740. +- `verify-mutation` in `cloud-deploy-relay-production.yml`, the multi-target workflow, and the capacity workflow still binds to an exact commit; same exposure #18754 fixed for the same-cap and rehome paths. +- `incident-live-preflight-cli.ts` reports only `source/code` (`active-probe/threshold_max`) with no signal name or observed value, so a failed mutation preflight (c27 recovery #3, run 33986948522) cannot be attributed to an endpoint without an out-of-band probe. Print the signal and observed/threshold pair. Related: the 2 000 ms `endpointLatencyMs` bar is shared by US and Asia cells while Asia /health round trips from a US runner sit at 0.7–1.3 s idle; consider a per-region bar or the p50 of the gate window instead of one shot. Gates #44 and #45 (2026-09-05) both froze on `cell.production-gce-c27.latency_ms` at 2.6–2.7 s with c28 showing the identical tail under operator probes; the bar is now blocking Asia rolls. **Fix: stablyai/orca #18877** (per-region `cellEndpointLatencyMs`, us-central1 2 000 / asia-east2 4 000, plus signal/observed/threshold in preflight messages). Residual: `probeEndpointHealth` in `resource-inventory.ts` still uses the flat 2 000 bar to decide whether to retry after the 10 s readiness-cache wait, so a healthy Asia cell over 2 s costs one extra probe per sample (latency, not verdict); thread the region bar into the retry decision. +- The root oxlint config ignores `cloud/**`, so `check:code-quality:changed` never inspects relay-ops or the cloud dev scripts; typecheck + vitest is the only gate there. +- Monitor bars that froze on non-health today: `directorInstancesMin: 5` with `latest-sum` (one-minute instance recycle), `endpointLatencyMs: 2000` on a US-runner probe to asia-east2, `cloudDataMaxAgeMs: 180000` vs Cloud Monitoring publish lag up to 255 s. Recalibrate with a week of data. +- `parsed()` in `resource-inventory.ts` still returns null on a 200 with a malformed MIG body; a second path to `runtime_power_unknown`. +- Deploy script strips `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` on every release (3.1 first item). +- `assignOnce` placement lock still global (4.1 remainder). +- Region preference (4.2), retries-bar recalibration after a week of Roll 2 data (4.4), pruner `stopReason` alert (1.5). +- Full apps-root apply for 4 unrelated drifts (1.4), from a host with the 1Password account. + +## Uplift ranking (reliability gained per unit of effort) + +| Rank | Item | Why it ranks here | +|---|---|---| +| 1 | 1.1 cell image roll | Removes the only crash mode we have seen in production. 22 of 23 cells still have it. One afternoon. | +| 2 | 3.1 refresh rotation grace window | Turns the entire "slow auth → mass sign-out" class into a slowdown. One day. | +| 3 | 4.1 inventory lock contention | The floor under every 503 and slow phone accept, every day, not just incidents. One week. | +| — | 2.2 relay/auth database split | **Deferred 2026-09-04** to ~2026-11-01. Biggest structural fix, but the concrete cause is fixed and alerts now page; see roadmap 2.2 for re-open triggers. | +| 4 | 1.2 + 1.3 pruning and reclaim | Defuses the 63 M-row time bomb. Low effort, mostly waiting. | +| 5 | 5.1 + 5.2 crash alert, page a human | Cheapest detection uplift; today's incident ran 4 h unpaged. | +| 6 | 2.1 private IP | Durable version of a fix that already landed (dynamic NAT ports). Do it on the existing instance. | +| 7 | 4.3 + 3.2 desktop hardening | Small, ride the normal desktop release. | +| 8 | 4.2, 4.4, 5.4, 1.4, 1.5 | Housekeeping and quality-of-life. | + +## The shared bottleneck: cell rolls + +Every change to what runs on a cell (image, proxy flag, env, relay code) needs a same-cap roll: drain → +recreate → verify, one wave at a time, gated by the 15-minute monitor, about an afternoon. Each wave forces +the desktops on that cell to re-dial (c7 canary: 807 controls re-dialed in ~10 s) and phones on those +desktops reconnect on their normal retry. Users see a few seconds of "reconnecting" per wave. + +So batch. Two rolls, not five: + +- **Roll 1 (now):** current image only (1.1). Do not wait for anything else. +- **Roll 2 (week 2–3):** proxy `--private-ip` (2.1) + relay pool `statement_timeout` (2.3) + lock-contention + fix (4.1), all in one image/template. Prerequisite: 2.1's peering and private IP exist first. + +## Lanes (independent; different people can own them) + +``` +Lane A data plane 1.1 roll ──────────────────► Roll 2 (2.1 flag + 2.3 + 4.1) ──► 4.4 recalibrate +Lane B auth/DB 1.2 enable pruning ──(10 d)──► 1.3 reclaim 3.1 grace window (any time) +Lane C network 2.1 peering + private IP ─────┐ (feeds Roll 2) (2.2 DB split deferred) +Lane D desktop 3.2 no same-token retry, 4.3 lease jitter (any release; wire-compatible) +Lane E observability 1.5, 5.1, 5.2, 5.4 (Terraform only, any time) +Lane F director 4.2 region preference (Cloud Run deploy, any time) +Misc 1.4 full apps-root apply (any time; see its check) +``` + +Hard dependencies: Roll 2 waits on 2.1's network work; 1.3 waits on 1.2 finishing. Everything else is +independent. (2.2 deferred; if revived, do it after 2.1 so the new instance is private from day one.) + +## Disruption summary + +| Item | User-visible? | What they see | Mitigation | +|---|---|---|---| +| 1.1 / Roll 2 | **Yes, transient** | Per wave, desktops on that cell reconnect within seconds; phones follow on retry. | Waves gated by the monitor; run in the US night. Already rehearsed on c7. | +| 1.2 pruning | No | Background deletes, 5k rows per batch. | Small first budget; watch `stopReason` and Cloud SQL write throughput. Stop the scheduler if checkpoint alerts fire. | +| 1.3 reclaim | **Depends on tool** | `VACUUM FULL` takes an exclusive lock on `refresh_tokens`: sign-in and refresh block for its duration (minutes to tens of minutes on 16 GB). `pg_repack` holds only brief locks. | Use `pg_repack`. If VACUUM FULL, announce a maintenance window. | +| 1.4 full apps apply | Should be none, **verify** | Terraform will create a new auth revision (env added). Traffic is pinned to `00031-tox` by name, so the new revision should receive 0 %. | Confirm in the plan that no `traffic` change appears. If it does, stop: the Terraform image variable is not the serving image. | +| 1.5, 5.x alerts | No | | | +| 2.1 private IP | **Yes, certain** | Google: "Configuring an existing Cloud SQL instance to use private IP causes the instance to restart, resulting in downtime." No in-place path, HA does not avoid it. Expect 1–2 min DB unavailability: sign-in fails, relay renewals retry. **One-way door**: private IP cannot be disabled and the VPC link cannot be removed once set. The proxy flag change rides Roll 2. | Off-peak; only after Roll 1 (old image dies on a 2 min DB blip). Owner decision required before the foundation apply. | +| 2.2 DB split (deferred) | **Yes, scheduled** | Relay unavailable for the cutover (drain all cells → copy relay tables → flip `DATABASE_URL` → restart). Minutes if rehearsed. Desktops and phones reconnect automatically after. | Rehearse on staging; do it in the US night; announce. | +| 2.3 statement timeout | No beyond Roll 2 | | | +| 3.1 grace window | No | Auth deploys are no-traffic candidate → smoke → promote. | Security trade-off: a stolen token replayed inside the window is served once instead of revoking. 60 s is the usual choice. | +| 3.2, 4.3 desktop | No | Normal app update. | | +| 4.1 lock fix | No beyond Roll 2 | | Verify against real Postgres on 55440 with concurrent probes before shipping. | +| 4.2 region preference | **Minor, Asia users** | Phones that start being placed in Asia reconnect once to a nearer cell. | Roll out behind the existing region-preference flag. | +| 4.4 | No | | | + +## Checklists + +### 1.1 Cell image roll (Roll 1) +- [x] Confirm fleet is quiet: 15-min monitor dry-run passes. #19 green 23:07:53Z (run 33927238469). Canary then failed the evidence provenance check because main moved during the gate; re-gating with a same-commit chain. +- [x] Confirm director is on 519f4914 and c7 on 85bf6799 (confirmed 2026-09-04 via instance-template census; 20 serving cells still on `5aedbca5`) (`verify` mode of the same-cap workflow). +- [x] Dispatch `cloud-deploy-relay-production-same-cap` waves per the plan in the findings doc; one wave, verify, next. Done 2026-09-05 01:14Z–22:27Z: c8 canary, US batches c9–c10, c13–c16, c19–c26 at protocol 1, then Asia c27 (recovered via `mode=rollback` re-entry after gate freezes on the flat latency bar, fixed by #18877), c28, c29 as single-cell canaries at protocol 0. +- [x] After each wave: the transition verifier passed at migration-only and again at general on every cell (assignments carried, heartbeat fresh, hard cap 3 000); no `container die` fleet-wide across the whole roll. The 4408/1006 burst per wave was not measured separately; the verifier's assignment count before and after each restart is the recovery evidence recorded. +- [x] Record image census in the findings doc. 2026-09-05 22:27Z: all 19 general cells on `519f4914` except c7 on `85bf6799`; existing-only c1–c6, c11, c12 and migration-only c17, c18 untouched on their older images by design. Selector at gen 148. + +### 1.2 Enable pruning +- [x] `auth_token_pruner_image` = digest of `orca-cloud-auth-00031-tox` (`343a0915…`; it contains the entrypoint). orca-cloud #479 merged. +- [x] `auth_token_pruner_enabled = true`, `auth_token_pruner_max_rows_per_run = 20000` for the first day (orca-cloud #479). +- [x] Targeted plan asserted 9 create / 0 change / 0 destroy. Applied 2026-09-05 02:06Z. +- [x] Trigger one run by hand; read the summary event. 02:18Z: `time-budget`, 73 batches, 365k scanned, 1 040 deleted (1 021 revoked, 19 expired), no errors. Scan-bound. +- [ ] Raise the budget to the default 200k after a clean day; watch Cloud SQL write MB/s and the checkpoint alert. +- [ ] 1.5: log metric + policy on `stopReason != complete`. + +### 1.3 Reclaim +- [ ] Wait for steady-state runs deleting ~0 rows. +- [ ] `pg_repack -t refresh_tokens` off-peak (needs the extension; check `pg_available_extensions`). Not `VACUUM FULL` without a window. +- [ ] Confirm table + index size and `disk/utilization` dropped. + +### 1.4 Full apps-root apply +- [ ] Run from CI or a host with the 1Password account (local plan fails on the Cloudflare data source). +- [ ] Plan shows exactly the four known drifts and **no traffic change** on `google_cloud_run_v2_service.auth`. +- [ ] Apply; confirm `status.traffic` still pins `00031-tox` at 100 %. + +### 2.1 Private IP (PRs open: orca-cloud #477 foundation, stablyai/orca #18720 relay flag) +- [ ] **Owner decision**: the foundation apply restarts the instance and is irreversible on Google's side. Merging #477 arms the next foundation apply; hold the merge until the window is chosen. +- [ ] Director is out of scope: it uses the Cloud Run built-in connector (managed Google path, not the relay VPC NAT), so it consumed none of the exhausted ports; moving it needs Direct VPC egress + a separate DSN secret. Own PR if ever wanted. +- [ ] Step 7 (`ipv4_enabled=false`) is blocked until humans have IAP/bastion access and the director is moved; it breaks both today. +- [ ] Allocate a `/24` private services range on the relay VPC; `google_service_networking_connection`. +- [ ] Add `ip_configuration.private_network` to `google_sql_database_instance.auth` (foundation root). Plan must show update, not replace. +- [ ] Apply off-peak; expect a possible restart. Watch auth 5xx alert and relay `sqlFailures`. +- [ ] Cell template: proxy args add `--private-ip` (code merged #18720; flag not set). Director: Direct VPC egress or connector, then the same flag. Both ride Roll 2. +- [ ] After Roll 2: NAT `port_usage` for relay gateways drops to ~0; then consider `ipv4_enabled = false` (removes the public IP; breaks the local `cloud-sql-proxy --token` workflow unless it also goes private). + +### 2.2 Database split (deferred to ~2026-11-01; checklist kept for when it is revived) +- [ ] New `google_sql_database_instance.relay` (private IP from day one, its own size and flags). Staging first. +- [ ] Relay schema applies cleanly to an empty instance (it does at startup). +- [ ] Rehearsal on staging: drain → `pg_dump` relay tables → restore → flip `relay_database_url` secret → restart director + cells → phones/desktops reconnect. Time it. +- [ ] Production: announce a window; same steps; verify `orca_relay_runtime_metrics` controls recover to pre-cutover count. +- [ ] Update `production-cloud-sql-app-consumers` budget test and both alert policies' `database_id`. + +### 2.3 Relay pool statement timeout (merged stablyai/orca #18722; ships Roll 2) +- [x] `statement_timeout` on the relay `pg.Pool` (5 s, env-configurable; schema pool untimed; `57014` retryable), below the control-renewal deadline; DDL on an untimed connection (same pattern as auth #476). +- [x] Postgres test on 55440: a held lock fails the query fast and the bounded retry takes over. + +### 3.1 Refresh rotation grace window (orca-cloud #478 merged 2026-09-04; deploy pending owner go) +- [ ] Fix the deploy-script env strip for `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` (pre-existing; found by #478). +- [x] `rotateRefreshToken`: if `rotated_at` within 60 s and not revoked, return the existing successor (idempotent), no revoke, no audit. +- [x] Outside the window or a third presentation: unchanged (revoke + audit). +- [x] Tests: replay inside window returns same successor; outside revokes; concurrent double-present yields one successor. +- [x] Deploy via `deploy-auth-production` (candidate → smoke → promote). Deployed 2026-09-04 23:15Z as `orca-cloud-auth-00035-gos`, cap 20 kept, 0 5xx; `successor_material` column present; sealed successors being written. (candidate → smoke → promote). + +### 3.2 / 4.3 Desktop (merged stablyai/orca #18719; ships next desktop release) +- [x] 3.2: on refresh timeout, re-read stored session before retrying; do not re-send a token already rotated locally. +- [x] 4.3: ±10 % jitter on control lease renewal; unit test on the distribution; wire-compatible (server accepts early renewals already). + +### 4.1 Lock contention (partial: stablyai/orca #18722 merged; ships Roll 2) +- [x] Replace the global `FOR UPDATE` over `relay_cells` with per-cell row locks; counters delta-only. Remaining: `assignOnce` placement lock is still global (optimistic snapshot follow-up). with per-cell row locks or `pg_advisory_xact_lock(cell)`; counters delta-only. +- [x] Postgres tests on 55440 with concurrent probes (in #18722). Staging load run still owed; `postgres_retries` per hour drops in staging load run. +- [ ] Ships in Roll 2; then 4.4 recalibrates the retries bar from a week of data. + +### 4.2 Region preference +- [ ] Director: honor requested region when the preferred region has headroom, else sticky. Behind the existing flag. +- [ ] Measure with `orca_relay_runtime_metrics` region counters before/after. + +### 5.x Observability +- [x] **Relay-root runtime-metric drift**: resolved by dropping the `region` label to match live state (stablyai/orca #18734). Applied 2026-09-04 23:11Z: 8 never-applied `control_*` renewal metrics + the incident dashboard created, 0 destroyed, 21 live metrics untouched. +- [x] 5.1 `container die` log metric per cell (`relay_cell_process_exit`, applied 2026-09-04 via #18717), > 3 / 15 min, relay channel. +- [ ] 5.2 Add a paging channel (**needs owner input**: destination) to `auth_alert_notification_channels` for refresh rejections + latency. +- [x] 5.4 One dashboard (applied 2026-09-04 23:11Z): `orca_relay_cloud_sql_wal_checkpoint`, NAT drops, `orca_auth_refresh_401`, summed `controls`. diff --git a/cloud/docs/relay-improvement-roadmap-2026-09.md b/cloud/docs/relay-improvement-roadmap-2026-09.md new file mode 100644 index 00000000000..64f33c69a70 --- /dev/null +++ b/cloud/docs/relay-improvement-roadmap-2026-09.md @@ -0,0 +1,67 @@ +# Relay improvement roadmap (written 2026-09-04, after the auth/relay outage) + +Owner-facing list of what is left to make the relay more robust, in priority order. Evidence and history +for every item is in [`relay-reconnect-2026-09-findings.md`](./relay-reconnect-2026-09-findings.md) +(Findings 1–13). Everything already landed on 2026-09-04 is listed at the end so this file is complete on +its own. + +## 1. Finish what 2026-09-04 started (this week) + +| # | Item | Why | How | Size | +|---|---|---|---|---| +| 1.1 | **Roll all 23 cells onto the current relay image** | Every cell still runs the image that exits the whole process on a Postgres connect timeout (Finding 6). The fixed image runs only on the director and c7. Any future DB stall repeats the 200-crashes-in-48h pattern. | `cloud-deploy-relay-production-same-cap` waves, gated by the 15-min monitor. Roll inputs and canary results are in the findings doc ("Roll inputs", "Canary blast radius"). | one afternoon | +| 1.2 | **Enable the refresh_tokens pruning job** (orca-cloud #476, merged, off) | `refresh_tokens` is 63 M rows / 26 GB and grows forever; its size is what turned a slow disk into a sign-out storm (Finding 13). | Build an auth image from main (the 21:04Z deploy already contains the entrypoint: `orca-cloud-auth-00031-tox`, digest `343a0915…`), set `auth_token_pruner_enabled = true` and the image digest in `infra/terraform-apps/environments/production.tfvars`, apply targeted. First run with a small `auth_token_pruner_max_deleted_rows`. Watch the run summary's `stopReason`, not the exit code. ~48 M rows drain in ~10 days at 200k/hour. | 1 hour + 10 days of watching | +| 1.3 | **Reclaim the disk after pruning** | Deletes leave dead tuples; the 16 GB table does not shrink on its own. | `pg_repack` (or `VACUUM FULL` in a maintenance window; it takes an exclusive lock) on `refresh_tokens` off-peak, after 1.2 finishes. | 1 evening | +| 1.4 | **Full Terraform apply of the orca-cloud apps root** | The production plan carries four drifts from other merged work: `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` env on the auth service (#476), a skill-share log exclusion filter change, skill pressure threshold 16→8, an artifacts bucket lifecycle rule. Locally it also fails on the 1Password Cloudflare data source. | Run from CI or a machine with the 1Password account; review the four drifts as ordinary changes. | 30 min | +| 1.5 | **Alert on the pruning job** | A run that only ever times out exits 0 and reads as green. | Log metric on the job's summary event where `stopReason != "complete"`, policy on the relay channel. | 1 hour | + +## 2. Remove the shared fate between auth and relay (2.1 and 2.3 this quarter; 2.2 deferred) + +| # | Item | Why | How | Size | +|---|---|---|---|---| +| 2.1 | **Private IP for Cloud SQL, `--private-ip` on the cell proxies** (do this on the existing shared instance; do not wait for 2.2) | Cells reach the database's public IP through Cloud NAT. Dynamic port allocation (landed) raised the ceiling from 64 to 4096 ports per VM, but the NAT is still in the path and its logs are still the only place port exhaustion shows up (Finding 11). | Add a private IP to `orca-cloud-auth-db` (foundation root, orca-cloud), peer the relay VPC, switch the proxy flag in the cell template, roll. | 1–2 days | +| 2.2 | **Split the relay database from the auth database** — *DEFERRED 2026-09-04 (owner decision): revisit ~2026-11-01 once pruning is done and there is a month of alert history* | One Cloud SQL instance serves `orca_auth`, `orca_relay`, `orca_push`, `orca_skills`. The auth table's growth stalled the relay for a day (Findings 10, 13). Deferral rationale: the concrete cause is fixed (disk 250 GB, WAL 16 GB, index, pruning), 2.3 + 1.1 turn a future stall into retries, and the checkpoint/disk/headroom alerts now page. Re-open if the checkpoint-loop or connection-headroom alert fires, or a large new auth-side table is planned. | New instance for `orca_relay`; migrate with a short relay drain. Relay state is small so the cutover is minutes. | 1–2 weeks incl. rehearsal on staging | +| 2.3 | **Statement timeouts on the relay pool** (the auth pool got one in #476) | A relay query stuck behind a checkpoint fsync should fail fast and let the bounded retry take over rather than hold a pool slot for seconds. | `statement_timeout` on the relay `pg.Pool` in `cloud/apps/relay`, tuned under the lease renewal deadline. | half a day | + +## 3. Make the desktop refresh path forgiving (next 2 weeks) + +| # | Item | Why | How | Size | +|---|---|---|---|---| +| 3.1 | **Refresh-token rotation grace window** | The server revokes the whole family the first time a just-rotated token is presented again. On 2026-09-04 that turned a 30 s server slowdown into 21,605 sign-outs. A short window (e.g. 60 s) where the immediately-previous token is still accepted, returning the same new token, is standard practice. | In `apps/auth/src/tokens/refresh-tokens.ts`: accept `rotated_at` within the window, return the successor instead of revoking. Keep true reuse (outside the window, or a third presentation) as revocation. | 1 day incl. tests | +| 3.2 | **Do not retry `/refresh` with the same token on timeout** | Desktop's 30 s `CLOUD_REQUEST_TIMEOUT_MS` expiring is treated like a network error and retried with a token the server may already have rotated. | In `src/main/orca-profiles/profile-cloud-session-refresh.ts`: on timeout, re-read the stored session first, and prefer a longer single attempt for the refresh call specifically. | half a day | +| 3.3 | **Un-revoke is impossible; make sign-out recovery obvious instead** | Server-side un-revoke does not help because the desktop deletes its local token on the 401. Landed: desktop notices immediately (#18694) and the phone says "desktop signed out" (#18698). | Nothing more unless we want a re-auth deep link from the phone to the desktop. | — | + +## 4. Chronic relay issues already characterised + +| # | Item | Why | How | Size | +|---|---|---|---|---| +| 4.1 | **Cell-inventory lock contention** (partial: PR #18722 narrowed the remaining non-placement sites; `assignOnce` placement lock is the follow-up) | `postgres_retries` is a global `FOR UPDATE` over the 23-row `relay_cells` table with a 1 s `lock_timeout`; it is the floor under every 503 and every slow phone accept (Findings 2, 5; memory `relay-cell-inventory-lock-contention`). | Per-cell row locks or an advisory lock keyed by cell; move capacity counters to delta writes. Verify against real Postgres on 55440. | 1 week | +| 4.2 | **Region preference is mostly inert** | Phones request an Asia cell on ~19 % of attempts and get one ~6 % of the time; the sticky lane wins silently, so Asia users ride the US path more than intended (memory `relay-region-preference-mostly-inert`). | Let a region preference override stickiness when the preferred region has headroom; measure with `orca_relay_runtime_metrics` region counters. | 2–3 days | +| 4.3 | **Desktop lease-rotation waves** | A cell recreate seeds a fleet-wide 1006/4408 reconnect burst ~54 min later, every ~54 min (Finding 3). | Jitter the desktop control lease renewal by ±10 % so the cohort spreads out. | half a day, desktop + wire-compatible | +| 4.4 | **Raise `postgres_retries` gate calibration** | The 300 bar was recalibrated (PR #18580) but should track the post-lock-fix baseline once 4.1 lands. | Re-derive from a week of `orca_relay_postgres_transaction_retry` counts. | 1 hour | + +## 5. Observability still missing + +| # | Item | Why | How | +|---|---|---|---| +| 5.1 | **Cell crash-rate alert** | 201 process exits in 48 h with no page (Finding 6). | Log metric on `container die` for `resource.type="gce_instance"` relay cells, > 3 per 15 min per cell. In `cloud/infra/terraform/relay-observability.tf`. | +| 5.2 | **Page a person for auth alerts** | Today's four auth policies (orca-cloud #475) route to the relay Slack channel only. A repeat of 2026-09-04 deserves a page. | Add a PagerDuty/phone notification channel to `auth_alert_notification_channels` for refresh rejections and latency. | +| 5.3 | **Pruning job alert** | See 1.5. | | +| 5.4 | **Dashboard that puts the four signals side by side** | Diagnosis took hours because checkpoint state, NAT drops, auth 401 rate, and fleet controls live in four consoles. | One Cloud Monitoring dashboard: `orca_relay_cloud_sql_wal_checkpoint`, NAT `dropped_sent_packets_count`, `orca_auth_refresh_401`, summed `controls`. | + +## Landed on 2026-09-04 (for completeness) + +- Auth service cap 2 → 20 (service-level manual scaling removed); Cloud SQL disk 49 → 250 GB PD-SSD; + `max_wal_size` 16384; partial index `refresh_tokens_family_unrevoked` built concurrently by hand. +- orca-cloud #474: the above in Terraform + deploy workflow; replayed dead token answers 401 without + re-revoking or re-auditing. Deployed as `orca-cloud-auth-00031-tox` 21:04Z. +- orca-cloud #475: auth alerts (refresh 401 > 100/5 min, 429 > 20/5 min, 5xx > 10/5 min, p99 > 10 s). Applied. +- orca-cloud #476: batched `refresh_tokens` pruner (disabled), auth pool `statement_timeout` 10 s, schema + DDL on an untimed connection. +- stablyai/orca #18693: both relay NATs on dynamic port allocation 64..4096 (applied US 21:01Z, Asia 21:05Z); + alerts for Cloud SQL WAL-checkpoint loop, disk > 70 %, NAT `OUT_OF_RESOURCES` drops. Applied. +- stablyai/orca #18694: desktop learns of a revoked session immediately, panes re-fetch on mount, pairing + notice says "Sign in again to use Orca Relay". +- stablyai/orca #18698: phone shows "Desktop signed out — sign in to Orca on your desktop to reconnect" via + the WebSocket close reason (only additive slot old phones tolerate). +- Director on image 519f4914; c7 on 85bf6799; other 22 cells still on the old image (see 1.1). diff --git a/cloud/docs/relay-reconnect-2026-09-findings.md b/cloud/docs/relay-reconnect-2026-09-findings.md new file mode 100644 index 00000000000..426a120c251 --- /dev/null +++ b/cloud/docs/relay-reconnect-2026-09-findings.md @@ -0,0 +1,991 @@ +# Relay reconnect investigation: findings and evidence + +Working notes for the 2026-09-04 mobile relay reconnect incident and the cell roll that follows. +Kept current across context compactions. Newest section first. All times UTC. Host ids are log digests, +never raw ids. Nothing here is a production mutation record unless the "Mutations" section says so. + +## Status board + +| Item | State | Where | +|---|---|---| +| PR #18565 relay accept abandonment + lease jitter + desktop rotation spread + phone probe fail-fast | Open, CI fully green again after the doc move (05:45Z), CodeRabbit + Pullfrog cleared, 3 review rounds; not merged (owner has not asked) | https://github.com/stablyai/orca/pull/18565 | +| PR #18569 monitor `relayPostgresRetryExhausted` 0 -> 300 | **Merged** 2026-09-04 ~04:20Z as 4101505b6b | https://github.com/stablyai/orca/pull/18569 | +| Same-cap `verify` of c7 (read-only) | **Passed** run 33836527159 | confirms identities, selector gen 110, rehome gen 12, protocol 1, digests | +| Monitor dry-run #1 | Froze min 5: `relay.postgres_retries` 380 > 300 | run 33836470590 | +| Monitor dry-run #2 | Green to min 13, froze 04:49Z: `director.concurrency` 76.7 > 64 (six-cell crash storm, Finding 6) | run 33837160275 | +| Monitor dry-run #3 | Froze min 3 at 05:01Z: `relay.postgres_retries` 339 > 300; no crash, concurrency 5–8 | run 33838698725 | +| Owner decision 2026-09-04 ~05:10Z | **Option B approved**: "you can raise the bar. or remove it altogether ... whats the most logical move". Kept the bar (removal would leave contention unwatched during the roll) and recalibrated from measured data. | this thread | +| PR #18580 monitor `relayPostgresRetries` 300 -> 2000 | Open, awaiting CI; mutation-checked (300 fails the new test) | https://github.com/stablyai/orca/pull/18580 | +| PR #18565 CI | Was red on `root directory guard` because this findings file sat at repo root; moved to `cloud/docs/` in 8ebff89106 | | +| PR #18580 | **Merged** 2026-09-04 05:23Z as 79d5fb469a (Pullfrog cancelled by the merge; independent Opus review requested instead, per owner) | | +| Monitor dry-run #4 | Froze min 12 at 05:37:35Z: `cell.production-gce-c27.health`/`.ready` = 0. Retries green all 12 samples under the new 2000 bar. Cause: c27 (asia-east2) container died 3x 05:37:00–05:38:01Z, Finding 6 crash class. | run 33840364323 | +| Monitor dry-run #5 | Froze at sample 1 (05:41Z): c27 health/ready still 0. MIG autoheal `recreateInstance` on c27 fired 05:38:12Z after the 3 crashes; instance RECREATING, process up with 0 controls (was ~395). Second c27 recreate in 7 h (Finding 3 seed pattern). Waiting for c27 to settle before dry-run #6. | run 33841327879 | +| Monitor dry-run #6 | **Passed** 06:06:31Z: 16 samples, no freeze (started 05:47:42Z) | run 33841783747 attempt 1 | +| c7 `canary-apply` | **Succeeded.** Dispatched 06:07:15Z; drain 06:10Z; MIG recreate 06:16–06:23Z; new image listening 06:23:42Z; verify + trust proof passed; restored to `admission=general` 06:25:21Z; canary authority sealed. c7 is on `85bf6799…`. | run 33843071283 | +| PR #18581 doc reconcile (Aug 23 figure: 2,200–3,000 raw log lines vs 1,510 on the gate metric) | **Merged** | https://github.com/stablyai/orca/pull/18581 | +| Same-cap `verify` c7 target=519f4914 rollback=85bf6799, gen 112 | **Passed** (read-only) | run 33856355648 | +| Monitor dry-run #7 (gen 112) | Froze at sample 1 (09:05:31Z): `director.errors` 4 > 0, the four 2.0 s pg-connect 500s from the 09:00 cascade still inside the 5-min delta window. Dispatched 4 min too early. | run 33856521278 | +| Monitor dry-run #8 (gen 112) | Green for 15 of 16 samples (09:09:38–09:24), froze on the final sample 09:25:22Z: `director.errors` 1 > 0. The one 500 was `/v1/admin/evacuation-status` at 09:23:50Z, 2.01 s latency = director pg-connect timeout, called by **the monitor's own collector** (`incident-monitor-sources.ts:492`). First evacuation-status 500 since Sep 1. The gate froze on a request it made itself. | run 33856905229 | +| Monitor dry-run #9 (gen 112) | Froze: c13/c23 crashed 50 s after dispatch, then c14/c20/c9 at 09:34. | run 33858650691 | +| Monitor dry-run #10 | Dispatched 09:46:13Z; froze at sample 5 (09:56:59Z): `director.errors` 12. All twelve at 09:55:17–21Z, 0.8–2.1 s latency, 10 on `/v1/regions` + 2 on `/v1/assign`; c16 and c8 crashed at 09:55:19 in the same second. A single 4-second Postgres connect stall hit director and cells together. | run 33859947207 | +| Monitor dry-run #11 | Froze at sample 2 (10:08:07Z): `director.concurrency` 79.8 > 64, the c8/c20 re-dial. They crashed 10:05:54, 3 s before the waiter's quiet check passed (log ingestion lag). | run 33861578009 | +| Monitor dry-run #12 | Dispatched 10:17:38Z after 10 quiet min; froze at sample 2 (10:19:24Z): `cell.production-gce-c16.health` 0. c16 did **not** crash (no container die, MIG NONE/HEALTHY, readiness=true throughout, `/health` 200 in 230 ms at 10:21). At 10:19:07–16 it logged "control activity renewal failed" x4 and a burst of 1006 closes, sqlFailures 1 -> 14, sqlLatencyMsMax 2588: a pg stall on the old image that did not reach the unhandled path. The probe's single fetch (30 s timeout) came back unavailable during that stall and `unavailableIsZero` turned it into health=0. | run 33862504601 | +| Monitor dry-run #13 | Green 14 of 16 samples (10:48:38–11:03), froze 11:04:43Z: c9 crashed 11:04:23, c28 11:04:25 (then looped 11:05:04, 11:05:41); c15 probe also read 0 (stall, no crash). Missed by ~90 s. **Dispatched by hand 10:48:15Z** into a 43-min crash lull (last die 10:05:54; last director 500 10:31:49). The re-armed waiter never fired: its MIG-stable check used `grep -vc True`, which exits 1 when nothing matches, so `&&` short-circuited on the *healthy* case. Waiter armed 10:20Z: 10-min quiet + every MIG stable + 60 s recheck, then dispatch, then canary c7 on green. Held at 10:24 and 10:31 by lone director `/v1/assign` 500s (2 s pg-connect stalls, no cell crash). Director 500 events since 08:46: 6 (gaps 2.7/21/31/29/7.6 min). At 10:39 the waiter was re-armed with a 6-min director-500 window (the monitor's own delta is 5 min) instead of 10, since the gate only needs the 15 min *after* dispatch to be clean. Cell crashes have stopped since 10:05 (33+ min, longest gap since 08:40). 12 dry-runs: 1 pass (#6), 11 freezes, none on a real fleet-health regression. | Cascade gaps since 09:00: 31, 2.9, 5.1, 16.1, 4.0 min (median 5); a 15-min clean window is ~28% per attempt at this rate. | | +| Monitor dry-run #14 | Dispatched 11:26:53Z by the fixed waiter (first autonomous dispatch); c14, c23, c25, c15, c24, c19 died 11:30:59–11:31:08 (six cells, 13 min after the last cascade). Froze on c8 (and others) health/ready probes. Waiter re-armed 11:06Z (grep bug fixed: `grep -c` under `|| true`), same chain; held through the 11:17 cascade and c14/c28 recreates. 13 dry-runs: 1 pass, 12 freezes. Since 08:40: 10 cascades, 75 container dies, gaps 20/31/3/5/16/4/6.5/58/13 min; only 3 windows of >=17 clean minutes existed in 2.6 h, and dry-runs hit two of them (#6 passed, #13 lost the third by 90 s). | +| Monitor dry-run #15 | Waiter armed 11:33Z (6-min director-500 window, 8-min crash window, all MIGs stable), chained canary; still holding at 12:04Z. Since 11:00: 8 cascades, 98 dies, gaps 13/13.6/3.6/14.5/4.4/6.1/3.0 min, **max gap 14.5 min**, so no 15-min clean window has existed in the last hour. 14 dry-runs: 1 pass, 13 freezes. | +| Monitor dry-run #15 verdict | Dispatched 12:28:49Z; froze at sample 2 (12:30:41Z): **12 cells** health/ready = 0 at once (c4, c5, c7, c10, c15, c16, c18, c20, c22, c25, c27, c28), including c4/c5 (0 controls all day, `/health` 200 in 190 ms a minute later) and c7 (new image). Six old-image cells also crashed 12:30:02–21. This was a fleet-wide SQL stall, not a cascade: every cell's `sqlLatencyMsMax` hit 4–6 s (c7 4865, director 5140), director pool waiting 1258, 15 cell pg-connect timeouts, director sqlFailures 92. Cloud SQL CPU 0.73, backends 160, new connections normal, memory 0.46, so the *instance* was not saturated; something held the database for ~5 s. Postgres log 12:31:23–28 shows a burst of `could not obtain lock on row in relation "relay_cells"` from NOWAIT (single-row and full-inventory) sweeps, i.e. the row locks were held during recovery. Cloud SQL transactions/min flat (~30k), reads flat, +network flat: the database was neither busy nor saturated, it was *waiting*. The stall bracket +(12:30:02–12:30:41) is where every cell's SQL max hit 4–6 s at once. Lock retries in that window were +ordinary (49/29/13 per min). Best reading: a ~5 s Postgres-side wait event shared by every session +(lock on a hot row held across a long transaction, or an instance-level pause), not CPU/IO. Cell +`sqlLatencyMsMax` was already 1.5–2.2 s fleet-wide in the four minutes before, i.e. the old cells' 1 s +`lock_timeout` plus queueing. | run 33872946111 | +| Monitor dry-run #16 | Dispatched 12:38:57Z; froze at sample 1 (12:40:11Z): `cell.production-gce-c27.latency_ms` 2071 > 2000, a fifth distinct freeze signal, the probe's own round-trip absorbing a checkpoint sync. **Loop stopped by me at 12:41Z**: with the disk in the checkpoint loop (Finding 10) no bar can hold for 15 min, so further dry-runs only burn the shared rollout lease. 16 dry-runs: 1 pass, 15 freezes. Re-arm after the disk change lands. | +| Cloud SQL checkpoint loop | **Broke on its own 12:39–12:45Z**: disk writes 48 -> 4 MB/s at 12:39 with transactions and network flat and no Cloud SQL operation; 12:40:17 checkpoint synced 0.047 s; 12:45:53 checkpoint was `time`-triggered again (first since 11:55) with sync 0.096 s and write spread over 269 s. Cause of the break unknown (most likely WAL fell back under `max_wal_size` once a burst of full-page writes aged out). It can re-enter the loop on the next large checkpoint; the disk-size fix remains the durable one. | +| Monitor dry-run #17 | Dispatched ~12:49Z (all guards clean); froze at sample 1 (12:52:05Z): `director.errors` 4, from the c9/c22 crash loop that began 12:50:34, ~90 s after dispatch. Checkpoints stayed healthy (85 ms), so this is the old image's baseline crash rate, not the disk. 17 dry-runs: 1 pass, 16 freezes. | +| Monitor dry-run #18 | **Dispatched by mistake 13:48:56Z into the outage**: my gcloud credentials expired ~13:45Z, every guard query returned empty, and the waiter's `grep -c . || true` read empty as "quiet". Froze at sample 1 (13:49:43Z) on `director.ready=0`, `auth.health=0`, and cell probes; no canary dispatched, no production mutation. All waiter loops killed at 13:51Z. Lesson: a quiet-window check must fail closed when its data source errors. Waiter had been re-armed 12:53Z. | +| Gate decision | Owner asked at 09:36Z to choose: A keep looping / B recalibrate `directorErrors` 0 -> small n / C human bypass. Ten dry-runs, four froze on this bar. Recommendation B+A. Note: B alone would not have passed #9 or #10 (cell health probes and a 12-error burst); it fixes the single-500 false freezes (#7, #8) only. | | +| Batch roll | **Deferred by plan**: roll once with the lock-fix image instead of twice. | | +| PR #18606 lock removal (root cause) | **Merged** 09:2xZ as 7b108abf71 after review, fix, re-verify; CI green | https://github.com/stablyai/orca/pull/18606 | +| Image publish for 7b108abf71 | **Done** 08:36:49Z run 33854111305: `sha256:519f4914217f08cabcdcd34825965db8473ec37c6591553a3af0d65dcdeeb183` | | +| Director deploy on 519f4914 | **Succeeded** 08:45Z run 33854355791; serving `orca-cloud-relay-00570-siv`, rollback tag on 00569-ret (also 519f4914), 00565-fes (85bf6799) still deployable. Dispatched 08:37:45Z (blue/green; prior revision 00565-fes on 85bf6799 kept as rollback). Note: `predecessor-image-digest` is a required input even with bootstrap=false; pass the serving digest. | `cloud-deploy-relay-production-director.yml` | +| c7 on new image, 2 h in | 817 controls, **0 container die** since restore (was ~1 per 15 min on old image); `sqlLatencyMsMax` still 1.0 s = lock wait unchanged, which #18606 targets | | +| Terraform alert `relay_postgres_retry_exhausted` at `> 0` | Firing continuously since #18521; recalibration not done (own change) | `cloud/infra/terraform/relay-observability.tf:447,469` | + +## Mutations performed (complete list) + +1. Merged PR #18569 to main (code/docs only). +2. Merged PR #18580 and #18581 to main (monitor bar + docs). +2b. Merged PR #18606 to main (relay lock change; no serving effect until the image is deployed). +2c. Dispatched `cloud-publish-relay-production` for 7b108abf71 (builds and pushes an image; changes nothing serving). Done: 519f4914. +2d. Dispatched `cloud-deploy-relay-production-director` on 519f4914 (preserve placement, no prune, rehome gen 12). Succeeded 08:45Z; serving revision 00570-siv. Rollback: `gcloud run services update-traffic orca-cloud-relay --region us-central1 --to-revisions orca-cloud-relay-00565-fes=100` (85bf6799, still Ready). Not needed so far. +3. 2026-09-04 06:07:15Z: dispatched `cloud-deploy-relay-production-same-cap` `canary-apply` for production-gce-c7 only (run 33843071283). Completed successfully 06:26Z: c7 isolated, drained (807 controls re-dialed), template + MIG rolled to 85bf6799, verified, restored to general admission. Selector generation advanced 110 -> 112 (isolate + restore). +4. Nothing else. Both monitor dispatches were `mode=dry-run` (read-only). The same-cap dispatch was `mode=verify` (read-only, confirmed by step gates `if: inputs.mode != 'verify'` on every mutating step). + +## Finding 6 (2026-09-04 ~05:00Z): the old cell image crashes the whole process on a Postgres connect timeout + +**This is the most important open finding.** The 23 GCE cells run image `sha256:5aedbca5…` = orca-cloud +commit e3e92d95d3 (2026-08-14). In that build `beginProof` is called as `void this.beginProof(...)`. +When `verifyCellAssignment` inside it throws (pg-pool `timeout exceeded when trying to connect`, 2 s +`connectionTimeoutMillis`), the rejection is unhandled and Node exits 1. Docker restarts the container +in ~1 s, but every control on that cell (~800 hosts) drops and re-dials `/v1/assign` at once. + +Evidence, cell c7 instance 4545742188814054238, 2026-09-04: + +``` +04:46:47.951 stderr [orca-relay] control activity renewal failed (x5) +04:46:49.527 stderr Error: timeout exceeded when trying to connect + at pg-pool/index.js:45:11 + at async PostgresPoolPressure.connect (postgres-pool-pressure.js:30:20) + at async PostgresDatabase.query (database.js:645:24) + at async RelayAssignmentStore.verifyCellAssignment (assignment-store.js:2024:22) + at async HostSessionRegistry.beginProof (host-session-registry.js:376:15) +04:46:49.527 stderr Node.js v24.19.0 +04:46:49.835 dockerd: container die … exitCode=1 image=…relay@sha256:5aed… +04:46:50.258 dockerd: container start +04:46:52.761 stdout [orca-relay] listening on https://c7.relay.onorca.dev +``` + +2026-09-04 05:36:59–05:38:01Z: c27 died 3x in 62 s plus one other instance (5464389947731541178); this froze dry-run #4 on c27's health probe. + +Fleet-wide `container die … exitCode=1` on the relay image, last 48 h: **201 events on 19 instances** +(c28 x38, c29 x37, c27 x19). Hourly counts track the lock-contention curve (peak 23/h at 21Z Sep 3). +Every one has the same `Node.js v24…` crash banner. On 2026-09-04 04:46:35–04:47:41Z six cells +(c7, c8, c19, c21, c22, c25) died within 66 s: ~4,800 hosts re-dialed, `/v1/assign` returned 16,321 +503s in one minute (baseline ~20), director concurrency hit 85 (Cloud Run cap 80), Cloud SQL +`new_connection_count` 119 -> 287/min. Fleet recovered by 04:51Z. That is what froze dry-run #2. + +Fix status: `guardSessionTask` wrapping `beginProof` landed in orca-cloud #436 (2026-08-27) and is in +the target image `sha256:85bf6799…` (main 11aace8dec). The roll is the fix. Not caused by anything in +this session: the same-cap verify finished ~04:25Z and never reached a mutating step; no compute +operations exist for those instances; heap/event-loop were flat before the crash. + +Autoheal amplifier: MIG health check is `/health` every 10 s, timeout 5 s, unhealthy after 3, so a +crash loop of ~30 s+ triggers `compute.instances.repair.recreateInstance`. All ~20 recreates in the +48 h to 2026-09-04 05:40Z were the three Asia cells (c27 x6, c28 x7, c29 x8; gcloud prints local +-07:00 times). c27 recreated 05:38:12Z after 3 crashes in 62 s; its ~395 controls went to 0 and the +monitor's `cell.production-gce-c27.health/ready` probe read 0 for the whole recreate (~several min), +freezing dry-runs #4 and #5. Each recreate also seeds a Finding 3 rotation cohort. Rolling the Asia +cells early in the batch phase should be weighed against the canary-first rule; c7 stays the canary. + +Implication for the gate: the monitor's `director.concurrency` freeze is *correctly* detecting these +crash storms. A dry-run only passes in a 15-minute window with no cell crash, roughly 1 in 3 windows +at current rates. Retrying in quiet hours is legitimate; the bar is not wrong. + +## Finding 5: `relay.postgres_retries` at 300 is 3x under today's baseline + +Retries per 5 min, cells + director, last 24 h: p50 579, p90 1039, p99 1398, max 1505; **65% of +windows over 300**. Quiet hours (03–08Z) p50 235, max 512. When the 300 bar was set (2026-08-26) +healthy bursts reached 234. Baseline has roughly tripled in 10 days. Skill notes say do not raise this +bar; I have not. Best odds for a clean 15 min are 02–04Z and 17–18Z (9/12 five-minute windows under +300 in each). + +## Finding 4: exhausted-retry bar was the wrong single blocker (fixed) + +`relayPostgresRetryExhausted: 0` never cleared after #18521 reached the director (22:12Z Sep 3): 236/236 +five-minute windows non-zero; post-#18521 p50 42 / p90 147 / max 220; Aug 23 incident peak 467. +Recalibrated to 300 in #18569 (merged). Dry-run #1 immediately revealed Finding 5 behind it. + +## Finding 3: the 00:50Z control-close wave was desktop lease rotation, not a rollout + +2026-09-04 00:49–00:51Z: 2,745 control closes on 19 instances; 1157/1632 code 1006 and 973/1030 code +4408 `control rebound` had ageMs in the 53-minute bin. Relay grants a flat 55 min lease; desktops +rebind 60–120 s early; so every host that (re)connected in the same minute rebinds as one cohort +forever. Seed: c27 MIG autoheal recreate 23:23Z (`compute.instances.repair.recreateInstance`) dumped +~420 controls. Harmonics at 23:55, 00:04, 00:25, 00:49Z. Each rebind is an `activateControl` +transaction that can take the inventory lock. Fix in #18565: relay lease 55 min ± 5 min (symmetric, +so mean rebind rate unchanged), desktop early window 1–6 min. + +## Finding 2: fleet-wide lock contention, worse on Sep 3 + +| window | 55P03 retries/h (cells) | cell sqlFailures/h | +|---|---|---| +| Sep 2 18Z – Sep 3 07Z | 660–1470 | 680–1620 | +| Sep 3 08Z–16Z | 3600–7100 | 3700–7700 | +| Sep 3 23Z | 7468 | 7585 | + +100% of sampled retries are 55P03; director phase is `cell-inventory`. Every cell pins +`sqlLatencyMsMax` at 1.0–1.2 s = the pre-#18521 1 s pool `lock_timeout`. Not load (controls flat +~26k, Cloud SQL CPU 46–53%). No `cloud-*` workflow explains the 08Z step. The lock is a global +`SELECT * FROM relay_cells FOR UPDATE` (23 rows) taken by assignment, control activation, activity +acquire, and sweeps, held to COMMIT. + +## Finding 1: root cause of the phone's 24 s hang (the original symptom) + +`acceptClient` runs four serialized Postgres calls; the fourth (`acquireActivity`) contends for the +global lock. Under contention the cell finishes after the phone's 12 s bound, then +`PendingHostDataReservation.bind` throws `host_data_reservation_already_bound` because the phone's +close already released the reservation. Every "first frame handler failed already_bound" line is that +post-mortem (31 events 23:06–01:01Z across 12 instances). Fix in #18565: abandon the accept after each +DB step once the socket is closed; new event `orca_relay_client_accept_abandoned {stage, elapsedMs}` +and metric fields `clientAcceptsAbandonedByStageDelta` / `clientAcceptAbandonedMsMax`. Phone side: +direct probe now fails fast on `reconnecting` so relay recovery is not queued behind three doomed +LAN redials (~3.5 s saved per foreground). #18518 (merged, not yet on the phone) covers the +stage-aware dial bound. + +Host 666077865f2e: stable throughout. 4408 rotation 00:27:45Z; 1006 quit 00:52:24Z on old adhoc; +sticky reassignment to c27 00:52:35Z on new build; rotation closes 01:44:55Z and 02:23:15Z with +splices intact. No drain/4404/wrong-cell. + +## Finding 7 (2026-09-04 ~05:10Z): retries bar recalibration basis (PR #18580) + +Chose 2000 over removal. The metric is the gate's own source (`orca_relay_postgres_retries` +log metric, director + cells summed per five minutes, ALIGN_DELTA 300 s): + +| window | p50 | p90 | p99 | max | > 300 | +|---|---|---|---|---|---| +| 2026-09-01 | 56 | 105 | 206 | 456 | 0% | +| 2026-09-02 | 109 | 186 | 294 | 377 | 1% | +| 2026-09-03 | 430 | 924 | 1320 | 1504 | 55% | +| 2026-09-04 to 05Z | 285 | 1012 | 1211 | 1211 | 44% | + +15-minute pass rate, last 24 h: bar 300 -> 22%, 800 -> 66%, 1000 -> 86%, 1500 -> 99%, 2000 -> 100%. +Aug 23 incident on this metric: 1510 then 646 (single windows), so retries no longer separate an +incident from baseline; exhausted (467 vs bar 300; healthy 72 h max 184), director concurrency, +and pool bars carry that role. Note: my earlier "p99 1398 / 65% over 300" in Finding 5 came from +raw log line counts; the metric-based numbers above are what the gate actually evaluates. +Baseline tripled between Sep 2 and Sep 3 with no deploy; still unexplained (Finding 2). + +## Decision needed from the owner (resolved: B) + +The same-cap roll is blocked only by the monitor gate, and the gate is blocked by `relayPostgresRetries: 300` +(Finding 5: 65% of windows breach it; even the 04:55Z quiet window hit 339). Three options: + +- A. Keep waiting for a naturally quiet 15 min. Odds per attempt ~1 in 3 in quiet hours, lower by day. + Each attempt is free and read-only. Could take hours. +- B. Recalibrate `relayPostgresRetries` from measured data, same method as #18569: 24 h p99 is 1398, the + Aug 23 incident ran 2200–3000, so ~1500 clears healthy windows with ~1.5–2x incident separation + (less margin than the exhausted bar had). Overrides the "do not raise" note in the skill facts. + Argument for: the roll being gated is the thing that reduces retries. Argument against: the bar is + doing its job of saying contention is high. +- C. A human dispatches the roll with a different gate policy. Not something I can or should do. + +My recommendation: B, with the number chosen from the table in Finding 5 and the roll following +immediately so the bar can be re-tightened after the fleet is on the 500 ms lock wait. + +## Finding 12 (2026-09-04 13:12Z): **INCIDENT IN PROGRESS. The auth service is at its 2-instance cap and rejecting 90% of desktop token calls with 429; the relay fleet has emptied.** + +Timeline: 13:04–13:06 the old-image cascades and NAT stalls drove ~1,400 desktops to re-dial. Their relay +JWTs (5-min TTL) expired mid-storm, so they hit `orca-cloud-auth` `/v1/desktop/auth/refresh` and +`/v1/desktop/auth/relay-token` together. The auth service is Cloud Run `maxScale=2`, `concurrency=80`, +1 vCPU throttled (`auth_max_instances = 2` in orca-cloud `infra/terraform-apps/environments/production.tfvars`, +applied by `deploy-auth-production.yml`). Both instances pinned at concurrency 85 from 13:02; from 13:07 +Cloud Run's front door returns **429 "no available instance"** (0 s latency, never reaches the container): +12,045 at 13:07, 54,292 at 13:08, 46,025 at 13:08, 42,529 at 13:09. Sep 3 total auth 429s: **0**. +Without a fresh relay token every desktop's `/v1/assign` gets 401 (1,433 distinct hosts 401'd, 0 got 200 +since 13:07) and every cell closes its control with `4401 relay authorization expired`. Fleet controls: +13,375 (12:55) -> 7,633 (13:08) -> **249 (13:12)**, splices 1. Auth container CPU 0.15–0.5, so the cap is +the limit, not the code. Every desktop is now in its refresh-retry loop hammering the same 2 instances: +this is a self-sustaining thundering herd and will not clear on its own. At 13:14Z: fleet **30 controls** +across 23 cells; successful relay-token issuance 5,000–6,500/min until 13:05, then 1,059 / 734 / 733 / +443 / 220 / 214 / 148 / **4** per minute through 13:13; auth 429s 54k -> 25k/min only because desktops +are backing off, not because the service recovered. Note `AUTH_MAX_INSTANCES: 2` is also hardcoded in +orca-cloud `.github/workflows/deploy-auth-production.yml` (lines 33–34), so a redeploy would re-pin it; +change both the workflow env and the tfvars. + +**Immediate mitigation (owner action, not applied):** raise the auth service's max instances. Fastest: +`gcloud run services update orca-cloud-auth --region us-central1 --max-instances 20` (or `10`, matching +the other apps' `max_instances = 10`), then land the same in `auth_max_instances` so Terraform does not +revert it. Auth is stateless behind Cloud SQL (`refresh_tokens` table); backends 210 of 400, so 20 +instances x a small pool is within budget. Also consider the desktop's refresh backoff: it re-dials on +401 immediately with no jitter, so a 429 storm sustains itself. + +**13:51Z status: my gcloud session lost auth at ~13:45Z; all production monitoring from this session is +blind until re-authenticated (`gcloud auth login`, interactive). Last confirmed state 13:40Z: fleet 0 +controls, auth maxScale 2, 7,600 auth 429/min. All autonomous dispatch loops are stopped.** + +**17:19Z–17:21Z MITIGATION APPLIED (owner said "fix it NOW").** State at 17:19Z, four hours in: all 23 +cells at 0 controls, auth 429 ~2,000/min, auth 2xx ~40/min, and the 2xx that got through took 13–28 s +(both instances saturated). Mutation 1: `gcloud run services update orca-cloud-auth --max-instances 20` +created revision `orca-cloud-auth-00018-4jc` (same image `auth@sha256:1710ff6c`, same env/concurrency, +only maxScale 2 -> 20) but the service pins traffic to `00023-qud` **by revision name**, so the new revision +was immediately `Retired` and nothing changed. Mutation 2 (17:21:30Z): `gcloud run services update-traffic +--to-revisions orca-cloud-auth-00018-4jc=100`. Lesson: the auth service's traffic block is name-pinned +(the deploy workflow does an explicit traffic switch), so a bare `services update` never reaches users. +Terraform still says `auth_max_instances = 2`; the next `deploy-auth-production.yml` run will revert this +unless the tfvars and the workflow's `AUTH_MAX_INSTANCES` are changed first. + +## Finding 13 (2026-09-04 17:19Z–18:10Z): **the auth outage is a database problem, not (only) a Cloud Run cap; `refresh_tokens` has 63 M rows and reuse-revokes scan whole families** + +Mutations this window (all online, no restarts, all by hand in project onorca-cloud): +1. 17:19Z `gcloud run services update orca-cloud-auth --max-instances 20` → new revision `00018-4jc`, but traffic is + pinned by revision name so it was `Retired`; 17:21:30Z `update-traffic --to-revisions 00018-4jc=100`. +2. Still 2 instances at 17:31Z: the SERVICE has its own `scaling.maxInstanceCount=2` in **manual scaling mode** + (`run.googleapis.com/maxScale: '2'` on service metadata, set by Terraform `infra/terraform-apps/auth.tf`), which + overrides the revision cap. `--scaling=auto` then `--max 20` at 17:31:45Z. Instances 2→20 by 17:38Z; 429s fell + 6,000/2 min → 60/2 min at 17:36Z and controls briefly reached 11. +3. Then latency, not capacity, became the wall: every refresh took 100+ s inside Postgres (desktop client timeout + is 30 s, `CLOUD_REQUEST_TIMEOUT_MS`), so 20 instances × 80 concurrency filled again with requests nobody was + waiting for, and 429s returned (~1,500/2 min from 17:40Z). +4. 17:27Z Cloud SQL disk 62 GB → 250 GB (IOPS ceiling 1,470 → ~7,500). 18:00Z `max_wal_size` 1.5 GB → 16 GB + (the checkpoint loop: `checkpoint starting: wal` every 45–60 s since 13:06Z). +5. 18:07Z `CREATE INDEX CONCURRENTLY refresh_tokens_family_unrevoked ON refresh_tokens(family_id) WHERE + revoked_at IS NULL` (an earlier attempt with `AND rotated_at IS NULL` was wrong for the revoke predicate; its + invalid remnant `refresh_tokens_family_live` was dropped). + +Evidence: `refresh_tokens` = 63.3 M live tuples, 16 GB table + 10 GB indexes; every refresh inserts a row and +nothing ever deletes (30-day TTL rows are never pruned). Query Insights 17:33–17:39Z: `UPDATE refresh_tokens SET +revoked_at = $1 WHERE family_id = $2 AND revoked_at IS NULL` = 21,000 s of execution per 6 min, ~90–120 k rows +updated per minute; io_time 15,000 s read; pg_stat_activity 180+ backends in `IO/DataFileRead` on that statement, +200 backends total for orca_auth (20 instances × pool max 10). `session-refresh-reuse-detected` audit events per +hour: ~100 all day → 8,805 (13Z), 15,511, 19,486, 24,897, 26,935 (17Z). Mechanism: a desktop's refresh times out +client-side at 30 s, the server had already rotated the token, the desktop retries with the same token, the +server calls that reuse and revokes the family (Bitmap scan on `refresh_tokens_family` + heap filter over every +row the family ever had), then the desktop retries the dead token again, and each retry re-runs the same +full-family scan (already-revoked families short-circuit nowhere). Reuse-detected 401 also **signs the user out** +on the desktop (`isOrcaCloudAuthFailure` → `clearCloudSessionIfUnchanged`), so every user who hit this during the +outage must sign in again. + +Durable fixes (orca-cloud PR in preparation on branch `auth-revoke-only-live-tokens`): `AUTH_MAX_INSTANCES` and +`auth_max_instances` → 20; Terraform disk 250 + `max_wal_size=16384`; the partial index in the schema; an +`already-revoked` short-circuit in `rotateRefreshToken` that skips the family UPDATE and the audit insert. Still +open after that: prune `refresh_tokens` (expired or revoked rows older than N days), a server-side statement +timeout shorter than the desktop's 30 s so the client and server agree on failure, and an alert on auth 429s. + +**19:11Z RESOLVED at the database layer.** `refresh_tokens_family_unrevoked` went valid at 19:11:17Z (build +18:07–19:11, two full table scans of 2.1 M blocks under load). Within 60 s: refresh latency 100 s → 0.1 s, auth 429 +→ 0, active orca_auth backends 200 → 2, checkpoints back on the 5-min timer (`checkpoint starting: time` at 18:35, +18:41, 19:00, 19:11). Director `/v1/assign` returning 200. Fleet controls 0 → 17 by 19:14Z. + +**Residual: mass sign-out.** 19:11–19:14Z: 3,857 refresh 401s from 3,829 distinct IPs, then near zero. Every one is +a desktop whose family was revoked by reuse-detection during the outage; the desktop clears its cloud session on +401 (`clearCloudSessionIfUnchanged`) and stops retrying. Those users must sign in again before the relay sees +them. Fresh `/session` sign-ins: 1, 5, 3 per minute at 19:10–19:12. Recovery of controls is now paced by users +signing in, not by infrastructure. Total `session-refresh-reuse-detected` events 13:00–19:00Z ≈ 100k, against a +~100/hour baseline. +**Affected-user count (19:22Z, from `refresh_tokens`):** 23,318 live token families revoked in the window, +**21,605 distinct users**. Only ~3,800 desktops had seen their 401 by 19:15Z; the rest were closed or asleep +and will find themselves signed out on next launch, so sign-ins will trickle for days. + +**Desktop UX finding (owner's own Mac, 19:22Z):** a revoked desktop keeps showing the account card as +"Connected" and the pairing pane as "Orca Relay: Unavailable" / `relay_control_not_active` indefinitely; the +local trace writes no relay events. Only quit + relaunch surfaced the sign-out prompt, after which sign-in → +relay-token → `/v1/assign` 200 (0.15 s) → working pairing, all within 10 s. Follow-ups: the relay coordinator's +401 path should flip the account card to reconnect-required immediately, and the pairing error should say "Sign +in again to use Relay" when the cause is an auth failure. Announcement wording: "If Relay shows Unavailable, quit +and reopen Orca, then sign in when prompted." + +orca-cloud PR #474 (branch `auth-revoke-only-live-tokens`): caps → 20, disk 250 / max_wal_size 16384 in +Terraform, partial index in the schema, `already-revoked` short-circuit. Do not deploy auth to any environment +with a large `refresh_tokens` before building the index concurrently there. + +**Wave 1 of the roadmap (2026-09-04 21:35Z onward):** five Opus agents in isolated worktrees: 3.1 grace window +(orca-cloud), 4.1+2.3 relay locks + pool timeout, 3.2+4.3 desktop refresh/jitter, 5.1+5.4 observability, +2.1 private IP (plan only, both repos). First back: stablyai/orca PR #18717 (crash alert + dashboard). Its key +finding: cell exits log to `cos_system` with uppercase `jsonPayload.MESSAGE` and `SYSLOG_IDENTIFIER=docker`, +so every earlier `jsonPayload.message:"container die"` count in this doc that read 0 was querying the wrong +field. Verified: 87 exits 12–13Z on the agent's filter, 0 in the last 6 h. Monitor dry-run 33922255205 +dispatched 21:41Z as the Roll 1 gate. +Dry-run 33922255205 froze at 21:46Z on `signal_missing cloud_sql.backends`. Cause: Cloud Monitoring published +no `num_backends` point for the auth instance between 21:40 and 21:46 (every other minute of the last 100 has +one; measured directly via the timeSeries API). A Google-side publish gap, not a database or monitor defect; +the monitor's freeze-on-missing rule is correct. The 12–13Z monitor failures were a different cause (active +probes reading 0 during the crash cascade). Re-dispatched at 21:50Z. +Dry-run #2 (33922844671) froze at 21:52:21Z on `auth.health observed 0` — verdict read from the state.json +artifact, not the log (the log only prints checkpoints). Auth served `/health` 200 continuously, including the +21:52:05 probe. Cause: the probe requires `/health` AND `/ready` on the first attempt; auth has no `/ready` +(404 by design), so every auth sample takes the forced 11 s retry, and on the third sample the retry fetch threw +at the network layer on the runner (no request reached Cloud Run) and `check()` recorded the exception as +health=false. Neither freeze was fleet health. Fix delegated (relay-ops: a thrown fetch is not a reading; auth +does not require `/ready`). **Sequencing constraint for Roll 1:** monitor evidence must be < 5 min old at +canary dispatch, so the owner's go must precede the dry-run, and a green dry-run must be followed by the +canary dispatch immediately. + +stablyai/orca PR #18719 (3.2 + 4.3, desktop): the replay engine was not the refresh function but +`RelayAuthCoordinator.scheduleRetry`, since `shouldRetryRelayConnectionError` treats any non-HTTP error +(including a refresh `TimeoutError`) as retryable and re-reads the same stored token on backoff. Fix: refresh +gets one 60 s attempt; an ambiguous failure (no status line) records the token and blocks re-sending it for +30 s (bounded, not permanent); definitive 5xx gets exactly one retry after re-reading the store; a 401 on an +ambiguously-attempted token logs `orca_cloud_refresh_possible_replay`. Lease renewal gets ±10 % full jitter +(base shrunk so the latest sample stays ≥ 90 s before expiry); server resets the full 55-min TTL on any rebind +(`host-session-registry.ts:736-743`) so early renewal is free. Verified the retry-path claim and both server +cites against main. + +2.1 private IP: orca-cloud PR #477 (foundation: servicenetworking API, /24 peering range 10.42.128.0, private +network on the instance, `prevent_destroy`; real production plan 3 add / 1 in-place change, staging unchanged) +and stablyai/orca PR #18720 (relay: `relay_cloud_sql_private_ip` variable, conditional `--private-ip` in the +cell startup template; default false renders byte-identical to main). Findings that change the plan: Google +states the private-IP change **restarts the instance** with no in-place path, and it is a one-way door (cannot +disable private IP or remove the network link). The director uses the Cloud Run built-in connector, not the +relay VPC NAT, so it never consumed the exhausted ports and is out of scope. Disabling public IP later breaks +the local proxy workflow and the director. #18720 merges (inert); #477 held for owner decision. + +4.1 + 2.3 relay: stablyai/orca PR #18722. Premise correction: #18521 and #18606 had already bounded and +narrowed most of the fleet-wide lock before today; what remained were the sticky-refresh retry (all 23 rows → +the one pinned row), reservation reconciliation (23 → the 2 involved rows), a dead pool-default fallback, and +an absolute counter write (→ delta with capacity guard). Placement (`assignOnce`) deliberately keeps the +ordered inventory lock: least-loaded selection is fleet-wide and dynamic target-only locking previously caused +cross-cell cycles; converting it to optimistic snapshot + conditional delta is the remaining 55P03 floor and a +follow-up. Pool `statement_timeout` was already 5 s but hardcoded; now env-configurable, `57014` added to the +retryable set (it was terminal before), schema DDL on an untimed max:1 pool. Independently re-ran the new and +adjacent suites here against 55440: 66/66. Harness note: 55440 is not idempotent across full runs (2 +pre-existing failures on a second run); reset the schema between runs. Rollout: director first, watch +`orca_relay_postgres_transaction_exhausted` and `cellInventoryHoldMsP95` before cells. + +#18719 first CI run failed only on `windows-host-job.win32.test.ts` (EPERM on temp-dir cleanup), a Windows +PTY test the PR does not touch and which no other recent run failed on; rerun dispatched rather than waved. + +3.1 grace window: orca-cloud PR #478 merged (not yet deployed; deploy is an owner gate because the startup +schema apply adds a nullable column to `refresh_tokens` with a brief ACCESS EXCLUSIVE). Semantics: within +`ORCA_CLOUD_REFRESH_ROTATION_GRACE_MS` (60 s default, 300 s cap, 0 = off) a re-presented rotated token gets the +SAME successor refresh token + a fresh access token, no revoke, no audit, provided the successor is still the +live head. Third presentation / outside window / revoked family: unchanged (revoke + audit). Successor plaintext +is stored sealed (AES-256-GCM, key = HKDF of the predecessor token; the DB never holds the key). Cost stated +plainly: a stolen token replayed inside 60 s is served once instead of tripping detection; DB-read + stolen +predecessor recovers the successor offline until pruned. Rotation now runs in one transaction (proved by a +forced-INSERT-failure rollback test; the 8-way race alone did not kill the non-transactional mutant). Verified +locally 27/27 incl. the Postgres suite against 55440, and CI ran it on PG 16 and 17 (4/4 each, not skipped). +Deploy wiring: env is set by BOTH Terraform and the deploy workflow, with a test pinning all three sources to +one value. **Pre-existing bug surfaced:** the deploy script strips every env var it does not own, so the +Terraform-set `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` (from #476) silently reverts to the compiled default on each +release. Latent only because both defaults are 30. Follow-up: add it to `authEnvironment` + the workflow env. + +Monitor probe fix: stablyai/orca PR #18723. A thrown fetch (DNS/TCP/TLS/8 s abort) is now "no reading" and is +re-asked once after 1 s; only a second throw is `false`. A non-ok HTTP answer is still `false` with no extra +retry. `latencyMs` is the slowest answering round trip, never a sleep. `requiresReady` is per endpoint: auth +(no `/ready` by design) is judged on `/health` + latency; director and cells unchanged. No threshold or rule +touched; `auth.ready` had no consumer. 81/81 relay-ops tests and 9/9 evidence-script tests locally. The monitor +runs at `main` head, so once merged the next dry-run uses it. + +Applying #18717 (22:10Z): the cell-exit log metric `orca_relay_cell_process_exit` is created; the alert policy +raced descriptor propagation (404) and is being retried. **Not applied, deliberately:** the dashboard. Its +targeted plan drags in `google_logging_metric.relay_snapshot[*]`, and that plan is `32 to add, 21 to destroy`: +the Terraform source adds a `region` label to every runtime metric (`EXTRACT(jsonPayload.region)`) which the +live metrics do not have, and a label change on a log metric is a delete+create. Replacing 21 live metrics +resets their history and would blank the 14 existing relay alert policies during the swap. That is +pre-existing drift in the relay root (unapplied since the region work), not something #18717 introduced. It +needs its own reviewed apply in a quiet window, ideally with the runtime-metric replacement acknowledged as +intentional. Dashboard apply waits on that. + +**Wave 1 closed 22:20Z.** Merged: orca-cloud #478 (grace window); stablyai/orca #18717 (crash alert + +dashboard TF), #18719 (desktop no-replay + jitter), #18720 (private-IP flag, off), #18722 (relay per-cell +locks + pool timeout), #18723 (monitor probe fix). Applied to production: cell-exit log metric + alert policy. +Held for owner: orca-cloud #477 private IP (restart, one-way); the dashboard apply (behind the runtime-metric +label drift); the auth deploy carrying #478; Roll 1. Every wave-1 code change now sits on main un-deployed: +the next relay image build carries #18722 + #18723's monitor runs at main head already; the next auth deploy +carries #478. + +**Landing (2026-09-04 20:50Z–21:02Z, owner: "if you are confident the cloud changes are valid, you can land them"):** + +- Merged: orca-cloud #474, #475, #476; stablyai/orca #18693, #18694, #18698. Neither repo has branch + protection or environment reviewers; `verify` / `cloud-verify` green on main after each. +- Applied to production by targeted saved plans (each plan asserted create-only / exact-attribute before + apply, via `terraform show -json`): 4 relay resources (WAL-checkpoint log metric + 3 alert policies), 8 auth + resources (3 log metrics, propagation sleep, 4 alert policies), and the us-central1 NAT + (`enable_dynamic_port_allocation` false→true, ports 64..4096). Google's docs: switching to dynamic does not + break existing connections when max ≥ 1024 and max ≥ old min; only lowering max or reverting to static is + disruptive. asia-east2 NAT deliberately left for after a US soak. +- Not applied: the untargeted apps-root plan also carries 4 unrelated drifts (`ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` + env on the auth service from #476, a skill log exclusion filter change, skill pressure threshold 16→8, an + artifacts bucket lifecycle rule) and fails on the 1Password Cloudflare data source locally. The foundation + root plans clean (disk 250 / max_wal_size already match). Those drifts belong to whoever runs the next full + apps apply in CI. +- `deploy-auth-production` on main 8034955 (run 33919143723) **succeeded 21:04Z**: serving revision + `orca-cloud-auth-00031-tox` at 100%, previous `00018-4jc`, cap 20, smoke passed on both URLs. First 15 min on + the new revision: 31×200 / 1×401 on `/refresh`, max latency 56 ms, no 5xx. The new + `refresh_token_prune_cursor` table exists, so the new schema applied. +- US NAT soak (21:01–21:06Z): 0 drops, 0 proxy dial errors, 0 cell exits, port_usage 11, sqlMax ~1.07 s. + Asia NAT then applied 21:05:28Z from the pre-verified saved plan (same three attributes). The deploy script strips env vars it does not own, so the Terraform + TTL var will not be on the new revision until the full apps apply lands; the auth code defaults to 30 d. +- Terraform locally needs `GOOGLE_OAUTH_ACCESS_TOKEN="$(gcloud auth print-access-token)"`; ADC is stale. + +**Alerting + NAT follow-ups (19:58Z, superseded by the landing block above):** + +- stablyai/orca PR #18693 (`relay-nat-ports-and-sql-alerts`): both relay NATs switch to dynamic port + allocation (64–4096 per VM); new relay-channel alerts for the Cloud SQL WAL checkpoint loop (log metric on + `checkpoint starting: wal`, > 3 per 5 min), Cloud SQL disk > 70%, and NAT `OUT_OF_RESOURCES` drops. No + existing workflow applies these resources; the PR body carries the targeted plan. +- orca-cloud PR #475 (`auth-observability-alerts`): log metrics + policies for auth refresh 401 (> 100 per 5 + min; Sep 3 baseline 20–80 per hour), 429 (> 20 per 5 min; baseline 0), 5xx (> 10 per 5 min), and Cloud Run + p99 latency > 10 s. Production routes to the relay Slack channel. +- Desktop stale auth-status fix: stablyai/orca PR #18694 (`desktop-cloud-session-revoked-status`). Main pushes + an auth-status-changed IPC when a 401 clears the session; panes re-fetch on mount; the pairing notice says + "Your Orca account session expired. Sign in again to use Orca Relay" and hides Retry. StrictMode regression + test verified red on the old guard. Does not help desktops already revoked today (session cleared before + this code); it fixes every future revocation. +- orca-cloud PR #476 (`auth-refresh-token-pruning`): batched `refresh_tokens` pruner as a scheduled Cloud Run + job (revoked rows kept 30 d, rotated rows 60 d against a 30 d TTL, 5k-row batches, 200 ms pauses, persisted + cursor, per-run budget) plus a 10 s `statement_timeout` on the auth request pool with schema DDL on an + untimed connection. Merges cleanly onto #474 and does not need its index (walks the primary key; + EXPLAIN-asserted no seq scan). CI ran the Postgres integration tests for real on PG 16 and 17. Ships + `auth_token_pruner_enabled = false` in both environments: enabling needs an image digest from a build that + contains the new entrypoint. Operating rules once enabled: monitor the run summary's `stopReason` and + `deletedRows`, not the exit code (a run that only ever times out exits 0); ~48 M rows drain in ~10 days at + 200k/hour; deleting them leaves dead tuples, so the 16 GB is not reclaimed without a separate VACUUM FULL or + pg_repack pass, which is its own change. +- Phone-side copy when the desktop is signed out: stablyai/orca PR #18698 (`phone-desktop-signed-out-reason`). + Real path traced: the director resolves the phone to the host's last cell (durable assignment row), and the + cell's `acceptClient` rejects with 4404. The only additive slot every shipped peer tolerates is the WebSocket + close *reason* (relay-hello and resolve schemas are zod strict; a new close code drops old phones off the + host-offline cadence). Desktop closes its control with reason `signed-out` only when the cloud session is gone + (null context after a 401, or explicit sign-out); quit and relaunch stay reasonless. Cell remembers it per + host for the dormant-assignment TTL, forgets on re-auth, and echoes it as the 4404 close reason; phone + renders "Desktop signed out — sign in to Orca on your desktop to reconnect" with the same retry cadence. + Old×new matrix in the PR body; nothing changes for any old peer. Merges cleanly with #18694. + +## What actually blocks the roll now (12:58Z summary for the owner) + +0. **Cloud NAT ports** (Finding 11, found 12:55Z): every us-central1 cell reaches Cloud SQL's public IP + through a NAT with the default 64 ports/VM; port_usage pinned at 64 and 1,514 dropped SYNs to + Cloud SQL:3307 in one 4-min window. This is the 2 s connect stall that kills old-image cells and is + still active after the disk loop broke. Fix: `min_ports_per_vm = 1024` (or dynamic allocation) on + `google_compute_router_nat.relay_gce` in `cloud/infra/terraform/relay-gce-foundation.tf`, targeted + apply; durable fix is a private IP on the Cloud SQL instance. Online, no VM restart. +1. **Cloud SQL disk** (Finding 10): 49 GB PD-SSD saturated since 11:58Z, checkpoint loop, fleet-wide + 4–6 s stalls every ~45 s. Fix: bigger disk and/or `max_wal_size`. Owner: `stablyai/orca-cloud` + `infra/terraform-foundation/database.tf` `google_sql_database_instance.auth` (no `disk_size`, + `disk_autoresize`, or `database_flags` set today, so Terraform is at defaults: 10 GB initial, autoresize + grew it to 49 GB). Add `disk_size = 200` (+ `disk_autoresize = true`) and optionally + `database_flags { name = "max_wal_size" value = "4096" }`; production tfvars are + `infra/terraform-foundation/environments/production.tfvars`; applied by `deploy-production.yml` in + that repo. Online, no restart for disk; `max_wal_size` is also a non-restart flag. Note Terraform + `disk_size` below the live 49 GB would be a destructive shrink, so 200 is safe and 49 is the floor. **This is now the first thing to do**; nothing else can pass a + 15-min gate while it persists, and it is also what is killing the old-image cells several times an hour. +2. **Old cell image** (Finding 6): dies on every stall. Fixed by rolling 519f4914 (canary inputs ready). +3. **Gate policy**: `directorErrors: 0` and per-cell health probes freeze on any single stall. Recalibrate + after 1 and 2, or bypass by hand for the canary. + +## Plan agreed with the owner (2026-09-04 ~06:45Z), in execution order + +Owner: "feel free to improve operations to make things more effective ... continue driving everything e2e +until this process is complete." Owner has had multi-day experiences with cell rolls and does not want a +9-hour sequential roll. + +1. **Lock-removal PR** (root cause). *Status 08:55Z: pushed as branch `relay-single-row-reservation` + (2 commits). Opus adversarial review found one real defect: `acquireActivity` moving a client-chosen + activity id across cells locked the old cell's row before the new one, cycling with placement's + ascending inventory lock (reviewer reproduced it as paired 55P03s on real Postgres; no 40P01 because + lock_timeout == deadlock_timeout == 1 s). Fixed with `lockCellRows` (ordered, 500 ms bound); census now + fails on any inline `relay_cells FOR UPDATE` outside the named helpers. Three-cell Postgres test moves + an activity high->low while the target row is held; 5/5 revert-mutants fail it. 480 SQLite tests + + tsc green. Also fixed a pre-existing test leak (`relay_cell_connection_snapshots`) that made + `assignment-control-supersession-postgres` fail on reruns. Reviewer re-verified 65569be3de: cycle + repro completes in 7 ms (was 1022 ms + paired 55P03); no remaining out-of-order pair in the store; + flagged two evasions in the new census guard, closed in the third commit (whole-statement scan, + covers query() too, mutation-checked with both evasions). Headroom Postgres test's one failure is + pre-existing on main (verified by swapping in main's store).* Make `activateControl` superseded-control cleanup, `acquireActivity` + existing-lease branch, and `changeActivity` use the existing single-row + `adjustCellReservationAtomically` instead of the 23-row `lockCellInventory`. Keep the global lock only + for placement (`resolve`/assignment) and sweeps. Real-Postgres contention test on port 55440. +2. **Faster same-cap rollout workflow.** (a) paced drain instead of `graceMs: 0` so a cell's ~800 hosts + re-dial over minutes, not one second (director cap is 5 x 80 = 400 in-flight); (b) cells in a batch run + in parallel once drains are paced; (c) post-canary batches use a short freshness check instead of a new + 15-min dry-run, since the in-job safety recheck already runs before each drain; (d) job timeout > 75 min. + Target: 22 cells in ~6 batches x ~25 min. +3. **Build image** with (1) merged, then one roll of the fleet with (2). Asia cells c27/c28/c29 first. +4. Re-tighten the monitor retries bar; recalibrate the Terraform exhausted alert. +5. Consider deleting the 55-min control lease rebind entirely (no recorded reason; liveness is the 75 s + watchdog + 90 s activity lease). Separate PR after (1) so its effect is measurable. + +## Faster same-cap rollout: design (step 2 of the plan), from reading the real limits + +What actually bounds parallelism today (measured on the c7 canary, run 33843071283): + +| step | c7 duration | bound by | +|---|---|---| +| prechecks (recheck, backend init, resolve, verify) | 43 s | none | +| isolate + drain + transition wait | 7 min | drain is `graceMs: 0`; `verify-relay-capacity-transition --activity restart-safe` polls until leases drain | +| Terraform template + MIG recreate + wait-until stable | 8 min | GCE recreate; per cell, independent | +| verify new incarnation + trust proof + restore | 1.5 min | none | + +Real constraints: (1) the director is 5 x 80 = 400 in-flight `/v1/assign`; a `graceMs: 0` drain of ~800 +hosts pins it at cap for ~2 min (observed 79.75/84.75 p99). (2) `production-cloud-sql-rollout` lease and +workflow concurrency group serialise the whole run, by design, and the per-cell job shares it via +`holder-key`. Nothing else forbids parallel cells. + +Changes, smallest first: +1. **Paced drain.** `HostSessionRegistry.drain(graceMs)` already sends `drain {graceMs}` and closes each + session after `graceMs`, but the desktop's `handleDrain` re-dials immediately regardless of graceMs + (`relay-origin-pool.ts:150-162`), so graceMs only delays the *close*, not the stampede. Fix on the + cell: stagger the drain *send* across sessions over a window (e.g. 800 sessions over 120 s = ~7/s), + which needs no desktop change and works for every desktop version in the field. New admin body field + `spreadMs` (optional, default 0 keeps today's behaviour); canary script passes `spreadMs: 120000`. + Requires the cell to be on an image with the change, so it applies to batches after the first + post-lock-fix roll, not to this one. +2. **Parallel cells in a batch.** In `cloud-deploy-relay-production-same-cap.yml` make `cell_2..cell_4` + `needs: [gate]` instead of chaining, gated on the same evidence (drop the `+75 min x wave-index` + allowance, it exists only because of chaining). Each job already takes the rollout lease with the + run's `holder-key`, so they re-enter it rather than fail. With paced drains, 4 cells x ~800 hosts + over 120 s is ~27 dials/s, well under the director cap. Raise `timeout-minutes` to 90. +3. **Post-canary batches skip the 15-min dry-run.** The in-job "Recheck aggregate SQL, pool, + reconnect, migration, and selector safety" step (`pnpm incident:relay-preflight`) already runs a + live one-shot check before each drain. For `batch-apply` with a sealed `canary-run-id` from the + same commit, accept a dry-run of any age (the canary's) plus that live recheck; keep the 15-min + requirement for `canary-apply`. Change lands in `relay-monitor-evidence.mjs verify-authority` + + `relay-production-same-cap-wave.mjs` + their node:test suites. + +**Correction after reading the cell job (07:35Z):** (2) parallel cells is not a flag flip. Each cell job +asserts the exact selector generation `expected + 2 x wave-index` and exact memberships derived from +predecessors having completed (`ISOLATED_*`/`RESTORED_*` in the job, `applyExactAdmissionSelector` +compare-and-swap), and all cells share one Terraform state lock. Making that concurrent means a batch-level +isolate/restore in the gate and a rewrite of the 650-line job's expectations. That is the multi-day trap +the owner described. Deferred. + +What is cheap and removes most of the wall-clock: (3). The per-batch 15-min dry-run costs 15 min each +*and* fails ~50% of the time on old-image crashes, which is where hours go. Implement: `batch-apply` with a +verified canary authority accepts a passed dry-run up to 6 h old and may re-use one already consumed +(the consumed-marker check exists to stop replaying stale evidence; the canary binding plus the in-job +live preflight at drain time replace it). Files: `relay-monitor-evidence.mjs` (`--after-canary`), +`incident-live-preflight-cli.ts` (same flag), the same-cap workflow + job, and both test suites. +Revised expectation: 22 cells = 6 sequential batches x ~70 min = ~7 h wall-clock but *unattended-safe* +and with one dry-run total, versus today's 6 dry-runs at ~50% each. (1) paced drain rides the lock-fix +image. + +## Recommended next steps (superseded by the plan above; kept for history) + +1. Resolve the gate decision above, then: monitor dry-run -> c7 `canary-apply` only -> verify -> stop. + Each rolled cell leaves the Finding 6 crash class. +2. Merge #18565; publish; a later same-cap roll carries it to cells. +3. Remove the global inventory lock from per-connection paths (`acquireActivity` existing-lease + branch, `activateControl` superseded-control cleanup, `changeActivity`) by using the existing + `adjustCellReservationAtomically` single-row update. Own PR, after the roll. +4. Recalibrate the Terraform alert `relay_postgres_retry_exhausted` to 300/300 s (observability root). +5. Whether to raise `relayPostgresRetries` is a human call; the data is in Finding 5. + +## Canary blast radius (read before dispatching c7) + +- What `canary-apply` does to c7, in order: isolate (selector -> migration-only, no new + assignments), `/v1/admin/drain graceMs:0` (every control on c7 re-dials the director and is + reassigned), Terraform template + MIG update to the target image, wait stable, verify new + incarnation + exact digest + protocol, prove per-host trust, restore c7 to general admission. + On any failure c7 is left isolated (migration-only) with rehome disabled; nothing else is touched. +- c7 at 05:20Z: 788 controls, 5 splices, 800 connections. So ~790 desktops re-dial once. The fleet + already absorbs this exact event 201 times / 48 h uncontrolled (Finding 6); the controlled version + isolates first, so no new assignment lands on c7 mid-roll. Expect a director concurrency blip, not + a freeze-class one (six cells at once gave 85; one cell should stay well under 64). +- Precedent: the identical workflow (pre-move, in orca-cloud) ran 9 successful `apply` canaries and + batches on 2026-08-27 (last: c20 -> 5aedbca5). Its failures that day all stopped at the read-only + "Recheck aggregate SQL..." or "Require durable rehome disabled" step, before `MUTATION_STARTED`. + The moved copy in this repo has one run: the read-only `verify` of c7 (passed, including WIF auth). +- c7 side note: MIG autoheal recreated the c7 instance four times on 2026-09-01 08:02-08:42 PDT + at ~13 min spacing. Same crash class as Finding 6 (health check failing during restart loops). + +### Canary observed effect (c7 drain, 2026-09-04 06:10Z) + +- c7 807 controls -> 0 between 06:08:52Z and 06:10:52Z. Director `/v1/assign`: 200s 32 (06:09) -> 2628 (06:10) + -> 340 (06:11); 5xx 1969 (06:10) -> 31 (06:11). Director max-concurrency p99 7.9 -> 79.75 (06:10) -> 84.75 + (06:11), i.e. at the Cloud Run cap of 80 for ~2 min. My pre-dispatch estimate ("well under 64") was wrong. +- Confounder: c10 (us-central1, instance 2803000337345335589) crashed 06:09:56Z on the old-image class + (Node.js banner + container die), so ~1,600 hosts re-dialed in the same minute, not ~800. Coincidental; + the fleet has one of these every ~15 min. +- Recovery: 06:13 903 / 06:14 1471 assign 200s from 640 distinct desktop IPs; 503s 78 -> 183 -> 29/min. + No cell crash 06:12–06:16Z. Drain step passed ~06:16Z; template/MIG apply started. +- 06:16:03–06:17:08Z, during c7's template apply (not its drain): c27 (x4) and c29 (x3) crash-looped on the + old-image pg-pool connect timeout in `beginProof`, both MIGs autoheal-recreated (c27's second recreate in + 40 min). Fleet 23 -> 21 reporting cells, controls 13286 -> 12462, assign 503s 1000/min at 06:17, director + concurrency p99 74.8. Cloud SQL CPU 0.70 max, backends 174 max (bar 250). Same multi-cell pattern occurred + at 01:31Z (4 cells) and 04:47Z (5 cells) with nothing rolling; the c7 drain's SQL load 6 min earlier may + have nudged the pool timeouts but the class is pre-existing. c7 MIG RECREATING onto new template + `…20260904061618…` = the expected image swap. +- 06:20Z: 849 assign 503s. Closes 06:19:30–06:21: 162x1006 age<5min (hosts bouncing off the recreating + c27/c29), 73x4408 + 53x1006 in the 50-min age bin (Finding 3 rotation cohort). Not roll-caused. + c7 MIG `recreating=1` on the new template since 06:16:18Z; c27 and c29 MIGs also RECREATING (autoheal). +- 06:23:16Z c7 instance restarted in place (MIG RECREATE keeps name/id relay-c7-bwjc / 4545742188814054238), + pulled `relay@sha256:85bf6799…` 06:23:37Z, listening + readiness true 06:23:42Z. Apply step passed 06:24Z; + verify step running. Isolate -> ready on new image took ~14 min end to end. +- Post-restore c7 on new image (06:25:42–06:26:42Z): controls 143 -> 273 -> 377 refilling, sqlQueries + ~1,500/30 s, `sqlLatencyMsMax` 518 -> 1003 -> 1155 ms, still 55P03 `cell-inventory` retries. So the new + image alone does not remove lock waits; the request-path 500 ms cap from #18521 applies to the director's + paths, and cell-side `acquireActivity`/`activateControl` still ride the global lock (step 3 in next steps). + Watch: does c7's sqlLatencyMsMax settle below the old 1.0–1.2 s pin once refill finishes, and does c7 stop + appearing in `container die` (the real win: guardSessionTask). +- 08:25Z (2 h after restore): c7 817 controls, 0 crashes since 06:25Z. Fleet crashes last 2 h: c27 x6, + c28 x5, all old-image Asia cells. The new image stops the crash class as predicted; it does not move + lock latency (c7 sqlLatencyMsMax 1005 ms), which is #18606's job. +- Implication for the batch phase: every drain will push director concurrency past the monitor's 64 bar + for ~1-2 min. The batch job rechecks safety *before* it drains (read-only step), so that is fine per wave, + but never run a monitor dry-run concurrently with a wave, and prefer batches of 2 over 4 until the fleet + is on the new image and the crash class is gone. + +## Post-merge dispatch plan for #18606 (image -> director -> cells) + +1. `gh workflow run cloud-publish-relay-production.yml --ref main -f mode=publish` (after the squash lands + on main). Resolve the digest by tag, never by parsing the log (it mixes relay and fence-broker digests): + `gcloud artifacts docker images describe us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay:sha- --format='value(image_summary.digest)'`. +2. Director: `gh workflow run cloud-deploy-relay-production-director.yml --ref main -f image-digest= + -f regional-placement-mode=preserve -f prune-incompatible-revisions=false -f expected-rehome-generation=12 + -f bootstrap-runtime-identity=false -f predecessor-image-digest=` + (no monitor evidence needed; requires rehome disabled at gen 12, which it is). Last run 33826514754 used + the same shape. Watch director `orca_relay_postgres_transaction_retry` per minute before/after. +3. Cells: same-cap `verify` c7 with target=, rollback=85bf6799; fresh dry-run; `canary-apply` c7; + then batches (3 per batch, Asia c27/c29/c28 first). Each batch: new dry-run unless the batch-reuse + change (design section above) has shipped. + +## Finding 8 (2026-09-04 08:40Z): ten-cell crash cascade during the director deploy, not caused by it + +Timeline: candidate revision 00570-siv created 08:38:39Z, first log 08:39:20Z; traffic still 100% on +00565-fes through 08:43 (assign logs by revision). Cell crashes: c28 (5031087219978409220) looped 08:37:55– +08:40:07 (9x), then at 08:40:20–08:40:45Z **ten** instances died within 25 s (c10 2803…, 5110…, 532…, 5464…, +7536…, 7726…, 8671…, 8928…, 8966…). All old-image `beginProof` pg-pool timeouts. Fleet controls 13,423 -> +6,157 by 08:43; assign 503s 3,912 (08:42) and 4,624 (08:43) per minute, director concurrency 85 (cap 80), +Cloud Run autoscaled 5 -> 10 instances, Cloud SQL CPU 0.55 -> 0.99. Deploy finished cleanly at 08:45Z with +the new director taking the tail of the storm; by 08:46 503s were ~30/15 s, controls 7,913 and rising, +director lock retries 29/min (vs 105–157/min pre-deploy) and exhausted 2/min (vs 65/min at 08:36). +Same class as 01:31Z (4 cells) and 04:47Z (5 cells) today; this was the biggest. c7, on the new image +since 06:25Z, did not crash. What triggered the pool timeouts fleet-wide at 08:40 is not established; Cloud +SQL CPU was 0.78–0.88 in the minutes before, the highest of the day, so the cells' 2 s connect timeout is +the plausible tipping point under a busy database. Every cell still on 5aedbca5 remains exposed to this. + +## Finding 9 (2026-09-04 08:56Z): #18606 on the director cut lock retries ~10x + +`orca_relay_postgres_retries` per 5 min, director only: 08:21–08:41 windows 419–689 (old image, incl. the +crash storm); 08:46/08:51/08:56 (new image 519f4914, refilling ~7k hosts): **61 / 69 / 54**. Exhausted: +104–178 -> **11 / 14 / 12**. Inventory hold p95 ~200 ms, max 255 ms, ~366 holds/min. Cells (still old +image) 17–44 -> 0–3, because the director no longer holds the 23-row lock on their behalf. This is the +first direct measurement of the root-cause fix under real load. Cloud SQL CPU peaked 0.99 during the +cascade and is decaying (0.86 at 08:55); the monitor freezes above 0.80, so no dry-run until it clears. + +Fourth cascade 09:00:12–09:00:18Z: c23, c8, c16, c26, c22 (five cells, 11 container-die events in 6 s, +all `5aedbca5`, exitCode 1, Node banner, pg-pool `client closed the connection` burst right before). Cloud +SQL CPU 0.84 -> 0.78 in the preceding minutes, director concurrency 18–22 (idle), so this one fired +*without* a database or director spike. Fleet had just recovered to 13,015. Cadence today: 01:31 (4), +04:47 (5), 08:40 (10), 09:00 (5), 09:31 (c13, c23), 09:34 (c23 again, c14, c20, c9; c14/c20 crash-looping), +09:39 (c21, c24), 09:55 (c16, c8), 09:59 (c20), 10:05 (c8, c20), 10:19 (c16 stalled, no crash), then a 58-min +lull, 11:04 (c9; c28 died 13x in 4 min, autoheal recreate 11:09Z, its 3rd recreate today), 11:17 (c10, c28 +again, c22, c23, c14 x9 looping; 23 dies in ~90 s; fleet 13.3k -> 10.8k), 11:31 (c14, c23, c25, c15, c24, c19), 11:34 (c20, c26, c29 x4, c14, c27 x3, c25; fleet 13.1k -> 10.3k). +Three cascades in 17 min. 11:38–11:45 c27 crash-looped 17x and c28 4x (Asia cells), c29 recreating. +11:59 (c21, c9, c10, c23), 12:02 (c19; 4,109 assign 503s that minute, mostly hosts bouncing off the +recreating cells, code 1006 age<5min x217), 12:09–12:12 (c19, c27 x6, c28 x5, c13, c22, c15, c26, c14; +8 cells, c27/c28 recreating again). Cloud SQL CPU 0.62–0.85 through it. 12:20 (six more cells). Cascade +cadence since 11:00 is now ~every 8 min; the waiter has held correctly the whole time and there has been +no dispatchable window. Loop continues unattended; findings stop logging each cascade from here unless the +class changes. Every cell that has died today is on 5aedbca5; c7 (85bf6799, +5.5 h) has not. Cell dies per hour today: +01Z 5, 02Z 7, 03Z 4, 04Z 7, 05Z 4, 06Z 9, 07Z 11, 08Z 30, 09Z 26, 10Z 2, 11Z 68+ (to 11:42). +Director concurrency pinned at 85 for 09:32–09:33; 503s 4,141 and 4,396 per minute. 09:39: c21, c24 +(2,870 503s). Crashes per instance 08:10–09:40Z: c28 x14, c27 x5, c23 x5, c22 x4, c14 x4, c13/c20 x3, +then c26/c9/c24/c16/c8 x2. Mean gap between cascades since 08:40: ~12 min. Every 15-min gate attempt +now has well under even odds; the c7-style canary that ends this needs a gate it can pass. The old image is now cascading roughly hourly regardless of load; the +only cell on a fixed image (c7) has 0 crashes in 2.5 h across all four. + +Director 500s: 4 in the 09:00 window, all 2.0 s latency on `/v1/assign` or `/v1/resolve` = pg-pool connect +timeout surfacing as a 500. Pre-existing (Sep 3: 03h/08h/16h one each, same 2.0 s shape; 06:09Z today on +the old image during the c7 drain). The monitor's `directorErrors: 0` bar freezes on any of these, so a +dry-run needs a 15-min window with none; at ~1 per cascade that is a real but modest constraint. + +**Gate observation (09:26Z):** `directorErrors: 0` counts every non-503 5xx on the director, including +the monitor's own admin calls. The director on 519f4914 still sees an occasional 2.0 s pg-pool connect +timeout (~1 per 20 min under today's Cloud SQL load), which surfaces as a 500 on whichever request drew +it. Two consecutive dry-runs (#7, #8) froze on exactly this: one, isolated, 2 s 500. That bar was set for +"unexpected director 5xx"; a single connect timeout that the client retries is not an incident. Candidate +recalibration (own PR, not done): `directorErrors` 0 -> 2 per 5 min, or exclude the monitor's own +user-agent. Not changing it unasked; noting that at ~3 per hour the 15-min gate passes ~1 in 2 attempts. + +**Did the director deploy make cells crash more? (checked 09:45Z)** Cell `container die` per 30 min: +06:00 9, 07:00 2, 07:30 9, **08:30 30** (director candidate 08:38, traffic 08:43–08:45; the 10-cell burst +was 08:40:20, before the move), 09:00 11, 09:30 12. Per hour today 05:4 06:9 07:11 08:30 09:23 vs Sep 3 +same hours 2/7/8. So today is 2–3x worse than yesterday and was rising before the deploy; after the deploy +it is ~11–12 per 30 min, in line with 06:00–07:30. Cloud SQL backends (~230 max) and new connections +(~5k/30 min) are flat across the deploy. Latest crash (c21 09:39:11) is `Connection terminated due to +connection timeout` with cause `Connection terminated unexpectedly` in `verifyCellAssignment` <- +`beginProof`, the same unhandled path. Conclusion: no evidence the deploy worsened it; the old image's +crash rate simply climbed all day. Director lock retries stayed ~10x lower after the deploy. + +**Checkpoint-phase check (10:00Z, negative result):** Postgres checkpoints complete every 5 min at ~:07. +Cell crashes bucketed by phase within that 5-min cycle show a mild :00–:29 s cluster today (22 of 103) +that is absent on Sep 3 (7 of 114), so checkpoints are not the trigger. Disk write bytes in cascade +minutes are at or below the median except 09:00. Cloud SQL memory 0.47, transaction rate flat. The +09:55 stall (11 director + 4 cell pg-connect timeouts in the same 4 s) came with `could not obtain lock +on row in relation "relay_cells"` from a NOWAIT sweep at 09:55:36, i.e. someone was holding the full +inventory at that moment. On the new director that can only be placement or a sweep; on the old cells it +is still every rebind. What stalls *connections* (not locks) for 2 s fleet-wide remains unexplained; +Cloud SQL is `db-custom-4-15360` REGIONAL PD_SSD 49 GB at 0.5–0.75 CPU when it happens. + +**Stall census (10:01Z):** 33 pg-connect-timeout stall events today (clusters of timeouts < 20 s apart). +Before 08:35 they were 1–9 timeouts each and 10–60 min apart; from 08:35 the big ones are 16, 22, 21, +17 timeouts and 5–30 min apart. No second-of-minute phase (start seconds spread across all buckets), so +not a fixed timer. Cloud SQL backends by state at 09:55: active peaked 42 at 09:52, idle-in-transaction +≤ 10, nothing near the 400 ceiling; memory 0.47; disk normal. Each stall is a few seconds where *new* +connections to Cloud SQL (via the auth proxy socket) time out at the 2 s `connectionTimeoutMillis`, +hitting every process that happens to need a fresh pool connection in that window. Old-image cells die +on it (unhandled), new-image director logs a 2 s 500 and continues. Root cause of the stall itself is +outside the relay code (Cloud SQL proxy or instance); not chased further here. + +## Finding 10 (2026-09-04 12:40Z): Cloud SQL disk write saturation since 11:58Z is driving the stalls + +`orca-cloud-auth-db` is `db-custom-4-15360` on a **49 GB PD-SSD** (81% used). PD-SSD performance scales +with size: 49 GB gives roughly 1,470 write IOPS and ~23 MB/s write throughput. Measured: + +| | before 11:58Z | 11:59Z onward | +|---|---|---| +| disk write MB/s | 4–6 | **30–50** (over the ~23 MB/s cap) | +| disk write IOPS | 500–800 | 800–1,475 (at the ~1,470 cap in 11:59, 12:15, 12:24, 12:34) | +| checkpoint `sync=` | 0.07–0.2 s (Sep 3 max 0.65 s, 290 checkpoints) | 2–20 s; 27 of 39 checkpoints in 12Z were >= 2 s | +| checkpoints per hour | 12 (timed, every 5 min) | 39 (WAL-triggered, every ~45 s; `write=` fell from 270 s to 30 s) | +| Cloud SQL CPU / memory | 0.5–0.8 / 0.47 | same (not the bottleneck) | + +Every 4 s+ fleet-wide SQL stall since 11:04 (11:04, 11:17, 11:31, 11:34, 12:09, 12:10, 12:18, 12:20, +12:30) sits inside a slow checkpoint `sync` window; the 12:30:49 checkpoint synced 5.88 s (longest file +5.47 s), matching the 12:30:02–41 stall. During fsync the WAL writer stalls and every session waits, which +is why the stall hit all 23 cells and the director at once regardless of the relay lock changes. The +old-image cells then die on the pool timeout; the new image survives. What raised write volume ~8x at +11:58Z is not established (autovacuum ran on every relay table 11:55–11:57 and checkpoints are being +forced by WAL volume, so a write amplifier inside Postgres is the leading candidate; relay transaction +rate and Cloud SQL network bytes were flat). This is the first cause found today that is *upstream* of +the relay code and it explains the afternoon acceleration (11Z 68 dies, 12Z 47 by 12:34). + +Corrections after digging (12:45Z): relay query volume, renewals, reconnects, and assignments per 5 min +were **flat** across 11:58 (sqlQ ~330k, renewals ~115k), so the relay did not start writing more. WAL +recycling per checkpoint went 7 -> 10–11 files (16 MB each) at 45 s intervals, i.e. WAL output rose from +~0.4 MB/s to ~4 MB/s while data-file writes rose to 30–50 MB/s; checkpoints switched from `time` to `wal` +triggered at 11:58:24. No Postgres slow-statement or "checkpoints too frequently" lines. This is +write amplification inside Postgres (full-page writes after each of the now-frequent checkpoints on +hot pages, plus autovacuum on every relay table each minute) on a disk too small for its IOPS ceiling, +not new relay load. Instance label `managed_by=terraform`, created 2026-07-09; the instance resource is +**not** in `cloud/infra/terraform` (only the database, user, and secret are, via +`local.relay_database_instance_name`), so it lives in the other Terraform root (orca-cloud, per +[[orca-cloud-terraform-split-findings]]). `storageAutoResize=true` with limit 0, so Cloud SQL will grow +the disk only when it fills, not when IOPS saturate; disk is 81% full. + +Onset precisely: the 11:55:37 `time` checkpoint wrote 67,258 buffers (10.5% of shared_buffers, the +day's largest) over 163 s and completed 11:58:24. Every checkpoint since has been `wal`-triggered at +~45 s spacing (`max_wal_size` reached), each writing 13–20k buffers with 9–11 WAL files recycled. This is a +self-sustaining loop: a checkpoint completes -> every subsequent write to a hot page emits a full-page +image into WAL -> WAL fills `max_wal_size` in ~45 s -> next checkpoint -> repeat. The relay's hot rows +(`relay_cells`, `relay_assignments`, activity leases, cell runtime) are updated tens of thousands of +times a minute, so full-page-write amplification is large. Before 11:58 the 5-min timed checkpoints kept +WAL well under the limit; a one-off larger checkpoint tipped it over and the disk's write ceiling keeps +it there. Query Insights: io_time +30% in the 12:00 bucket, lock_time flat. + +**Owning workflow / mitigation (not applied):** raise the Cloud SQL data disk (PD-SSD IOPS and MB/s scale +linearly with GB; 49 -> 200 GB roughly quadruples the ceiling, online, no restart) in the Terraform root +that owns `google_sql_database_instance` for `orca-cloud-auth-db`, applied through that root's workflow. +A second, flag-level lever is raising `max_wal_size` (default 1 GB) so timed checkpoints resume; that is +also a Cloud SQL instance setting in the owning Terraform root. Per the standing rule, not applied from +this session. Until then the fleet-wide 4–6 s stalls recur on +every slow checkpoint sync, the old-image cells die on each one, and no 15-min gate window will exist. + +## Finding 11 (2026-09-04 12:55Z): **Cloud NAT port exhaustion** on the us-central1 cells is the second stall class + +`google_compute_router_nat.relay_gce` (us-central1, `AUTO_ONLY` IPs, no `min_ports_per_vm`, no dynamic +port allocation, i.e. the default **64 ports per VM**). `router.googleapis.com/nat/port_usage` per VM +hit **64 = the cap** in exactly the minutes the cells' Cloud SQL proxies logged `dial tcp +35.188.82.89:3307: i/o timeout` (12:20–12:22, 12:41–12:43, 12:51–12:53), and +`nat/dropped_sent_packets_count` went 0 -> 56/552/590, 82/272/133, 395/1565/1842 in those same minutes. +Hourly: port_usage max was 25–50 all of Sep 3 and until 10Z today, 64 in 11Z and 12Z; dropped packets 0 +until 11Z (219), then 5,491 in 12Z. Open NAT connections rose 400–600 -> 815–874. Every cell's Cloud SQL +traffic egresses through this NAT to the instance's public IP (the instance has no private IP: +`ipv4Enabled=true`, `privateNetwork` unset). When a VM's 64 ports fill, new TCP SYNs to 3307 are dropped, +the proxy's dial times out, and the relay pool's 2 s `connectionTimeoutMillis` fires: that is the exact +2 s stall the old image dies on and the new director surfaces as a 500. The dial timeouts hit c7 and c8 +hardest because they carry the most controls and open the most DB connections. + +What raised port demand today: each old-image crash re-opens a full pool through fresh NAT ports, the +autoheal recreates do the same, and the 55P03 retry storms keep more connections mid-transaction, so +crashes and NAT exhaustion feed each other. This is why the afternoon accelerated even after the disk +loop broke at 12:39. + +**Owning change (not applied):** `cloud/infra/terraform/relay-gce-foundation.tf` +`google_compute_router_nat.relay_gce` (this repo): set `min_ports_per_vm = 1024` (or enable +`enable_dynamic_port_allocation = true` with `max_ports_per_vm = 4096`) and, if needed, add manual NAT IPs +(each IP supplies 64,512 ports across VMs). Online change, no VM restart. The durable fix is giving the +Cloud SQL instance a **private IP** and pointing the proxy at `--private-ip`, which takes DB traffic off +NAT entirely; that is a Cloud SQL instance change in the orca-cloud foundation root plus a startup-script +flag here. Per the standing rule, not applied from this session. + +Direct proof: `resource.type="nat_gateway" AND jsonPayload.allocation_status="DROPPED"` shows **1,514 +dropped allocations to 35.188.82.89:3307** in 12:50–12:54 alone, every one of them the Cloud SQL public +IP. The NAT has zero manual IPs (AUTO_ONLY) and no port settings in Terraform, so it is at Google's +default 64 ports/VM. No workflow in this repo applies `relay-gce-foundation.tf` broadly (the roll +workflows apply cell templates with `-target`), so the NAT change needs a targeted apply of +`google_compute_router_nat.relay_gce`, which is an owner-run Terraform step. + +Original write-up of the symptom before the NAT correlation follows. + +The 12:50:30–12:50:50 stall (every cell 3.7–3.9 s SQL max, six old-image cells died) happened with +checkpoints healthy (85 ms) and disk at 6 MB/s, so it is not Finding 10. The cells' Cloud SQL Auth Proxy +logged `failed to connect to instance: dial error: dial tcp 35.188.82.89:3307: i/o timeout`. Count of +those per hour today: 08Z 1, 11Z 15, **12Z 416**; all of Sep 3: 4. Cloud SQL `up`/backends/connections +did not blip. So new TCP connections to the instance's public IP on 3307 are timing out from the cells' +proxies in bursts, which is exactly the "2 s connect timeout" the old image dies on. Query Insights for +12:49–12:54 attributes 1,380 s of lock wait to the placement CTE (`WITH assignment_state AS +MATERIALIZED …`) and 469 s to the single-row reservation UPDATE: the lock queue is the *consequence* of +connections stalling mid-transaction, not the cause. Not chased further; candidates are the proxy's +connection churn under the crash loops (each recreated cell opens a fresh pool) and the instance's +public-IP path. Relay code cannot fix this; it is Cloud SQL / network. Dial timeouts by minute today: 12:20 24, 12:21 +66, 12:41 22, 12:42 6, 12:51 160, 12:52 137, i.e. bursts of 20–160 s each, and they hit c7 (new image, +89 today) and c8 (93) hardest, so it is not the old image's connection churn either. Cloud SQL `up`=1 +throughout. The proxy dials the instance's public IP `35.188.82.89:3307`; a burst of i/o timeouts to a +healthy instance points at the path (public-IP egress / NAT / proxy connection limits), not at Postgres. +That is the same 2 s that the old image dies on and that the new director surfaces as a 500. + +## Roll inputs (verified by the read-only `verify` run) + +**Image census from instance templates, 2026-09-04 21:45Z (authoritative, read from `gcloud compute +instance-templates`):** 20 serving cells on `5aedbca5` (c8, c9, c10, c13–c16, c19–c29) — the image that exits +the process on a Postgres connect timeout (Finding 6); c7 on `85bf6799`; c4, c5, c17, c18 (draining / +migration-only) on `0e83408b` / `36a56b10`; c1, c2, c3, c6, c11, c12 (existing-only) on Jul/Aug images. Target +for Roll 1 is `519f4914` (director already on it). Monitor dry-run dispatched 21:45Z as the roll gate; waves +require owner go. + + +- target-image-digest `sha256:519f4914217f08cabcdcd34825965db8473ec37c6591553a3af0d65dcdeeb183` (lock fix; supersedes 85bf6799 as target) +- previous target `sha256:85bf67993869a769642995d0863f4c2b6b569c3850c2d8390ec2ca5f2b179e28` (c7 is on this; use as c7's rollback) +- rollback-image-digest `sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563` +- target/rollback rehome protocol 1 / 1; expected-rehome-generation 12; selector generation **112** (110 before the c7 canary) +- existing-only c1,c11,c12,c2,c3,c4,c5,c6; migration-only c17,c18; general c10,c13–c16,c19–c29,c7,c8,c9 +- confirmation for canary: `ROLL_RELAY_SAME_CAP production-gce-c7` +- monitor evidence is single-use and must be < 5 min old at dispatch (plus 75 min per predecessor wave) +- monitor dry-run dispatch (read-only, runs at `main` head so a merged bar change applies immediately): + `gh workflow run cloud-monitor-relay-production.yml --ref main -f mode=dry-run -f expected-selector-generation=110 + -f expected-existing-only-cells= -f expected-migration-only-cells=production-gce-c17,production-gce-c18 + -f expected-general-cells= -f migration-policy=strict -f recovery-source-cell-id=none -f capacity-cell-id=none` + +## Queries that worked (copy-paste) + +- Cell metrics: `resource.type="gce_instance" AND jsonPayload.event="orca_relay_runtime_metrics"` +- Container crashes: `resource.type="gce_instance" AND jsonPayload.MESSAGE:"container die" AND jsonPayload.MESSAGE:"relay@sha256"` +- Crash banner: `resource.type="gce_instance" AND jsonPayload.message:"Node.js v24"` +- Retries: `jsonPayload.event="orca_relay_postgres_transaction_retry"` (no resource filter to get both) +- Director lines are `textPayload`; cell lines are `jsonPayload.message` +- Cloud Run concurrency: Monitoring API `run.googleapis.com/container/max_request_concurrencies` +- Dry-run final state: download artifact `relay-monitor-dry-run--`, read `*.state.json` (the log's `schemaVersion` lines are only checkpoints, not the final verdict) + +## 2026-09-04 22:50Z onward: owner go received; driving the gates + +Owner: "sure, feel free to drive these." Sequence chosen: Roll 1 first (highest uplift), auth deploy with +#478 second, pruner enable third, label drift resolved by matching Terraform to live state, #477 still held. + +| Step | Result | +| --- | --- | +| Monitor dry-run #19 (gen 112, strict) | **Passed** 23:07:53Z, run 33927238469 attempt 1. First green since the probe fix (#18723). 16 samples, no freeze. Dispatched 22:51:33Z after confirming: 0 `container die` in 3 h, director 5xx in the last 4 h were all 503s (excluded by the `director.errors` filter). | +| c8 `canary-apply` onto 519f4914 (rollback 5aedbca5) | **Failed at 23:09:07Z before any mutation**: `relay monitor evidence provenance does not match` in `verify-authority`. Run 33928330631. Gate job passed, `cell_1 / rollout` failed on the manifest check, `seal_canary` skipped, lease released. Cause: the manifest binds `commitSha`; the dry-run ran at main `264c9ed8d2`, the canary dispatched at `--ref main` resolved to `4fab8e2f15` because unrelated PRs merged to main during the 15-minute gate. Verified no side effects: c8 MIG still on template `…c8-20260827…` (5aedbca5), stable, 25 controls; no `/v1/admin/drain` or isolate calls in the director log. | +| Constraint learned | Both workflows must run at the **same main commit**. The production environment's deployment branch policy allows only `main`, and the job gates on `github.ref == 'refs/heads/main'`, so a pinned tag/branch is not an option. Any merge to stablyai/orca main during the 15-minute dry-run invalidates the evidence. Mitigation for the retry: dispatch the canary within seconds of the green, and do not merge anything to stablyai/orca main myself during the window. A durable fix (accept evidence whose commit is an ancestor with identical workflow/script content) is a follow-up, not a same-day change to a safety check. | +| Label drift (5.x) | Resolved by dropping the `region` label from Terraform to match the 21 live metrics (stablyai/orca #18734, merged). Targeted plan asserted `27 no-op, 9 create, 0 destroy`; applied 23:11Z: 8 `orca_relay_control_*` renewal metrics that had never been applied, plus `google_monitoring_dashboard.relay_incident`. `orca_relay_controls` createTime unchanged (2026-07-13), label extractors unchanged. | +| Pruner enable (1.2) | orca-cloud #479 merged: `auth_token_pruner_enabled = true`, image digest of `00031-tox`, `max_rows_per_run = 20000`. Targeted plan asserted 9 create / 0 change / 0 destroy (job, scheduler at `41 * * * *` UTC, two service accounts, five IAM grants). **Not yet applied**: waiting until the roll canary has landed so the first hourly run does not overlap a drain. | +| Auth deploy with #478 (3.1) | Dispatched 23:13Z from orca-cloud main `f0fa4b5` (run 33928663526). Candidate startup adds nullable `successor_material` under a brief ACCESS EXCLUSIVE lock. | +| Auth deploy result | **Succeeded** 23:15:37Z: `orca-cloud-auth-00035-gos` serving 100 %, cap 20 preserved, 0 5xx. `refresh_tokens.successor_material` present (nullable text); 298 sealed successors written in the first 15 min against 924 rotations; `session-refresh-reuse-detected` at baseline (5 / 15 min). Grace window is live. | +| Monitor dry-run #20 | Froze 23:35:38Z on `runtime_power_unknown cell.production-gce-c11.powered`. Two window restarts earlier (23:24, 23:25) on `signal_stale auth.errors` (Cloud Monitoring publish lag 181–255 s vs 180 s bar). Cause: one transient rejection of the per-cell MIG GET in `readResourceInventory` yields `targetSize: null` → `runtimeKnown=false` → hard freeze. c11 is a parked existing-only cell (MIG size 0, stable) and was fine. Not fleet health. Fix delegated: stablyai/orca #18740 (retry the MIG read once, mirroring #18723). Run 33928912676. | +| Monitor dry-run #21 | **Green** 23:54Z at main `8064d1f991`, but main had moved to `0a821e5bc8` during the window; the chain re-gated instead of dispatching (the canary would have failed provenance again). Run 33930229711. | +| Monitor dry-run #22 | **Green** 00:10Z at `0a821e5bc8`; main moved to `2e80972450`. Re-gated. Run 33931177390. | +| Monitor dry-run #23 | Froze 00:18:31Z on `cell.production-gce-c29.latency_ms` 2635 > 2000, the probe's own round-trip from a US runner to asia-east2; c29 controls 17→19 and `sqlLatencyMsMax` flat ~1050 through the minute, no crash, no checkpoint stall. c29 probe max was 0 in the three previous gates, so a one-off. Run 33932092775. | +| Blocking constraint | Main receives unrelated merges every 5–10 min (23:08, 23:15, 23:17, 23:40, 23:42, …). A 15-min gate bound to an exact commit cannot be consumed under that traffic. Delegated a durable fix: `verify-authority` accepts evidence whose commit is an ancestor of the canary commit **and** has no diff on the monitor/deployer trusted paths; fails closed on shallow clones or unknown commits. Chain re-armed on dry-run #24 (run 33932679796) meanwhile. | +| Monitor dry-run #24 | Froze 00:28:00Z on `director.instances` 4 < 5. Cloud Run active-instance count read 4 for exactly one minute (00:27), 5 in every other minute for 3 h; min/max scale is pinned at 5; no new revision. A routine single-instance recycle. Not fleet health. Bar `directorInstancesMin: 5` with `latest-sum` cannot tolerate that; recalibrate to 4 or use a 3-min window minimum (follow-up, not same-day). Run 33932679796. Chain dispatched #25 (run 33933193511) at `86cd327749`. | +| Monitor dry-run #25 | **Green** 00:46Z at `86cd327749`; main moved to `8096cb2803`. Fourth green gate lost to unrelated main traffic (#19, #21, #22, #25). Run 33933193511. Chain's re-gate #26 (run 33934079533) cancelled by me. | +| Fixes merged 00:55Z | stablyai/orca #18740 (MIG inventory read retried once before `runtime_power_unknown`; 2 tests) and #18754 (`verify-authority` and the batch canary authority accept evidence sealed at an **ancestor** commit when every trusted monitor/deployer path is byte-identical; fails closed on shallow clones and unknown commits; deploy/rehome jobs now check out with `fetch-depth: 0`; 5 new tests, 18/18 pass). Reviewed both diffs; trusted-path set verified to exist on main. | +| Monitor dry-run #27 | Dispatched 00:56Z at `74ad08ec66` (first gate whose evidence the new rule can consume). Run 33934541092. Chain re-armed with the same ancestor + identical-trusted-code rule so an unrelated merge no longer forces a re-gate. | +| Monitor dry-run #27 | **Green** 01:11:35Z at `74ad08ec66`; main had moved to `38bde20121` with identical trusted code, so the new rule (#18754) let the chain dispatch. Run 33934541092. | +| c8 `canary-apply` #2 (run 33935407461) | Provenance check **passed** (first consumption of ancestor evidence). Isolate → gen 113, drain, template+MIG applied 01:14–01:22, new c8 came up on `519f4914` and `relay_capacity_transition_verified` (migration-only, image exact, heartbeat fresh) at 01:23:50. Then the step's next call, `curl --fail-with-body` to c8 `/v1/admin/runtime-status`, got a **503 with a 27-byte body** at 01:23:51 and the step exited 22. Director `cell-status` at 01:23:50.8 returned 200; c8's own logs show nothing at that second; c8 health/ready both 200 seconds later; backend HEALTHY (the health check had just flipped TIMEOUT→HEALTHY at 01:22:16 and UNKNOWN→HEALTHY at 01:23:47 as the new instance warmed). Read: a single 503 at the load-balancer/warm-up edge on a curl with no retry, on a cell that was already verified healthy one line earlier. Failsafe ran: c8 kept **migration-only**, rehome control disabled, selector gen 113. c8 is serving (40 controls at 01:39, sqlLatencyMsMax ~30 ms) on the target image, just not admitted for general traffic. Nothing to roll back. | +| Recovery | The job has an explicit resume path: `mode=rollback` with `rollback-image-digest` = the image the cell already runs skips isolate/apply, verifies, and restores general admission (`ROLLBACK_RESUME=true`). Dispatched gate #28 (run 33936966508) at gen 113 with c8 in migration-only; on green the chain dispatches that resume for c8 with rollback digest `519f4914` and target `5aedbca5` (the validator only requires them to differ). | +| Follow-up | The verify step's bare `curl --fail-with-body` needs the same "no reading is not a verdict" retry the monitor got (#18723/#18740); a 503 immediately after `verify-relay-capacity-transition` passed is not evidence of a bad cell. | +| Monitor dry-run #28 | **Green** 01:58:59Z at gen 113 with c8 in migration-only. Run 33936966508. | +| c8 recovery (run 33937756402, `mode=rollback`, rollback digest = 519f4914) | **Succeeded** 02:02Z. `ROLLBACK_RESUME=true` path: isolate/apply skipped, converged-Terraform check passed, verify passed (`relay_capacity_transition_verified` general, image `519f4914`, heartbeat fresh), activate → **gen 114**, c8 general. No restart, no drain. c8 at 43 controls, sqlLatencyMsMax 36 ms. **c8 is the second cell on 519f4914** (with c7 on 85bf6799). Because the recovery ran as `rollback`, `seal_canary` was skipped, so no canary authority exists for a `batch-apply`; the next cell runs as another `canary-apply`. | +| Merged 02:05Z | stablyai/orca #18769: bounded retries on every admin-endpoint curl/fetch in the same-cap job and the rehome/canary/verify scripts (`--retry 3 --retry-delay 2 --retry-connrefused`, per-attempt bodies to a file; script helper 2 attempts on network error or 500/502/503/504 only; 4xx never retried; 650/650 tests). Trusted-path change, so the next gate runs at a commit containing it. | +| Pruner enabled (1.2) | Terraform applied 02:06Z (8 creates, then the deploy-identity job IAM grant after a propagation 404, 9/9). Job `orca-cloud-auth-token-pruner`, image `343a0915…`, scheduler `41 * * * *` UTC, budget 20 000 rows/run. First run by hand (exec `sf5ct`): cold start 3m20s, then `stopReason: time-budget` at 480 s: 73 batches, 365 000 scanned, **1 040 deleted** (1 021 revoked, 19 expired, 0 rotated), ~6.4 s/batch of 5 000, `completedFullPass: false`. No errors, no lock-wait or checkpoint alert. Scan-bound, not budget-bound: at this pace a full pass over the table takes many hourly runs, and the row budget is never the limiter. Leave the budget alone; watch hourly runs for `stopReason` and a rising `deletedRows` as the cursor reaches the rotated backlog. | +| Monitor dry-run #29 | **Green** 02:20:58Z at gen 114, main `e2b70a5eba` (contains #18740, #18754, #18769). Run 33938052374. | +| c9 `canary-apply` (run 33938818286) | **Succeeded end to end** 02:21–02:34Z: isolate → gen 115, drain, template+MIG to `519f4914`, verify passed on the first try (retry-hardened step), trust proof, activate → **gen 116**, general. `seal_canary` **succeeded**: batch authority now exists. c9 at 38 controls, sqlLatencyMsMax 33 ms. No `container die` in 30 min. Three cells on new images (c7 `85bf6799`, c8 and c9 `519f4914`); 17 serving cells still on `5aedbca5`. | +| Monitor dry-run #30 | Dispatched 02:36Z at gen 116 (run 33939533990). On green the chain dispatches **batch 1**: `batch-apply` c10,c13,c14,c15 bound to canary run 33938818286 (sealed at gen 116, same commit `e2b70a5eba`). Preflight: all four on `5aedbca5`, MIGs stable, no crash in 20 min. Sequential cells inside the job (wave-index 0..3), each with its own isolate/drain/apply/verify/restore, so ~12 min per cell, ~50 min total. | +| Monitor dry-run #30 verdict | **Green** 02:52:15Z at gen 116, `e2b70a5eba`. | +| Batch 1 (run 33940290163) | Dispatched 02:52:27Z: `batch-apply` c10,c13,c14,c15, canary authority run 33938818286, same commit. | +| Batch 1 attempt 1 (run 33940290163) | **Failed at 02:54:39Z in the live preflight, before any mutation**: `relay live preflight failed: cloud-monitoring/signal_stale`. The step's `--retry-freshness` (5 attempts, 15 s apart, freshness-only codes) is passed only for `WAVE_INDEX != 0`; the first cell takes a single sample, so one Cloud Monitoring publish lag > 180 s at that instant fails the batch. Every candidate series was current again by the time I checked. c10 untouched (template `…c10-20260827…`, 47 controls), no selector write, gen still 116, failsafe no-op. Gate #31 dispatched 02:57Z (run 33940508865); chain re-dispatches the same batch (canary authority 33938818286 still valid: same gen 116, same commit). Fix delegated: wave 0 gets the same freshness retry. | +| Monitor dry-run #31 | **Green** 03:13:26Z at gen 116; main at `cb7f7dd11a` with identical trusted code. Run 33940508865. | +| Batch 1 attempt 2 (run 33941253533) | Dispatched 03:13:38Z: c10,c13,c14,c15, canary authority 33938818286. Runs at `cb7f7dd11a` (batch authority is accepted across the ancestor since trusted paths are unchanged). | +| Merged 03:14Z | stablyai/orca #18778: `--retry-freshness` on every same-cap wave including the first, and the retry loop now stops before the next wait would push evidence past the wave's age bound (it was checked only at entry before). Twin carve-out in the capacity job filed as a follow-up. | +| Batch 1 cell 1 (c10) | **Succeeded** 03:14–03:27Z (preflight, drain, apply, verify, restore). c13 started 03:27Z. | +| Batch 1 cell 2 (c13) | **Succeeded** 03:27–03:38Z. c14 started 03:38Z. | +| Batch 1 cell 3 (c14) | **Succeeded** 03:38–03:50Z. c15 started 03:50Z. | +| Batch 1 complete (run 33941253533) | **All four succeeded** 03:13–04:00Z: c10, c13, c14, c15 on `519f4914`, selector **gen 124**. Fleet at 936 controls, 23 cells. Two `container die` at 03:35:41/44 were **c13's new container** exiting during boot (`applyPostgresSchema` → `Connection terminated due to connection timeout`, exit 1, 2 s runtime each) because the `cloud-sql-proxy` sidecar had not finished starting; the third start at 03:35:45 succeeded and c13 has been serving since (57 controls). A boot-order race in the container spec, not a serving-cell crash. Follow-up: schema pool should wait for the proxy socket, or the container should depend on the proxy's readiness. **8 cells on new images** (c7 85bf6799; c8, c9, c10, c13, c14, c15 519f4914), 12 on `5aedbca5`: c16, c19–c26 (US), c27–c29 (Asia). | +| Monitor dry-run #32 | **Green** 04:19:50Z at gen 124, main `436ef827dd` (contains #18778). Run 33943539025. | +| c16 `canary-apply` (run 33944255902) | Dispatched 04:20:02Z. On success it seals the authority for batch 2 (c19,c20,c21,c22). | +| c16 canary (run 33944255902) | **Succeeded** 04:20–04:32Z, activate → gen 126, batch authority sealed. 9 cells on new images. | +| Monitor dry-run #33 | Failed 04:58:56Z on `continuity_deadline_exceeded` (1 500 004 ms > 1 500 000 ms). One `signal_stale cloud_sql.lock_waits` at 04:46 (189 s vs 180 s bar, Cloud Monitoring publish lag) restarted the 15-min window at sample 12; the restart could not complete inside the 25-min continuity cap. No health failure at any sample; no `container die` since c16's own boot race at 04:30. Run 33944873727. Chain re-gates. Note for recalibration: `cloudDataMaxAgeMs: 180000` vs observed Cloud Monitoring publish lag of 181–255 s has now cost three gates (#20 twice, #33). | +| Freshness recalibration | stablyai/orca #18798 (open, merge after batch 2 dispatch): `cloudDataMaxAgeMs` 180 s → 330 s, derived from Google's documented visibility delays (Cloud Run 60+120 s, Cloud SQL 60+165 s) and the 5-min window-sum query (a label series that stops emitting reads as up to 300 s old while its sum is complete, which is the 255 s `auth.errors` case) plus ~30 s collect latency. Director-admin and the lock-wait carry keep their own 180 s pins. A freshness-only failure may miss 2 consecutive samples without restarting the window; the sample still counts and is still threshold-checked; a 3rd miss, collector failure, runner gap, or any breach restarts/freezes as before. 92/92 tests. | +| Monitor dry-run #34 | **Green** 05:17:31Z at gen 126, `436ef827dd`. Run 33946093029. | +| Batch 2 (run 33946819345) | Dispatched 05:17:43Z: c19,c20,c21,c22, canary authority 33944255902 (c16). | +| Merged 05:19Z | stablyai/orca #18798 (freshness bar 330 s + two-sample tolerance). Next gate runs at a commit containing it. | +| Batch 2 cell 1 (c19) | **Succeeded** 05:19–05:32Z. c20 started. | +| Batch 2 cell 2 (c20) | **Succeeded** 05:32–05:43Z. c21 started. | +| Batch 2 cell 3 (c21) | **Succeeded** 05:43–05:59Z. c22 started. | +| Batch 2 complete (run 33946819345) | **All four succeeded** 05:17–06:12Z: c19, c20, c21, c22 on `519f4914`, selector **gen 134**. Fleet at 1 090 controls, 23 cells, refresh 401s at baseline (1–4 per 3 min). One `container die` at 06:08:45 was **c22's new container** exiting during boot (exit 1, 2 s runtime; started 06:08:43, restarted 06:08:46 and serving since), the same proxy-sidecar boot race seen on c13 and c16. No serving-cell crash. **Census: 15 of 23 serving cells on new images** (c7 `85bf6799`; c8–c10, c13–c16, c19–c22 `519f4914`), 7 on `5aedbca5`: c23–c26 (US), c27–c29 (Asia). Next: gate at gen 134 → canary c23 → batch c24,c25,c26; then canary c27 → batch c28,c29. | +| Monitor dry-run #35 | **Green** 06:32:01Z at gen 134, `b33d1972bc` (contains #18798, first gate at the 330 s freshness bar). Run 33949334606. | +| c23 `canary-apply` (run 33950075843) | Dispatched 06:32:13Z at main `b0c67eaf88` (ancestor gate SHA, identical trusted code). On success it seals the authority for batch 3 (c24,c25,c26). | +| c23 canary (run 33950075843) | **Succeeded** 06:32–06:46Z, activate → gen 136, batch authority sealed. No `container die` during boot. 16 of 23 serving cells on new images; 6 on `5aedbca5` (c24–c26 US, c27–c29 Asia). | +| Monitor dry-run #36 | Dispatched 06:46Z at gen 136, run 33950746574 (`58553bfe1c`). On green the chain dispatches batch 3 (c24,c25,c26) under canary authority 33950075843. | +| Monitor dry-run #36 result | **Green** 07:02:49Z at gen 136, `58553bfe1c`. | +| Batch 3 (run 33951468008) | Dispatched 07:03Z: c24,c25,c26, canary authority 33950075843 (c23). | +| Batch 3 cell 1 (c24) | **Succeeded** 07:04–07:18Z. c25 started. | +| Batch 3 cell 2 (c25) | **Succeeded** 07:18–07:31Z. c26 started. | +| Batch 3 complete (run 33951468008) | **All three succeeded** 07:03–07:44Z: c24, c25, c26 on `519f4914`, selector **gen 142**. Fleet at ~1 230 controls, 23 cells, refresh 401s at baseline. **Zero `container die`** during the batch (no boot race on c24–c26). **All 20 US serving cells now on new images** (c7 `85bf6799`; c8–c10, c13–c16, c19–c26 `519f4914`). Remaining on `5aedbca5`: c27, c28, c29 (asia-east2, probe hard cap 3000 ms). | +| Monitor dry-run #37 | Dispatched 07:48Z at gen 142, run 33953555224 (`4c5077d57a`). On green the chain dispatches the c27 canary (first Asia cell). | +| Monitor dry-run #37 result | **Green** 08:04:24Z at gen 142, `4c5077d57a`. | +| c27 `canary-apply` (run 33954264945) | Dispatched 08:04Z, first Asia cell (asia-east2-a). On success it seals the authority for batch 4 (c28,c29). | +| c27 canary (run 33954264945) | **Failed closed before any mutation** 08:07:25Z at "Verify exact current generation, digest, cap, and rollback point": `runtime predecessor mismatch fields=regionalRehomeProtocol`. **Operator input error, not a cell fault**: the chain script hardcoded `target-rehome-protocol=1 / rollback-rehome-protocol=1` for every cell, but `relay_region_rehome_source_cell_ids` lists only the 16 US cells (c7–c10, c13–c16, c19–c26), so the Asia startup template omits `ORCA_RELAY_REHOME_*` and c27–c29 report protocol 0 by design. `MUTATION_STARTED` never set, failsafe no-op, selector stays gen 142, c27 still serving on `5aedbca5`, no `container die`. Gate #37 evidence consumed. Fix: chain script now takes `PROTO`; Asia round dispatches with protocol 0 (the per-host trust proof step is protocol-gated and skips, as designed for non-source cells). Follow-up: the job already reads `relay_region_rehome_source_cell_ids`; it could derive the expected protocol from membership instead of trusting the operator input. | +| Monitor dry-run #38 | Dispatched 08:12Z at gen 142, run 33954621425 (`e95d247be1`). On green the chain dispatches the c27 canary with protocol 0. | +| Monitor dry-run #38 result | **Green** 08:28:36Z at gen 142, `e95d247be1`. | +| c27 `canary-apply` #2 (run 33955359385) | Dispatched 08:28Z with `target/rollback-rehome-protocol=0`. | +| c27 canary #2 (run 33955359385) | **Failed closed, no mutation** 08:31:19Z. Predecessor check passed with protocol 0; the isolate step then died at argument parsing: `production capacity target is not approved`. The same-cap job shells out to `prepare-relay-production-capacity-canary.mjs` for isolate/drain/activate, whose `PRODUCTION_CAPACITY_CELL_IDS` allowlist is the 16 US capacity cells (c7–c26), while the same-cap wave validator (`SAME_CAP_CELLS`) approves all 19 serving cells including c27–c29. The Asia cells have never been through this job (their Aug 14 rollout used the asia-topology workflow). Both the isolate step and the failsafe threw before any HTTP call, so `MUTATION_STARTED=true` was written but nothing was isolated: selector stays gen 142, c27 general and serving on `5aedbca5`, no `container die`. Gate #38 evidence consumed. Fix: stablyai/orca #18811 (`--approved-cells same-cap` on all four invocations, default unchanged for the US capacity job, census test over every `SAME_CAP_CELLS` member × isolate/drain/activate + the job's cell-shape bash block; 525/525 script tests). Sweep of the other job scripts found no further Asia blocker; gate #39 (run 33955668701) dispatched at gen 142 to prove the selector is unchanged before the next attempt. | +| Monitor dry-run #39 | **Green** 08:51:28Z at gen 142: independent proof the selector was untouched by both failed c27 attempts. Not used for dispatch (its commit predates #18811). | +| Merged 08:51Z | stablyai/orca #18811 → main `12e05203a4`. | +| Monitor dry-run #40 | Dispatched 08:51Z at gen 142 on main `12e05203a4` (contains #18811), run 33956408337. On green the chain dispatches the c27 canary, protocol 0, third attempt. | +| Monitor dry-run #40 result | **Green** 09:08:03Z at gen 142, `12e05203a4`. | +| c27 `canary-apply` #3 (run 33957151726) | Dispatched 09:08Z, protocol 0, on main containing #18811. | +| c27 canary #3 (run 33957151726) | **Failed after isolate; failsafe held** 09:17:21Z. Live check 09:26Z: c27 at 0 controls (drained), template still `…20260814235757`, c28/c29 absorbed the hosts (37 each), fleet 1 404 controls / 23 cells, refresh 401s baseline, no `container die` in 60 m. Predecessor check and allowlist passed; isolate → **gen 143** (c27 migration-only), drain sent (graceMs 0, hosts reconnected via director to c28/c29/US). Terraform plan built correctly (template replace + MIG update to `519f4914`), then `validate-relay-capacity-plan.mjs --mode same-cap-cell` rejected it: `cell plan does not contain the reviewed image and capacity`. Its same-cap rule demands exactly one `ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT` and one `ORCA_RELAY_REHOME_AUDIENCE` printf in the startup script; Asia templates omit both because c27–c29 are not rehome sources (same root as attempt 1, third US-only assumption in the job). **No apply ran**: c27 template unchanged, still `5aedbca5`, isolated and draining (drain is one-way in-process; only a restart clears it). Failsafe re-asserted migration-only at gen 143 and rehome disabled. Recovery plan: fix validator (protocol-0 path: require the rehome lines *absent*), merge, gate at gen 143, then `mode=rollback` with rollback-image=`519f4914` (the failed-canary re-entry path; accepts draining + migration-only) to restart c27 onto the target image and restore it; then single-cell canaries for c28 and c29 (batch needs ≥2 cells). | +| Plan-validator fix | stablyai/orca #18818 (merged 09:41Z → main `9f2a9a248e`): `validate-relay-capacity-plan.mjs --regional-rehome-protocol 0|1` in same-cap-cell mode; protocol 0 requires the rehome lines *absent*, protocol 1 unchanged; both plan-validation calls in the job pass `DESIRED_REHOME_PROTOCOL`; census test now validates a correct plan for every `SAME_CAP_CELLS` member at its tfvars-derived protocol. 529/529. Residual: the operator-supplied protocol is still unbound for Asia cells (no `SOURCE_CELLS` cross-check outside us-central1), so a wrong value fails late at plan validation rather than early; deriving it from membership is the checklist follow-up. | +| Monitor dry-run #41 | Dispatched 09:42Z at gen 143 (c27 expected migration-only) on main `9f2a9a248e` (contains #18811 + #18818), run 33958728141. On green: c27 recovery via `mode=rollback`, rollback-image `519f4914`, protocol 0, confirmation `ROLL_BACK_RELAY_SAME_CAP`. | +| Monitor dry-run #41 result | **Green** 09:58:51Z at gen 143, `9f2a9a248e`. | +| c27 recovery #1 (run 33959789773, `mode=rollback`) | **Failed closed, no mutation** 10:09:21Z at `Verify monitor evidence provenance`: `relay monitor dry-run authority is incomplete or stale`. The dry-run authority is valid for 5 min after `completedAt` at wave 0 (`EVIDENCE_MAX_AGE_MS`); the gate completed 09:58:51Z but the operator poller (20 s `gh run view` loop) only observed completion at 10:07:09Z during a local network outage, so the dispatch landed at 10:07:11Z, 8 m 20 s after completion. Failed before the rollout lease, isolate, or any Terraform step; c27 unchanged (migration-only, drained, `5aedbca5`, gen 143). Every prior canary dispatched ≤15 s after gate green, so this is a dispatch-latency miss, not a job defect; the freshness bound behaved as designed. | +| Monitor dry-run #42 | Dispatched 18:39Z at gen 143 on main `af82126058` (trusted paths byte-identical to `9f2a9a248e`), run 33984753269. Recovery script re-armed behind it (same `mode=rollback` onto `519f4914`, protocol 0). | +| Monitor dry-run #42 result | **Green** 18:55:47Z at gen 143, `af82126058`. | +| c27 recovery #2 (run 33985902062, `mode=rollback`) | **Failed closed, no mutation** 19:05:02Z, same `authority is incomplete or stale`. Dispatch landed 19:02:39Z, 6 m 52 s after the gate completed. Root cause of both misses is the operator laptop sleeping during the 15 min gate wait (`pmset -g log`: asleep 18:52:28Z → 19:02:17Z; the morning miss coincided with a sleep/dark-wake cycle too), so the 20 s poller never ran inside the 5 min window. Not a job or evidence defect: the freshness bound did its job. Operator fix: poller now runs under `caffeinate -i`. | +| Monitor dry-run #43 | Dispatched 19:06Z at gen 143 on main `af82126058`, run 33986121849. Recovery armed behind it under `caffeinate`. | +| Monitor dry-run #43 result | **Green** 19:22:53Z at gen 143, `af82126058`. | +| c27 recovery #3 (run 33986948522, `mode=rollback`) | **Failed closed, no mutation** 19:25:48Z. Dispatched 13 s after gate green (authority accepted this time), then the live preflight recheck failed: `relay live preflight failed: active-probe/threshold_max`. That is the 2 000 ms `endpointLatencyMs` bar on one endpoint's slowest /health or /ready round trip from the runner (8 s fetch timeout, one retry). The error names no endpoint and the job log prints none; gate #43 had zero failures across 16 samples, so this was a transient probe slow-down in the ~3 min between gate and preflight. Live probe 19:32Z from the operator: director and auth ~130–190 ms, US cells ≤540 ms, Asia cells 690–1 315 ms (c28/c29 /health ~1.3 s, the closest to the bar; c27 ~0.9 s). Existing-only cells c1–c3, c6, c11, c12 return 503 on both paths as expected (unpowered). Failed before the rollout lease, isolate, or any Terraform step; c27 unchanged. Follow-up (checklist): preflight should print the failing signal and observed value. | +| Monitor dry-run #44 | Dispatched 19:33Z at gen 143 on main `062db77118`, run 33987646501. Recovery re-armed behind it. | +| Monitor dry-run #44 result | **Frozen red** 19:50:01Z after 13 samples: `active-probe/threshold_max cell.production-gce-c27.latency_ms observed=2568 threshold=2000`. No other failure, no continuity event, no `container die` fleet-wide in 60 m. `/health` is a static JSON reply (`app.ts`), so the slow round trip was `/ready` (the probe reports the max of the two) or the path to the cell. Cloud SQL logs for 19:49:38Z–19:51:58Z show six `could not obtain lock on row in relation "relay_cells"` errors and a time-triggered checkpoint completing at 19:50:36Z (write phase 270 s, the spread target, not a stall). c27 is drained with 0 controls, so its `/ready` dependency check was the only thing it was doing. Recovery script stopped as designed (no auto re-gate). Operator probe 19:53Z: c27 and c28 both bimodal, ~0.27 s or ~0.89 s per `/health` from the US, identical shape, nothing c27-specific. Attributing the one 2.6 s sample to the same shared-DB contention that produced the lock errors is the best available reading; the retry at gate #45 tests whether it recurs. | +| Monitor dry-run #45 | Dispatched 19:53Z at gen 143 on main `062db77118`, run 33988383401. Recovery re-armed behind it. | +| Monitor dry-run #45 result | **Frozen red** 19:54:47Z after 3 samples, same signal: `cell.production-gce-c27.latency_ms observed=2668 threshold=2000`. Two gates in a row now attribute a >2 s round trip to c27 while every other cell passes. | +| c27 `/ready` tail analysis | `/health` is static; `/ready` (`relay-readiness.ts`) fetches the auth JWKS (2 s timeout) then runs `SELECT 1`, cached 10 s. Operator probes 19:57Z–20:00Z, 15 each from the US: c27 and c28 have the **same** tail (0.27 s / 0.88 s modes, then 1.3 s, then 2.17–2.27 s at the top); US cells c8/c20 sit at 0.08–0.18 s. Auth JWKS latency over the last hour: 400 requests, max 20 ms, none over 1 s. So the tail is cell→Cloud SQL (US) round trips plus the runner→Asia hop, not auth and not c27-specific; c27 is drained (0 controls) so nothing local competes. Cloud SQL `could not obtain lock on row in relation "relay_cells"` runs at 17–78 per 10 min all day (NOWAIT inventory locks, expected under placement bursts) with no spike in the failing minutes. The bar (`endpointLatencyMs` 2 000 ms, one shot per minute, max of two paths) leaves Asia cells ~10% of samples from tripping; the gate got unlucky twice on c27 and lucky on c28/c29. Not a health finding. | +| Monitor dry-run #46 | Dispatched 20:01Z at gen 143 on main `062db77118`, run 33988810139. Recovery re-armed behind it. If this also freezes on an Asia probe, the next move is a per-region latency bar (or p50 over the window) in `incident-monitor.ts`, reviewed and merged before further Asia gates rather than retrying blindly. | +| Monitor dry-run #46 result | **Frozen red** 20:07:44Z after 7 samples, third time on `cell.production-gce-c27.latency_ms` (observed 2 685). Operator 40-sample `/ready` probe per Asia cell at 20:10Z: c27 p50 0.88 s / p90 2.15 s / max 2.26 s / 6 over 2 s; c28 p50 0.88 / p90 1.25 / max 2.25 / 1 over; c29 p50 0.88 / p90 0.89 / max 1.27 / 0 over. All 200. `/ready` (`relay-readiness.ts`) fetches the auth JWKS in us-central1 then `SELECT 1` on Cloud SQL in us-central1, so an Asia cell's readiness is two trans-Pacific hops plus the runner→Asia hop; the fleet-wide 2 000 ms bar was calibrated on US cells (0.08–0.5 s). c27 being drained and idle has no local load, so this is path latency, not health. **Stopped retrying gates.** Fix in flight: per-region `cell..latency_ms` bar (us-central1 stays 2 000, asia-east2 4 000; hard faults still caught by the health/ready equal-1 checks and the 8 s probe timeout) plus attributable preflight failure messages, via review + CI before the next Asia gate. | +| Merged 20:33Z | stablyai/orca #18877 → main `a3c1d32995`: per-region `cellEndpointLatencyMs` (us-central1 2 000, asia-east2 4 000; director/auth rules and the `endpointLatencyMs` key unchanged), region carried from tfvars onto every cell expectation, preflight failures now print `source/code signal observed= threshold=`. relay-ops 95/95, cloud suite 633 + 529 + 148 green. | +| Monitor dry-run #47 | Dispatched 20:34Z at gen 143 on main `a3c1d32995` (first gate with the per-region bar), run 33989896150. Recovery re-armed behind it. | +| Monitor dry-run #47 result | **Green** 20:38:09Z at gen 143 on `a3c1d32995`: first gate under the per-region bar, 16/16 samples, no Asia latency failure. | +| c27 recovery #4 (run 33990715317, `mode=rollback`) | **Success** 20:51Z. Dispatched 13 s after gate green. Isolate re-asserted migration-only at gen 143 (already isolated, no change), Terraform applied the same-cap template `…20260905204141` and the MIG replaced the instance, new incarnation on `519f4914`, protocol 0, transition verifier passed at migration-only (1 180 assignments carried, hard cap 3 000, heartbeat fresh), then activate → **gen 144**, c27 general, verifier passed again. No `container die` fleet-wide 19:55Z–20:52Z. c27 now runs the target image; c28/c29 remain on `5aedbca5` (template `…20260814235757`). | +| Monitor dry-run #48 | Dispatched 20:53Z at gen 144 (c27 back in general, MIG = c17,c18) on main `61ebffa86e` (trusted paths identical to `a3c1d32995`), run 33991385880. On green the chain dispatches the c28 `canary-apply`, protocol 0. | +| Monitor dry-run #48 result | **Green** 21:08Z at gen 144, 16/16 samples, no Asia latency failure. Main had moved to `5cec2c2dfc`; the chain verified the trusted paths were identical to the gate commit and dispatched 12 s after green. | +| c28 canary (run 33992169289, `canary-apply`) | **Success** 21:27Z. Isolate → migration-only at **gen 145**, drain already clear, verifier passed on the old image (1 220 assignments carried, hard cap 3 000, heartbeat fresh), Terraform applied same-cap template `…20260905211352`, new incarnation on `519f4914` at protocol 0, verifier passed again at migration-only, activate → **gen 146**, c28 general, verifier passed (1 219 assignments). Seal step recorded the canary. No `container die` fleet-wide 21:08Z–21:30Z. Only c29 remains on `5aedbca5`. | +| Monitor dry-run #49 | Dispatched 21:33Z at gen 146 (c28 back in general, MIG = c17,c18) on main `dce5ebd83d` (trusted paths identical to `a3c1d32995`), run 33993075948. On green the chain dispatches the c29 `canary-apply`, protocol 0, the last Roll 1 cell. | +| Monitor dry-run #49 result | **Frozen red** 21:52:24Z, `active-probe/continuity_deadline_exceeded observed=1500005 threshold=1500000`. One continuity event at 21:41:27Z, `cloud-monitoring/collector_failed` (a Cloud Monitoring read failed, not tolerated), which reset the continuous window at sample 14; the restarted window reached 10 samples before the 25-minute lineage cap (`INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS`) expired. No health failure in any of the 25 samples, no Asia latency failure, no `container die`. Monitor-side transient, not a fleet finding. The chain re-gated automatically after its 2-minute back-off. | +| Monitor dry-run #50 | Dispatched 21:54Z at gen 146 on main `51eed5a1bc`, run 33994385666. **Green** 22:10Z, 16/16 samples. Main had moved to `d7767fb196`; trusted paths identical to `a3c1d32995`. Chain dispatched the c29 `canary-apply` (run 33995164002, protocol 0) 12 s after green. | +| c29 canary (run 33995164002, `canary-apply`) | **Success** 22:27Z. Isolate → migration-only at **gen 147**, verifier passed on the old image (1 199 assignments), Terraform applied same-cap template `…20260905221622`, new incarnation on `519f4914` at protocol 0, verifier passed at migration-only, activate → **gen 148**, c29 general, verifier passed (1 199 assignments carried). No `container die` fleet-wide 22:11Z–22:30Z. | +| **Roll 1 complete** | Image census 22:30Z from MIG templates: c8–c10, c13–c16, c19–c29 on `519f4914` (18 cells); c7 on `85bf6799` (the earlier rehearsal image, carries the same fix); existing-only c1–c6, c11, c12 and migration-only c17, c18 untouched by design. No serving cell remains on `5aedbca5`. Selector gen 148, membership unchanged from the start of the roll. Zero relay container exits fleet-wide across the roll (01:14Z–22:30Z). Gates used: #19–#50; freezes were all monitor-side (provenance, freshness, flat Asia latency bar, one Cloud Monitoring collector failure), none a fleet health finding. Roll 2 (fresh image with #18722 + #18720) is the next data-plane step and waits on the owner's private-IP window decision. | diff --git a/cloud/docs/relay-roll2-plan-2026-09.md b/cloud/docs/relay-roll2-plan-2026-09.md new file mode 100644 index 00000000000..f84de39163f --- /dev/null +++ b/cloud/docs/relay-roll2-plan-2026-09.md @@ -0,0 +1,154 @@ +# Relay Roll 2 and close-out plan (2026-09-05) + +Owner-approved scope 2026-09-05: finish the relay reliability work with one more cell image roll, +deferring the Cloud SQL private-IP move (2.1, orca-cloud #477) to a separate owner decision. Roll 1 +is complete (see `relay-reconnect-2026-09-findings.md`, "Roll 1 complete"); every serving cell runs +`519f4914` except c7 on `85bf6799`. + +Estimate: about two working days of effort over one week of calendar time. The cell roll itself is +6 to 7 hours of mostly unattended wall clock, run in the US night. + +## Phase 0. Land the code (half a day, no production change) + +### 0a. Split PR #18565 + +The branch mixes three relay/mobile/desktop fixes with the operator record. Split so the record +lands regardless of how the code review goes. + +- **Docs PR** (new branch off main): `relay-reconnect-2026-09-findings.md`, + `relay-improvement-checklist-2026-09.md`, `relay-improvement-roadmap-2026-09.md`, this file. + Docs only, merge on CI green. +- **Code PR** (rebase #18565 onto main, resolve two conflicts): + - `cloud/apps/relay/src/host-session-registry.ts`: conflict with #18698 (signed-out signal). + Keep both; the accept-abandonment and lease changes are orthogonal to the signed-out path. + - `src/main/runtime/relay/relay-origin-pool.ts`: **drop this branch's version**. #18719 already + merged the desktop early-window jitter (1 to 6 min). Also drop + `relay-session-broker.test.ts` additions that only exercise the dropped change. + - Keep: relay accept abandonment (`orca_relay_client_accept_abandoned` event), relay-side lease + jitter, mobile direct-probe fail-fast, and their tests. + +### 0b. Lengthen the control lease (same code PR) + +In `cloud/apps/relay/src/host-session-registry.ts`: + +``` +CONTROL_LEASE_MS = 6 * 60 * 60 * 1000 // was 55 min +CONTROL_LEASE_JITTER_MS = 30 * 60 * 1000 // was 5 min +``` + +Why 6 h: the lease bounds how long a host stays on a cell after a missed drain and is the only +passive rebalancing; 6 h keeps both and cuts control-activation traffic on the inventory lock by +about 6x. Nothing else depends on it: the relay JWT (5 min) is refreshed by the desktop on its own +schedule and liveness is the 75 s silence watchdog. Wire-safe: the relay sends `leaseExpiresAt` in +the hello ack and old desktops schedule from that value. + +Update the comment above the constants and the three assertions in +`host-session-client-accept.test.ts` that pin the lease arithmetic. Check that nothing in +`cloud/apps/relay-ops` or the monitor thresholds assumes a 55 min rotation period (grep +`55`, `CONTROL_LEASE`, `rotation`). + +### 0c. Review and merge + +Review rounds per the standing process (Opus review, then Codex pass). Merge order: docs PR first +(no dependency), then the code PR. Record the merge SHA of the code PR; that is the Roll 2 image +source. + +## Phase 1. Build and stage the image (half a day) + +Roll 2 image = code PR merge SHA. It carries, relative to `519f4914`: + +| Change | PR | Effect | +|---|---|---| +| Per-cell inventory locks, delta counters | #18722 | Removes the global `relay_cells FOR UPDATE` behind the phone accept hang | +| Relay pool `statement_timeout` 5 s | #18722 | A relay query can no longer hang a cell | +| Accept abandonment | #18565 | Cell stops finishing accepts for phones that already closed | +| Control lease 6 h ± 30 min | #18565 | Fewer, spread-out rebinds | +| `--private-ip` proxy flag support | #18720 | Code only; flag stays unset until 2.1 | + +Steps, in order (from the findings doc's post-merge dispatch plan): + +1. `gh workflow run cloud-publish-relay-production.yml --ref main -f mode=publish`. Resolve the + digest by tag, not from the log: + `gcloud artifacts docker images describe us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay:sha- --format='value(image_summary.digest)'`. +2. Staging: `cloud-deploy-relay-staging.yml` with the new digest; paired phone plus desktop smoke + (connect, background, reconnect). Confirm `orca_relay_client_accept_abandoned` appears only when + a client closes early, and that `sqlLatencyMsMax` no longer pins at the lock timeout. +3. Director: `cloud-deploy-relay-production-director.yml -f image-digest= + -f regional-placement-mode=preserve -f prune-incompatible-revisions=false + -f expected-rehome-generation=12 -f bootstrap-runtime-identity=false + -f predecessor-image-digest=`. Blue/green; prior revision stays as rollback. + Watch director `orca_relay_postgres_transaction_retry` per minute before and after. The director + goes first so the per-cell locks are live before any cell restart burst. +4. Same-cap `verify` mode against c7 with target=, rollback=`519f4914`. Read-only. + +Go/no-go for Phase 2: director serving the new image for at least 30 min, retries per minute at or +below the pre-deploy baseline, no `container die`, no auth 5xx. + +## Phase 2. Roll the cells (one US night, mostly unattended) + +Same machinery as Roll 1: `cloud-monitor-relay-production.yml` dry-run gate, then +`cloud-deploy-relay-production-same-cap.yml`. Cells roll one at a time by design (exact selector +assertions, single Terraform state, and one cell's ~1.2k-host reconnect burst per restart). Do not +add parallelism for this roll. + +Inputs: target=, rollback=`519f4914` (c7: rollback=`85bf6799`). Selector membership is +unchanged from the end of Roll 1 (gen 148; existing-only c1–c6, c11, c12; migration-only c17, c18). + +Order: + +1. **c7 canary** (`canary-apply`, protocol 1). c7 is the rehearsal cell and the only one not on + `519f4914`. +2. **c8 canary**, then **batch c9, c10, c13, c14**. +3. **c15 canary**, then **batch c16, c19, c20, c21**. +4. **c22 canary**, then **batch c23, c24, c25, c26**. +5. **Asia c27, c28, c29** as three single canaries at protocol 0 (`PROTO=0`). Batch mode cannot + take Asia cells yet and needs at least two cells. + +Each batch needs a same-commit canary authority; each wave needs a fresh 15 min gate. Use the +chain script pattern from Roll 1 (wait gate green, check trusted-path ancestry, dispatch within 5 min, +log `CANARY `) under `caffeinate -i`. Budget: 11 to 13 min per cell plus 15 min per gate, +about 6 to 7 h total. + +Per wave checks (same as Roll 1): transition verifier passes at migration-only and again at general +with assignments carried; no `container die` fleet-wide; selector generation advances by exactly 2 +per cell. After the Asia cells: image census from MIG templates; every general cell on the new digest. + +Failure handling: a failed canary re-enters through `mode=rollback` with rollback-digest = desired +image (Roll 1 c27 pattern). A gate freeze on an Asia latency probe despite the 4 000 ms bar is a +stop-and-investigate, not a retry. Monitor-side freezes (freshness, continuity deadline) re-gate +after a 2 min back-off; the chain does this on its own. + +Record every gate and wave in the findings doc as in Roll 1. + +## Phase 3. After the roll (spread over the following week) + +- **4.4 Recalibrate the retries bar.** After one week of `orca_relay_postgres_transaction_retry` + on the new image, re-derive the `postgres_retries` monitor threshold from the new baseline + (PR against `cloud/apps/relay-ops/src/incident-monitor.ts` thresholds). About 2 h. +- **1.2 Pruner budget.** Raise `auth_token_pruner_max_rows_per_run` to the default 200k after a + clean day; watch Cloud SQL write MB/s and the checkpoint alert. Then **1.5** log metric plus + policy on `stopReason != complete`. +- **1.3 Reclaim.** Once pruner runs delete ~0 rows: `pg_repack -t refresh_tokens` off-peak (check + `pg_available_extensions` first; not `VACUUM FULL`). Confirm table, index, and `disk/utilization` + dropped. +- **Monitor residuals** already in the checklist: `probeEndpointHealth` retry decision still uses the + flat 2 000 ms bar; operator protocol unbound for Asia; `probe-relay-rehome-trust` regex. +- Update the checklist status header; tick 2.3, 4.1, 4.3 relay-side as deployed. + +## Deferred, owner decision required + +- **2.1 Private IP** (orca-cloud #477). One-way door with a Cloud SQL restart. When chosen: apply the + foundation off-peak, then a template-only change that sets the `--private-ip` proxy flag. That is + another cell roll unless bundled with a future image. +- **5.2 Paging channel** for auth alerts: needs a destination. +- **Parallel cell rolls** (2 or 3 at a time): about 1.5 days (relax exact-selector assertions to + "exact except in-flight", single coordinator Terraform apply, parallel job shape, tests). Only + worth building if more image rolls are planned after Roll 2, and only once the per-cell locks are + live so a multi-cell reconnect burst is safe. +- **2.2 Database split**: deferred to ~2026-11-01. + +## Not in this plan + +Desktop and mobile changes already merged (#18719 desktop early-window jitter and no same-token +refresh retry; #18565 mobile fail-fast once merged) ship with the next desktop and mobile releases +on their own schedules. No relay action needed. From 6a3e446c69b46ee63e13304cb7402d5c893915fa Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 17:24:12 -0700 Subject: [PATCH 11/17] test: make SSH artifact regression fixtures reliable at narrow widths (#18947) --- tests/e2e/ssh-codex-display-artifacts-repro.spec.ts | 2 +- tests/e2e/ssh-codex-repro-remote-fixtures.ts | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts b/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts index e19a090df4d..f4c02d04c94 100644 --- a/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts +++ b/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts @@ -48,7 +48,7 @@ import { resetWebglAndCaptureGraySlabAnalysis } from './terminal-webgl-reset-cap const RUN_DOCKER_SSH = process.env.ORCA_E2E_SSH_DOCKER === '1' const RUN_REAL_REMOTE_CODEX = process.env.ORCA_E2E_REAL_REMOTE_CODEX === '1' -const EXPECT_NO_ARTIFACTS = process.env.ORCA_E2E_EXPECT_NO_CODEX_ARTIFACTS === '1' +const EXPECT_NO_ARTIFACTS = process.env.ORCA_E2E_EXPECT_NO_CODEX_ARTIFACTS !== '0' const CAPTURE_WHILE_REMOTE_TUI_RUNNING = process.env.ORCA_E2E_CAPTURE_WHILE_REMOTE_TUI_RUNNING === '1' const HIDE_UNTIL_REMOTE_TUI_DONE = process.env.ORCA_E2E_HIDE_UNTIL_REMOTE_TUI_DONE === '1' diff --git a/tests/e2e/ssh-codex-repro-remote-fixtures.ts b/tests/e2e/ssh-codex-repro-remote-fixtures.ts index 3ee48187e7c..14597bc084a 100644 --- a/tests/e2e/ssh-codex-repro-remote-fixtures.ts +++ b/tests/e2e/ssh-codex-repro-remote-fixtures.ts @@ -135,7 +135,7 @@ async function insertCodexHistory(frame) { const phase = String(frame).padStart(4, '0') + '.' + index await write('\\r\\n') await write(\`\\x1b[48;2;72;72;72m\\x1b[K\`) - await write(\`\\x1b[38;2;220;220;220;48;2;72;72;72m\${pad('gpt-5.5 high · ~/code/pr-12250-migration-compare-move-baseprice-claim · /ps to view · /stop to close ' + phase, width)}\\x1b[0m\`) + await write(\`\\x1b[38;2;220;220;220;48;2;72;72;72m\${pad('gpt-5.5 high · ' + phase + ' · ~/code/pr-12250-migration-compare-move-baseprice-claim · /ps to view · /stop to close', width)}\\x1b[0m\`) } await write('\\x1b[r') await write(\`\\x1b[\${viewportBottom};1H\`) @@ -176,7 +176,7 @@ for (let frame = 0; frame < ${REMOTE_CODEX_FIXTURE_FRAMES}; frame += 1) { await reverseIndexCodexHistory(frame) } if (frame % 9 === 0) { - await grayScrollLine(\`gpt-5.5 high · ~/code/pr-12250-migration-compare-move-baseprice-claim · /ps to view · /stop to close \${frame}\`) + await grayScrollLine(\`gpt-5.5 high · \${frame} · ~/code/pr-12250-migration-compare-move-baseprice-claim · /ps to view · /stop to close\`) } await sleep(${REMOTE_CODEX_FIXTURE_FRAME_DELAY_MS}) } From 2e2ecc5193313fcded1a8b340340d2075fde4db9 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 17:26:54 -0700 Subject: [PATCH 12/17] test: order restart fixture readiness around daemon recovery (#18949) --- ...minal-host-restart-background-sync.spec.ts | 21 +++++++++++++++++++ .../restart-restore-terminal-input.spec.ts | 3 ++- 2 files changed, 23 insertions(+), 1 deletion(-) diff --git a/tests/e2e/paired-remote-terminal-host-restart-background-sync.spec.ts b/tests/e2e/paired-remote-terminal-host-restart-background-sync.spec.ts index 9855d8d92b0..cf2f96e9c84 100644 --- a/tests/e2e/paired-remote-terminal-host-restart-background-sync.spec.ts +++ b/tests/e2e/paired-remote-terminal-host-restart-background-sync.spec.ts @@ -283,6 +283,24 @@ async function expectTerminalInteractive( } async function moveHostAwayFromWorktree(page: Page, targetWorktreeId: string): Promise { + await expect + .poll( + () => + page.evaluate(async (targetId) => { + const state = window.__store?.getState() + const target = state?.allWorktrees().find((worktree) => worktree.id === targetId) + if (!state || !target) { + return false + } + await state.fetchWorktrees(target.repoId) + return window + .__store!.getState() + .allWorktrees() + .some((worktree) => worktree.repoId === target.repoId && worktree.id !== targetId) + }, targetWorktreeId), + { message: 'Seeded alternate host worktree never loaded' } + ) + .toBe(true) const alternateWorktreeId = await page.evaluate((targetId) => { const state = window.__store?.getState() const alternate = state?.allWorktrees().find((worktree) => worktree.id !== targetId) @@ -423,6 +441,9 @@ test('foregrounds a preserved daemon PTY after the paired host relaunches', asyn expect(reconnectControl.ptyId).not.toBe(target.ptyId) await openClientTab(client.page, worktreeId, reconnectControl.webTabId) await waitForPaneConnected(client.page, reconnectControl.webTabId) + await expect + .poll(() => readPaneContent(client!.page, reconnectControl.webTabId), { timeout: 30_000 }) + .toContain('READY') await expectTerminalInteractive(client, reconnectControl, 'y') } finally { if (client) { diff --git a/tests/e2e/restart-restore-terminal-input.spec.ts b/tests/e2e/restart-restore-terminal-input.spec.ts index 1ceed4254c5..79528fabffe 100644 --- a/tests/e2e/restart-restore-terminal-input.spec.ts +++ b/tests/e2e/restart-restore-terminal-input.spec.ts @@ -239,7 +239,6 @@ test('restored pane recovers input after the daemon un-wedges', async (// oxlint const second = await session.launch() secondApp = second.app - await settleRestoredLaunch(second.page) // Field-fidelity check, not a hard gate: does the pane paint restored // content while its PTY attach cannot complete? That visible-but-dead @@ -258,6 +257,8 @@ test('restored pane recovers input after the daemon un-wedges', async (// oxlint } stoppedDaemonPid = null + // Session readiness requires a daemon response; resume it before waiting for restoration. + await settleRestoredLaunch(second.page) await expectRestoredPaneAcceptsInput( second.page, `daemon wedged during relaunch (painted while wedged: ${paintedWhileWedged}, ` + From 54a8afc91de07e53e3d1de3791dc1c5ffe709f9b Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:27:29 -0400 Subject: [PATCH 13/17] fix(orchestration): typed error codes for dispatch and worker-start refusals (#18902) * fix(orchestration): typed error codes for dispatch and worker-start refusals orchestration dispatch (and worker-start, which composes it) surfaced task not found, task not ready, and inject rejected as the same bare runtime_error, so an agent reading the receipt could not choose between creating the task, waiting on dependencies, or picking another terminal. Add task_not_found (data.taskId), task_not_ready (data.status, data.unmetDependencies), and inject_rejected (data.terminal, data.reason), each carrying data.nextSteps so every shipped CLI already prints the recovery. worker-start's not-ready refusal moves from task_not_startable to task_not_ready with the same detail. runtime_error stays for genuinely unexpected failures. Proven red-first from RpcDispatcher through the CLI's own failure formatting, plus an SSH bridge test that the host CLI's typed refusal relays unchanged. * test(orchestration): load CLI formatter at runtime in the dispatch-code test The composite node typecheck (config/tsconfig.node.json without --composite false, as CI runs it) rejects a static import of src/cli from a main test with TS6307. Load the formatter and error class dynamically behind narrow structural types, as the CLI/runtime boundary test does. * fix(orchestration): keep task_not_startable and split the CLI-format proof Review on #18902: - Drop task_not_ready. worker-start already published task_not_startable for a not-ready Task, so renaming it would change an existing receipt value under old clients. dispatch now emits task_not_startable too (it was a bare runtime_error before, so this is purely additive), with the new data.status / data.unmetDependencies / data.nextSteps. - Move the refusal receipts (code, message, data) into src/shared/orchestration-dispatch-refusal-contract.ts so the runtime emits them and the CLI test formats the identical envelope. The RPC test under src/main asserts toEqual against the contract; the new src/cli/orchestration-dispatch-refusal-format.test.ts feeds those same receipts to formatCliError / reportCliError. Neither tsconfig widens and the composite typecheck CI runs is clean. * fix(orchestration): keep published refusal messages and type the DB claim guards Codex review of #18902: - Every call site keeps the exact message it published on main ("Task not found: ", "only a ready Task can start.", "cannot retry from Dispatch"); the shared contract now takes the message per site and only owns the code and data. Baseline strings are pinned as literals. - createDispatchContext's own missing/non-ready guards, including the atomic-claim loser, now emit the same typed receipt instead of a bare Error, so a dispatch that races a status change no longer flattens to runtime_error. Covered by a dispatcher-level race test. - Invalid --retry-of keeps task_not_startable but now carries status, unmetDependencies, retryOf, and a retry-specific next step. - Dependency recovery text distinguishes waiting on running deps from retrying/unblocking failed ones. - CLI test adds an unknown-code case so the old-client claim rests on an assertion, not a comment; SSH test asserts exact stdout. - Guide table narrowed to the covered preflight cases; occupancy stays runtime_error and is named as such. --- skill-guides/orchestration.md | 9 + src/cli/bundled-skill-guides.ts | 2 +- ...hestration-dispatch-refusal-format.test.ts | 77 ++++++ .../dispatch-context-store.ts | 18 +- .../worker-dispatch/worker-dispatch-start.ts | 18 +- .../orchestration-worker-dispatch-db.test.ts | 7 +- .../orchestration/task-dispatch-refusal.ts | 61 +++++ src/main/runtime/rpc/errors.ts | 1 + ...orchestration-dispatch-error-codes.test.ts | 242 ++++++++++++++++++ .../methods/orchestration-dispatch-methods.ts | 24 +- ...estration-inject-rejection-message.test.ts | 31 --- .../orchestration-inject-rejection-message.ts | 16 -- .../orchestration-tasks-dispatch.test.ts | 2 +- .../rpc/methods/orchestration-workers.ts | 9 +- ...e-cli-dispatch-refusal-passthrough.test.ts | 64 +++++ ...stration-dispatch-refusal-contract.test.ts | 67 +++++ ...orchestration-dispatch-refusal-contract.ts | 102 ++++++++ 17 files changed, 676 insertions(+), 74 deletions(-) create mode 100644 src/cli/orchestration-dispatch-refusal-format.test.ts create mode 100644 src/main/runtime/orchestration/task-dispatch-refusal.ts create mode 100644 src/main/runtime/rpc/methods/orchestration-dispatch-error-codes.test.ts delete mode 100644 src/main/runtime/rpc/methods/orchestration-inject-rejection-message.test.ts delete mode 100644 src/main/runtime/rpc/methods/orchestration-inject-rejection-message.ts create mode 100644 src/main/ssh/ssh-remote-cli-dispatch-refusal-passthrough.test.ts create mode 100644 src/shared/orchestration-dispatch-refusal-contract.test.ts create mode 100644 src/shared/orchestration-dispatch-refusal-contract.ts diff --git a/skill-guides/orchestration.md b/skill-guides/orchestration.md index 0878532c447..eab866f13d0 100644 --- a/skill-guides/orchestration.md +++ b/skill-guides/orchestration.md @@ -180,6 +180,15 @@ Dispatch rules: - After 3 consecutive failures on one task, the dispatch context circuit-breaks and the task is marked failed. - Use `task-list --brief --json` for coordinator sweeps; it collapses whitespace and caps each echoed spec at 160 characters (`spec_truncated` marks shortened rows). Omit `--brief` when the full spec is required, or when an older CLI rejects it as an unknown flag. +`dispatch` and `worker-start` refuse the following preflight cases with a stable `error.code`; read it before choosing a recovery, and treat `error.data.nextSteps` as the exact recovery text. Older hosts may omit `data`, so treat every field as optional. + +| Code | Meaning | Recovery | +| -------------------- | --------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------ | +| `task_not_found` | No Task with that id, or not in the bound Run (`data.taskId`, `data.runId`) | Check `task-list --json`; create the Task with `task-create` if it does not exist | +| `task_not_startable` | Task cannot start now: not `ready`, or invalid `--retry-of` (`data.status`, `data.unmetDependencies`, `data.retryOf`) | Wait for running dependencies with `check --wait`; retry or unblock failed ones; inspect `dispatch-show` if already dispatched | +| `inject_rejected` | `--inject` refused because no recognized agent runs in the target (`data.terminal`, `data.reason`) | Start a recognized agent there or pick another terminal; or dispatch without `--inject` and use `terminal send` | +| `runtime_error` | Any other failure, including a target terminal that already owns an active Dispatch | Read the message, inspect state, and do not retry unchanged | + ## How deep workers can nest A dispatched worker normally cannot dispatch sub-workers. Attempting it fails with diff --git a/src/cli/bundled-skill-guides.ts b/src/cli/bundled-skill-guides.ts index 06aae7bdd7d..5e68efbe8da 100644 --- a/src/cli/bundled-skill-guides.ts +++ b/src/cli/bundled-skill-guides.ts @@ -30,7 +30,7 @@ const ORCA_LINEAR_MARKDOWN = "---\nname: orca-linear\ndescription: >-\n Use Orc const ORCA_PER_WORKSPACE_ENV_MARKDOWN = "---\nname: orca-per-workspace-env\ndescription: >-\n Set up, review, debug, or validate Orca per-workspace environment recipes —\n on-demand, disposable runtimes (cloud sandboxes, VMs, or local) created fresh\n for each workspace. Covers first-time setup (provider prerequisites, the\n reusable base snapshot, the coding-agent auth snapshot, credentials, and\n state), not just the per-workspace lifecycle scripts. Use to stand up\n per-workspace environments, fix an `environmentRecipes` entry in `orca.yaml`, scaffold\n provider lifecycle scripts, or resolve an `orca vm recipe doctor` failure.\n---\n\n# Per-Workspace Environments\n\nHelp a user stand up and maintain a repo-owned per-workspace environment recipe end to end. Each\nworkspace gets its own on-demand, disposable runtime (a cloud sandbox, a VM, or a local one),\ncreated fresh and torn down after.\n\nOrca is a **thin wrapper**: you guide, detect, and scaffold; you never own the user's cloud account,\nbilling, images, or credentials.\n\n- **You DO:** sequence the setup, detect what's detectable (provider CLI present/logged-in? recipe\n present? `doctor` passing?), scaffold provider-templated scripts the user fills in, drive the slow\n snapshot/auth phases with the user, and always show the next action.\n- **You DO NOT:** create accounts, choose plans/regions, invent org/project/scope ids, store or print\n secrets, or run anything that spends money without an explicit user OK.\n\nFirst-time setup has **four phases before the per-workspace recipe runs** — easy to miss, so walk\nthem in order:\n\n1. **Prerequisites** — cloud account, provider CLI, scope/project, plan limits, git token (§2).\n2. **Base snapshot** — reusable image: tools + repo + headless build, snapshotted once (§3).\n3. **Agent-auth snapshot** — boot the base, run interactive device-auth, re-snapshot (§4).\n4. **State** — thread snapshot id / scope / project / port between phases via a state file (§6).\n\nThen the **per-workspace contract** (create/suspend/resume/destroy) runs fast (§8).\n\n**The one branch that shapes everything — connection mode:** **Orca-server** (`create` runs `orca serve`\nin the env and emits a `pairingCode`; §7c/§7f) vs **SSH** (`create` runs no server and emits a\n`connection.type:\"ssh\"` block Orca dials into; §7g/§7h). Settle this first — it changes the `create`\noutput shape and half the templates.\n\nKeep Orca's checkout behavior unchanged by default: omit `checkoutMode`, emit schema version 1, and\nlet Orca create a linked worktree. Only use `checkoutMode: provisioned-root` when the user explicitly\nwants one ephemeral machine to clone the finished workspace itself. This niche mode currently requires\ndirect SSH, an ordinary non-bare/non-sparse primary checkout at `projectRoot`, and schema version 2.\n\n**Quick-start (happy path):** interview the user (connection mode Orca-server vs SSH, provider, agent CLI,\ngit auth — §1.2) + read the provider's CLI docs → scaffold `scripts/orca-vm/` from §7 → run the\nbase-snapshot script, then the auth script (you invoke these by hand; not via `orca.yaml`) → wire\n`environmentRecipes` in `orca.yaml` → `orca vm recipe doctor --json` (free) → then the `--provision`\nself-test loop (§9) until it passes.\n\n---\n\n## 1. Setup workflow\n\nDrive these with the user. **[CHECKPOINT]** steps need explicit confirmation — they spend money, take\na long time, or need the user at the keyboard. Never create an Orca workspace or commit unless asked.\n\n1. **Inspect the repo** for an existing `environmentRecipes` entry, `scripts/orca-vm/`, a state file, or setup\n notes. If a working recipe exists, jump to Doctor (§9) instead of rebuilding.\n2. **Interview the user up front** — gather these choices and confirm them back before scaffolding\n anything. Don't pick for them (§11); don't guess.\n - **Connection mode:** how Orca attaches to the environment — an **Orca server** (the VM runs\n `orca serve` and Orca pairs over its pairing URL; worked example §7f) or **SSH** (Orca connects to\n the host over SSH; §7g). This decides the recipe's connection shape, so settle it first.\n - **Checkout ownership:** do not ask by default. Only when the user requires the environment to\n create the exact final checkout, confirm `provisioned-root` and direct SSH; otherwise omit it.\n - **Provider:** Vercel Sandbox, Fly, Modal, an existing SSH host, … For non-obvious providers, also\n ask scope/project/region and plan limits (§2). Then **read that provider's CLI/SDK docs** (or\n ` --help`) before scaffolding — you need its exact create/exec/snapshot/remove verbs.\n If a provider advertises `ssh`, verify whether it exposes a real dialable SSH target\n (host/port/user/key or proxy command) or only a provider-mediated interactive shell; Orca SSH mode\n needs the former.\n - **Coding-agent CLI + account:** which agent runs in the VM (`codex`, `claude`, …) and that the user\n has an account for it — it gets logged in during the Phase-3 auth snapshot (§4).\n - **Git auth:** the token source for cloning a private repo (`GH_TOKEN`/`GITHUB_TOKEN` or `gh auth\ntoken`; §5).\n3. **Check prerequisites (§2)** — detect the provider CLI + auth and confirm the items above are in\n place before any paid step.\n4. **Scaffold scripts + state file** from §7 (worked Vercel example: §7f; SSH host: §7g; Docker SSH:\n §7h; Windows: §7i), filling in the provider's real commands. Make them executable.\n5. **[CHECKPOINT] Build the base snapshot (§3)** — paid, slow.\n6. **[CHECKPOINT] Authenticate the agent (§4)** — interactive; the user follows a URL/code. **You cannot\n drive this step** — you run commands non-interactively, so there's no TTY for `docker exec -it` /\n `ssh -t` to prompt against. The **user** runs the Phase-3 login in their own terminal (or via the\n Claude Code harness bang-prefix — `! `, with the required space after `!`); you scaffold and drive\n the non-interactive phases around it. After kicking it off, **ask the user to report back once the login\n finishes** — you can't observe it completing, and you need that confirmation before resuming the\n non-interactive steps (base/auth commit, doctor, provision).\n7. **Wire the recipe** so `orca.yaml` points create/suspend/resume/destroy at the scripts (§8). The\n workspace composer reads `environmentRecipes` from the project's primary checkout of `orca.yaml`, **not** from\n a feature branch or worktree. So a recipe added only on a branch won't appear as a \"Run on\" option\n until that `orca.yaml` change is committed and merged to the project's primary branch. Tell the user\n this up front: `doctor`/`--provision` validate the scripts from the working copy on any branch, but\n creating a workspace from the recipe in the picker needs it on primary.\n8. **Dry-run doctor** — `orca vm recipe doctor --repo-path --json` (free, static; §9).\n Fix every failure before going live.\n9. **[CHECKPOINT] Live self-test** — get the user's OK once, then run\n `orca vm recipe doctor --provision --json` as a loop: it runs create → validates →\n destroys, and on failure returns a full transcript. Read it, fix the scripts, and re-run yourself until\n it passes (§9). Spends cloud money; the one approval covers the loop.\n10. **[CHECKPOINT] Optional workspace test** — only if asked: create a workspace via the picker, then\n verify sleep/wake/delete.\n\n---\n\n## 2. Phase 1 — Prerequisites\n\nThe user's responsibility; verify what's verifiable, ask for the rest, invent nothing. State which\nitems you verified vs. which the user asserted.\n\n- **Connection mode** (Orca server vs SSH) confirmed with the user — see §1 step 2; it shapes the recipe.\n- **Cloud account + plan** that allows sandboxes/VMs. Ask.\n- **Provider CLI installed + authenticated** — detect (`command -v `), check auth (e.g.\n `vercel whoami`). If missing, point at the provider's docs; don't log them in.\n- **Scope / project / region** the sandboxes live under. Ask; flows into every script via state.\n- **Plan / timeout / RAM caps.** Record them — e.g. Vercel Hobby caps sandbox timeout at **45m**,\n which limits both the base build and per-workspace runtime (see §10).\n- **Git token for private repos** (`GH_TOKEN`/`GITHUB_TOKEN`, or the provider's git auth; can fall back\n to `gh auth token`). See §5.\n- **Coding-agent CLI choice** (`codex`, `claude`…) and that the user has an account — it gets\n authenticated into the VM in Phase 3.\n\n---\n\n## 3. Phase 2 — Base snapshot (the reusable image)\n\nBuild **once**, snapshot, and every workspace boots from it in seconds instead of rebuilding.\nProvisioning + building takes a while (often ~20–30 min), so it runs behind a checkpoint. The script\nshape is §7a; key points:\n\n- Build the **headless Electron main only** (not the renderer) so it fits in plan RAM.\n- Use the VM image's package manager (`apt`/`dnf`/`apk`, per the base distro — not the provider brand).\n- Clone with the git token via `GIT_ASKPASS` (§5).\n- **Trap errors and remove the half-built sandbox** so a crash doesn't leave a paid resource running.\n- **Never snapshot a machine on which the Orca runtime has already run.** The first `orca serve` creates\n the runtime's user-data dir, and everything in it gets baked into the image and shared by every VM\n booted from it: the pairing keypair and device-token registry (`orca-devices.json`,\n `orca-e2ee-keypair.json`), `agent-session-authority.key`, and the build box's logs, terminal history\n and orchestration db. Confirmed: two VMs from one such snapshot emitted **identical `deviceToken` and\n `pairedDeviceId`**. Snapshot **before** the runtime has ever run, or delete the resolved user-data\n directory first: `orca_user_data_path=\"${ORCA_USER_DATA_PATH:-${XDG_CONFIG_HOME:-$HOME/.config}/orca}\"; rm -rf -- \"$orca_user_data_path\"`.\n This matches Orca's Linux precedence for custom and default paths; deleting a named file list will\n drift as Orca adds state.\n- Snapshot the stopped sandbox, parse the snapshot id, and write it + scope/project/port/repo to state.\n\n---\n\n## 4. Phase 3 — Agent-auth snapshot (interactive)\n\nThe base snapshot has the agent CLI installed but **not logged in**, and per-workspace VMs are\nephemeral — so authenticate once and bake it into a second snapshot layer. Script shape is §7b:\n\n1. Boot a sandbox from the base `snapshotId` (from state).\n2. Run the agent's login **interactively** (`--interactive --tty`); the user completes the URL/code in\n their browser. On a **headless VM this must be the device-auth flow** (e.g. `codex login --device-auth`),\n **not** plain `codex login`: the default OAuth login starts a loopback callback server on a container\n port the host browser can't reach, so it hangs. Device-auth instead prints a URL + code the user opens\n on the **host**.\n3. Verify login; **refuse to snapshot an unauthenticated VM.** Prefer the status command's **exit code**\n (most agent CLIs exit non-zero when unauthenticated). If you grep instead, agent status often goes to\n **stderr** (e.g. `codex login status` prints \"Logged in using ChatGPT\" there), so **fold stderr first**\n (`... 2>&1 | grep …`) and match the agent's **exact success line** — never `grep -qi 'logged in'`, which\n also matches \"**not** logged in\" and would commit an unauthenticated image.\n4. Re-snapshot, parse the new id, and overwrite `snapshotId` in state to the authenticated image\n (recording `authSourceSnapshotId`). Remove the auth sandbox.\n\n**You can't drive step 2 yourself** (you run commands non-interactively — no TTY). The **user** runs it in\ntheir own terminal, or via the Claude Code harness bang-prefix (`! `, with the required space after\n`!`). You scaffold/boot the sandbox and run steps 3–4, but **you cannot observe the interactive login\nfinishing** — so **ask the user to tell you when it's done** before you verify and re-snapshot.\n\nThis layer inherits §3's rule: if you started `orca serve` on the base or auth sandbox to smoke-test it,\ndelete the runtime's user-data dir (`~/.config/orca` on Linux) before re-snapshotting, or every workspace\nbooted from this image shares one pairing identity and one `agent-session-authority.key`.\n\nIf the agent's credentials are short-lived, warn that the snapshot may need periodic re-auth (§10).\n\nFor disposable runtimes, do **not** treat a host agent config directory (for example `~/.codex`) as the\nauth snapshot by bind-mounting or copying it wholesale. Agent homes often contain sqlite state, hook\napproval state, caches, logs, and host-specific env/config. Instead, authenticate/configure the agent\ninside the disposable runtime and snapshot/commit that runtime layer.\n\n---\n\n## 5. Credentials\n\n- **Never** commit secrets or put them in `userData`, recipe JSON, comments, docs, or the state file.\n- **Git token:** read from env (`GH_TOKEN`/`GITHUB_TOKEN`), falling back to `gh auth token`. Pass to the\n VM only via the provider's ephemeral `--env`. Inside the VM, use a `GIT_ASKPASS` helper with\n `x-access-token` (not the token in the clone URL) and `GIT_TERMINAL_PROMPT=0` so a missing token fails\n fast instead of hanging. When you write the helper from inside `bash -lc` under `set -u`, escape the\n positional arg and the token (`\\$1`, `\\$GH_TOKEN`) so they land **literally** and resolve at git-runtime\n — an unescaped `$1` aborts with \"unbound variable\", and a literal `$GH_TOKEN` keeps the real token out of\n the written file. `rm -f` the helper after the clone/fetch.\n- **Provider auth:** rely on the provider CLI's logged-in session, not checked-in keys.\n- **Agent auth:** lives in the authenticated snapshot (Phase 3) — never a file you write or commit.\n- State holds only **non-secret** wiring (snapshot ids, scope, project, port, repo url/ref).\n\n---\n\n## 6. State file\n\nA repo-local JSON file (e.g. `scripts/orca-vm/-state.json`) threads non-secret values between\nphases. Each script resolves values as **env var → state → built-in fallback**, and merges its outputs\nback. Phase 2 writes the base `snapshotId`; Phase 3 overwrites it with the authenticated snapshot;\nper-workspace `create` boots from `snapshotId`.\n\n```json\n{\n \"baseName\": \"orca-base\",\n \"snapshotId\": \"snap_authenticated_image_id\",\n \"authSourceSnapshotId\": \"snap_base_image_id\",\n \"scope\": \"\",\n \"project\": \"\",\n \"port\": 7331,\n \"repoUrl\": \"https://host/org/repo.git\",\n \"repoRef\": \"main\",\n \"projectRoot\": \"/abs/path/on/remote/repo\"\n}\n```\n\n---\n\n## 7. Script templates (provider-agnostic shapes)\n\nScaffold under `scripts/orca-vm/`. These are **shapes** — fill in the provider's real commands. All\nreserve stdout for the final JSON and log progress to stderr. Include a shared `json_value ` /\n`env_value ` reader (env → state → fallback) in each.\n\n**Where each script runs:**\n\n- **Local-side** (`create`/`suspend`/`resume`/`destroy` + the base-snapshot/auth scripts the user\n invokes) runs **on the user's desktop**, so it must run on their OS. macOS/Linux: `#!/usr/bin/env\nbash`, `set -euo pipefail`, quoted paths. **Windows:** a bare `.sh` won't run — scaffold `.ps1`/`.cmd`\n or require WSL/Git-Bash and point `orca.yaml` at the right launcher.\n- **Remote-side** (commands you `exec` _inside_ the Linux VM) always runs in the VM's Linux shell, so\n bash is fine there regardless of the user's OS.\n\n### 7a. Base-snapshot (`-base-snapshot.sh`) — Phase 2\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve base_name/repo_url/repo_ref/project_root/port/scope/project/timeout (env→state→fallback)\n# resolve gh token: GH_TOKEN | GITHUB_TOKEN | `gh auth token`\n# 1. provision a sandbox (timeout/vcpus/published port/snapshot retention); trap: remove on error\n# 2. remote exec (long timeout): install pkgs + gh + corepack/pnpm + agent CLI;\n# clone with GIT_ASKPASS(token); write headless main-only build config;\n# dev setup; pnpm install; build CLI; build headless electron main; smoke-check tools\n# 3. snapshot stopped sandbox; parse snapshot id (fail if unparseable)\n# 4. merge { baseName, snapshotId, projectRoot, repoUrl, repoRef, port, scope, project } into state\n# print only the state JSON to stdout\n```\n\nWorked Vercel commands for this phase are in §7f. You run this script by hand (not via `orca.yaml`),\nafter exporting the first-run inputs the state file doesn't have yet — e.g. provider scope/project, the\nrepo URL/ref, and a git token (`GH_TOKEN`); later runs read them back from state.\n\n### 7b. Auth (`-base-auth.sh`) — Phase 3\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# read source snapshot from state.snapshotId (fail if absent); auth_name=\"${base_name}-auth\"\n# 1. boot sandbox from source snapshot; trap: remove on error\n# 2. INTERACTIVE/TTY remote exec: agent login — user completes URL/code. Headless VM: MUST use the\n# device-auth flow (e.g. `codex login --device-auth`) — plain OAuth login binds a loopback callback\n# port the host can't reach and hangs. User runs this themselves (you have no interactive TTY); ask\n# them to report back when it's done before continuing.\n# 3. verify login, then refuse to snapshot if not logged in. Prefer the status command's EXIT CODE (most\n# agent CLIs exit non-zero when unauthenticated) over string-matching. If you must grep, fold stderr\n# first (`status 2>&1 | grep …` — many agents print the success line there) and match the agent's exact\n# success line; never `grep -qi 'logged in'`, which also matches \"not logged in\". Codex example: §7f.\n# 4. snapshot; parse new id\n# 5. merge { snapshotId:, authSourceSnapshotId: } into state; remove auth sandbox\n# print only the state JSON to stdout\n```\n\n### 7c. Create (`-create.sh`) — per workspace\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# read authenticated snapshotId/scope/project/port/repo*/project_root (env→state→fallback)\n# fail clearly if snapshotId is missing (point back to Phases 2–3)\n# name = orca-${ORCA_RECIPE_ID}-${ORCA_VM_INSTANCE_ID} (sanitized, length-capped)\n# 1. boot sandbox from snapshotId with a published port; capture the public URL → pairing address\n# (an externally reachable wss:// URL); trap: remove sandbox on error\n# 2. remote exec: ensure repo at desired commit; rebuild only if commit changed (cache marker)\n# 3. remote exec: start orca serve in the background and read the recipe JSON it writes (see below)\n# 4. print serve's JSON to stdout, optionally enriched with userData:\n# { schemaVersion:1, pairingCode, projectRoot, userData:{ provider, resourceId:name, snapshotId } }\n```\n\n**The exact `orca serve` invocation and its output (verified — do not improvise the flags).** Inside the\nVM, run:\n\n```bash\norca serve \\\n --port \"$PORT\" \\\n --project-root \"$ABS_REPO_PATH_ON_REMOTE\" \\\n --pairing-address \"$EXTERNAL_WSS_URL\" \\\n --recipe-json\n```\n\n**Binary name:** in a VM built from source (the Phase-2 flow), run it as `pnpm exec orca-dev serve …`\nfrom the repo root — `orca-dev` is the in-repo entrypoint and is what the §7f example uses. Plain\n`orca serve …` is the same command when the built CLI is installed on the VM's PATH. The flags/output\nare identical either way.\n\nThere is **no `--host` flag**. `--project-root` must be an absolute directory on the remote. With\n`--recipe-json` the server **stays running** and prints exactly this single object to **stdout**, then\nkeeps serving:\n\n```json\n{\n \"schemaVersion\": 1,\n \"pairingCode\": \"\",\n \"projectRoot\": \"\"\n}\n```\n\n`pairingCode` is the pairing URL, already pointing at whatever you passed as `--pairing-address` — so set\n`--pairing-address` to the externally reachable address and **pass `pairingCode` through unchanged; never\nhand-rewrite it**. Because serve runs in the foreground and doesn't exit, redirect its stdout to a file\nand poll until that file parses as JSON (and bail if the process dies — dump its stderr log). Your\n`create` script then prints that JSON (optionally merging `userData`). Concrete pattern: §7f.\n\n### 7d. Suspend / resume / destroy — per workspace\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\npayload=\"$(cat)\" # Orca passes lifecycle JSON on stdin\nresource_id=\"$(node -e 'const d=JSON.parse(process.argv[1]); process.stdout.write(d.recipeResult?.userData?.resourceId ?? \"\")' \"$payload\")\"\n[ -n \"$resource_id\" ] || { echo \"No resource id in lifecycle payload\" >&2; exit 1; }\n# suspend: provider suspend \"$resource_id\"\n# resume: provider resume \"$resource_id\"; then RE-EMIT fresh recipe JSON (pairing may change)\n# destroy: provider remove \"$resource_id\" (or set destroy: none in orca.yaml)\n```\n\n### 7e. State file — scaffold with scope/project/repo filled in and snapshot ids empty (§6).\n\n### 7f. Worked example — Vercel Sandbox (all three phases)\n\nA real, working shape (the Vercel surface is a CLI: `vercel sandbox create|exec|snapshot|remove`). Adapt\nnames; verify flags against `vercel sandbox --help` for the user's CLI version before relying on them.\nThese ground §7a (base snapshot) and §7b (auth), which are otherwise generic skeletons.\n\n**Phase 2 — base snapshot (§7a):** provision → install tools + clone + headless build → snapshot.\n\n```bash\n# provision a fresh build sandbox (retain a couple of snapshots); trap-remove on error\nvercel sandbox create --name \"$base\" --runtime node24 --timeout 30m --vcpus 4 --publish-port \"$port\" \\\n --snapshot-expiration 30d --keep-last-snapshots 2 \"${vercel_args[@]}\" >&2\n# remote build (long timeout): install pkgs+gh+pnpm+agent CLI, clone with GIT_ASKPASS (write the helper\n# with LITERAL \\$1/\\$GH_TOKEN so they resolve at git-runtime, not write-time — see §5/§7f create — then\n# `rm -f /tmp/askpass.sh`), write the headless main-only build config (drop the renderer), dev setup,\n# build CLI + headless main, smoke-check\nvercel sandbox exec \"$base\" \"${vercel_args[@]}\" --timeout 25m --env \"GH_TOKEN=$gh_token\" … -- bash -lc '…build…' >&2\n# snapshot the STOPPED sandbox and parse the id from CLI output (fail if unparseable)\nout=\"$(vercel sandbox snapshot \"$base\" --stop --expiration 30d \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$out\" >&2\nsnapshot_id=\"$(printf '%s\\n' \"$out\" | sed -nE 's/.*(snap_[A-Za-z0-9]+).*/\\1/p' | tail -1)\"\n# merge { baseName, snapshotId, scope, project, port, repoUrl, repoRef, projectRoot } into state; print state JSON\n```\n\n**Phase 3 — agent-auth snapshot (§7b):** boot the base, log the agent in interactively, re-snapshot.\n(`codex` below is an example — substitute the user's chosen agent's login/status verbs, e.g. `claude`.)\n\n```bash\nvercel sandbox create --name \"$auth\" --snapshot \"$snapshot_id\" --timeout 30m --publish-port \"$port\" \"${vercel_args[@]}\" >&2\n# INTERACTIVE — the USER runs this in their own terminal (you have no interactive TTY) and completes the\n# URL/code on the HOST. --device-auth is MANDATORY on a headless VM: plain `codex login` binds a loopback\n# callback port the host browser can't reach and hangs. Ask the user to report back when login finishes.\nvercel sandbox exec --interactive --tty \"$auth\" \"${vercel_args[@]}\" -- bash -lc 'codex login --device-auth'\n# refuse to snapshot an unauthenticated VM — fold stderr, match codex's exact success line (§4)\nvercel sandbox exec \"$auth\" \"${vercel_args[@]}\" --timeout 30s -- bash -lc 'codex login status 2>&1' | grep -Eqi 'Logged in using ChatGPT|Logged in via device' \\\n || { echo \"agent not logged in; not snapshotting\" >&2; exit 1; }\nout=\"$(vercel sandbox snapshot \"$auth\" --stop --expiration 30d \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$out\" >&2\nnew_id=\"$(printf '%s\\n' \"$out\" | sed -nE 's/.*(snap_[A-Za-z0-9]+).*/\\1/p' | tail -1)\"\n# overwrite state.snapshotId = new_id, record authSourceSnapshotId = snapshot_id; remove the auth sandbox\n```\n\n**Per-workspace `create`** (the fast path):\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve from env→state→fallback: snapshot_id, scope, project, port, repo_url, repo_ref, project_root\nvercel_args=(); [ -n \"$scope\" ] && vercel_args+=(--scope \"$scope\"); [ -n \"$project\" ] && vercel_args+=(--project \"$project\")\n[ -n \"$snapshot_id\" ] || { echo \"snapshotId missing — run Phases 2–3 first\" >&2; exit 1; }\ngh_token=\"${GH_TOKEN:-${GITHUB_TOKEN:-$(command -v gh >/dev/null 2>&1 && gh auth token 2>/dev/null || true)}}\"\nrecipe_id=\"${ORCA_RECIPE_ID:-vercel-sandbox}\"\nrecipe_id=\"${recipe_id//./-}\" # Vercel names forbid dots.\ninstance_id=\"${ORCA_VM_INSTANCE_ID:-$(date +%s)}\"\nmax_recipe_id_length=$((128 - ${#instance_id} - 6)) # Preserve the unique instance suffix.\n[ \"$max_recipe_id_length\" -gt 0 ] || { echo \"ORCA_VM_INSTANCE_ID is too long for a Vercel sandbox name\" >&2; exit 1; }\nname=\"orca-${recipe_id:0:max_recipe_id_length}-${instance_id}\"\n\n# Arm cleanup BEFORE create so a failing create can't leak a half-built paid sandbox.\ncleanup_on_error() { [ \"$?\" -ne 0 ] && vercel sandbox remove \"$name\" \"${vercel_args[@]}\" >/dev/null 2>&1 || true; }\ntrap cleanup_on_error EXIT\n\n# 1. boot from the authenticated snapshot, publish the serve port\ncreate_output=\"$(vercel sandbox create --name \"$name\" --snapshot \"$snapshot_id\" \\\n --timeout 30m --publish-port \"$port\" \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$create_output\" >&2\n# Vercel prints the published https URL; derive the external wss:// pairing address from it\npublic_url=\"$(printf '%s\\n' \"$create_output\" | sed -nE 's#.*(https://[^[:space:]]+\\.vercel\\.run).*#\\1#p' | head -1)\"\n[ -n \"$public_url\" ] || { echo \"no published URL in create output\" >&2; exit 1; }\npairing_ws=\"${public_url/https:\\/\\//wss://}\"\n\n# 2. (remote) ensure the repo is at the right commit; rebuild only if the commit changed (cache marker)\nvercel sandbox exec \"$name\" \"${vercel_args[@]}\" --timeout 20m \\\n --env \"GH_TOKEN=$gh_token\" --env \"ORCA_PROJECT_ROOT=$project_root\" \\\n --env \"ORCA_REPO_URL=$repo_url\" --env \"ORCA_REPO_REF=$repo_ref\" \\\n -- bash -lc 'set -euo pipefail; cd \"$ORCA_PROJECT_ROOT\"; \\\n # Re-establish git auth for the private-repo fetch (why + full rationale: §5); else it hangs on a prompt.\n # Load-bearing escaping: \\$1 and \\$GH_TOKEN must land LITERALLY and resolve at git-runtime. Test after\n # any edit here — reformatting the nested printf/node quoting silently breaks the fetch or leaks the token.\n if [ -n \"${GH_TOKEN:-}\" ]; then \\\n printf \"%s\\n\" \"#!/usr/bin/env bash\" \"case \\\"\\$1\\\" in *Username*) echo x-access-token;; *Password*) echo \\\"\\$GH_TOKEN\\\";; esac\" > /tmp/askpass.sh; \\\n chmod 700 /tmp/askpass.sh; export GIT_ASKPASS=/tmp/askpass.sh GIT_TERMINAL_PROMPT=0; fi; \\\n git fetch origin \"$ORCA_REPO_REF\"; \\\n git checkout -B \"$ORCA_REPO_REF\" FETCH_HEAD; \\\n rm -f /tmp/askpass.sh; \\\n c=\"$(git rev-parse HEAD)\"; [ -f .orca-built ] && [ \"$(cat .orca-built)\" = \"$c\" ] || { \\\n pnpm install --prefer-offline && pnpm run build:cli && \\\n node config/scripts/run-electron-vite-build.mjs --config config/electron-vite.vm-serve.config.ts && \\\n printf \"%s\" \"$c\" > .orca-built; }' >&2\n\n# 3. (remote) start orca serve in the background, writing recipe JSON to a file; poll until it parses\nrecipe_json=\"$(vercel sandbox exec \"$name\" \"${vercel_args[@]}\" --timeout 60s \\\n --env \"ORCA_PORT=$port\" --env \"ORCA_PROJECT_ROOT=$project_root\" --env \"ORCA_PAIRING_ADDRESS=$pairing_ws\" \\\n -- bash -lc 'set -euo pipefail; cd \"$ORCA_PROJECT_ROOT\"; rm -f /tmp/orca-recipe.json /tmp/orca-serve.log; \\\n nohup pnpm exec orca-dev serve --port \"$ORCA_PORT\" --project-root \"$ORCA_PROJECT_ROOT\" \\\n --pairing-address \"$ORCA_PAIRING_ADDRESS\" --recipe-json >/tmp/orca-recipe.json 2>/tmp/orca-serve.log /dev/null 2>&1 && { cat /tmp/orca-recipe.json; exit 0; }; \\\n kill -0 \"$pid\" 2>/dev/null || { cat /tmp/orca-serve.log >&2; exit 1; }; sleep 0.25; \\\n done; cat /tmp/orca-serve.log >&2; echo \"serve recipe JSON timed out\" >&2; exit 1')\"\n\n# 4. print serve's JSON enriched with userData (single object on stdout)\nnode -e 'const p=JSON.parse(process.argv[1]); console.log(JSON.stringify({...p, schemaVersion:1,\n userData:{...p.userData, provider:\"vercel-sandbox\", resourceId:process.argv[2], snapshotId:process.argv[3]}}))' \\\n \"$recipe_json\" \"$name\" \"$snapshot_id\"\ntrap - EXIT\n```\n\n`suspend`/`resume`/`destroy` use `vercel sandbox stop|...|remove \"$resource_id\"` reading\n`userData.resourceId` from stdin (§7d). This is the **Orca-server** connection mode (the recipe emits a\npairing URL). If the user chose **SSH** in the §1 interview, use §7g instead.\n\n### 7g. Worked example — existing SSH host (SSH connection mode)\n\nSSH mode is **fundamentally different from §7c/§7f**, not a relabeling of them:\n\n- **`create` does NOT run `orca serve` and does NOT emit a `pairingCode`.** Orca itself connects to the\n host over its SSH relay, brings up the git + filesystem providers, and imports the repo. The script's\n only job is to make the host ready and **print SSH connection details** Orca will dial.\n- The result uses a `connection` block with `type: \"ssh\"` and a `target`, **not** the flat\n `pairingCode`/`projectRoot` shape. Exact shape (Orca rejects anything else):\n\n```json\n{\n \"schemaVersion\": 1,\n \"connection\": {\n \"type\": \"ssh\",\n \"projectRoot\": \"/abs/path/to/repo/on/host\",\n \"target\": {\n \"label\": \"my-box\",\n \"host\": \"192.0.2.10\",\n \"port\": 22,\n \"username\": \"ubuntu\",\n \"identityFile\": \"~/.ssh/id_ed25519\",\n \"jumpHost\": \"bastion.example.com\",\n \"proxyCommand\": \"cloudflared access ssh --hostname %h\",\n \"relayGracePeriodSeconds\": 0,\n \"portForwards\": []\n }\n }\n}\n```\n\n`label`, `host`, `port`, `username` are required; the rest are optional — omit any you don't need.\n\nFor an explicitly requested one-VM-per-workspace checkout, the create script must read\n`ORCA_RECIPE_RESULT_SCHEMA_VERSION`, `ORCA_REPO_URL`, `ORCA_REPO_REF`, `ORCA_REPO_REF_HEAD`, and\n`ORCA_REPO_BRANCH`. Use `ORCA_REPO_REF` to fetch the selected source, but create\n`ORCA_REPO_BRANCH` at the exact `ORCA_REPO_REF_HEAD` commit; resolving the symbolic ref again can race\nwith an upstream update. `ORCA_REPO_URL` and `ORCA_REPO_REF` are a matched fetch pair, including when\nthe desktop source uses multiple remotes. Return that primary checkout at `projectRoot` and emit the\nsame SSH result with:\n\n```bash\n[ -n \"${ORCA_REPO_REF_HEAD:-}\" ] || { echo \"missing pinned source commit\" >&2; exit 1; }\ngit fetch origin \"$ORCA_REPO_REF\"\ngit cat-file -e \"${ORCA_REPO_REF_HEAD}^{commit}\"\ngit checkout -B \"$ORCA_REPO_BRANCH\" \"$ORCA_REPO_REF_HEAD\"\n```\n\n```json\n{\n \"schemaVersion\": 2,\n \"checkoutMode\": \"provisioned-root\",\n \"connection\": {\n \"type\": \"ssh\",\n \"projectRoot\": \"/abs/repo\",\n \"target\": { \"label\": \"my-box\", \"host\": \"192.0.2.10\", \"port\": 22, \"username\": \"ubuntu\" }\n }\n}\n```\n\nFail if the requested schema is not `2`; do not silently fall back to the ordinary recipe shape.\n\n**Networking → which `target` fields to set** (how _your desktop_ reaches the box — there is no\n`orca serve` URL in SSH mode):\n\n- Public IP / DNS, or a Tailscale/VPN address → `host`; SSH port → `port` (usually 22).\n- Key auth → `identityFile` (add `identitiesOnly: true` if the agent has many keys).\n- Through a bastion → `jumpHost` (a `user@host` ProxyJump) **or** a full `proxyCommand` (e.g. an access\n proxy). Use one, not both.\n- A service port the workspace needs → add entries to `portForwards`.\n- `relayGracePeriodSeconds` (optional): how long Orca keeps the SSH relay alive after the workspace\n detaches before tearing it down; `0` = tear down immediately. Leave it off unless the user wants a\n reconnect grace window.\n\n**Toolchain & agent auth on a persistent (no-snapshot) host — do this ONCE, by hand, before wiring the\nrecipe** (there's no base image to bake; the host _is_ the base). Run the §7f Phase-2 install steps and\nthe §7f Phase-3 ` login --device-auth` **directly over SSH on the host** (interactive, e.g.\n`ssh -t user@host ' login --device-auth'`). After that the host stays ready across workspaces.\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve from env→state→fallback (default unset optionals to \"\"): ssh_username, host,\n# ssh_port (default 22), identity_file, jump_host, proxy_command, project_root, repo_url, repo_ref\n: \"${identity_file:=}\"; : \"${jump_host:=}\"; : \"${proxy_command:=}\" # avoid set -u aborts on optionals\ngh_token=\"${GH_TOKEN:-${GITHUB_TOKEN:-$(command -v gh >/dev/null 2>&1 && gh auth token 2>/dev/null || true)}}\"\nssh_target=\"${ssh_username}@${host}\"\nssh_opts=(-p \"$ssh_port\"); [ -n \"$identity_file\" ] && ssh_opts+=(-i \"$identity_file\")\n# Why: a fresh host's key isn't in known_hosts; a StrictHostKeyChecking prompt would HANG a\n# non-interactive create. Pre-add the key (or set the option) so it can't block.\nssh-keyscan -p \"$ssh_port\" \"$host\" >> \"$HOME/.ssh/known_hosts\" 2>/dev/null || true\n\n# 1. ensure the repo is present and at the right commit on the host (NO orca serve here)\nssh \"${ssh_opts[@]}\" \"$ssh_target\" \\\n \"GH_TOKEN='$gh_token' GIT_TERMINAL_PROMPT=0 bash -lc '\n set -euo pipefail\n [ -d \\\"$project_root/.git\\\" ] || git clone \\\"$repo_url\\\" \\\"$project_root\\\"\n cd \\\"$project_root\\\" && git fetch origin \\\"$repo_ref\\\" && git checkout -B \\\"$repo_ref\\\" FETCH_HEAD\n '\" >&2\n\n# 2. print the SSH connection block (NO pairingCode, NO orca serve). host/port/username tell Orca's\n# relay how to dial in; identityFile/jumpHost/proxyCommand/portForwards are emitted when set.\nnode -e 'const [host,port,user,idf,jh,pc,root]=process.argv.slice(1);\n const target={ label:\"per-workspace-host\", host, port:Number(port), username:user };\n if(idf) target.identityFile=idf; if(jh) target.jumpHost=jh; if(pc) target.proxyCommand=pc;\n // add target.portForwards=[...] here if the workspace needs forwarded service ports\n console.log(JSON.stringify({ schemaVersion:1, connection:{ type:\"ssh\", projectRoot:root, target } }))' \\\n \"$host\" \"$ssh_port\" \"$ssh_username\" \"$identity_file\" \"$jump_host\" \"$proxy_command\" \"$project_root\"\n```\n\n`suspend`/`resume`/`destroy`: on a persistent host there's usually nothing to tear down — set\n`destroy: none` and omit suspend/resume. (Orca still disconnects/reconnects its own SSH relay on\nsleep/wake/delete — that's separate from these scripts.)\n\nIf the SSH host is instead an **ephemeral/snapshot-capable VM** (your hypervisor, or a cloud VM with\nimage support), keep the §7f Phase-2/3 base-image model for provisioning, but still emit the\n`connection.type:\"ssh\"` block above instead of starting `orca serve`.\n\n### 7h. Worked example — local Docker SSH (SSH connection mode)\n\nLocal Docker can model an ephemeral SSH VM without cloud cost: build a base image with `sshd`, tools,\nrepo prerequisites, and the agent CLI; run an **interactive auth container** once; then `docker commit`\nthat container as the authenticated image used by per-workspace `create`.\n\nKey points:\n\n- Publish container SSH to a random localhost port (`-p 127.0.0.1::22`) and emit\n `connection.type:\"ssh\"` with `host:\"127.0.0.1\"`, that port, `username`, `identityFile`, and\n `identitiesOnly:true`.\n- Generate a repo-local SSH key if needed, but gitignore the private/public key files.\n- **Bake SSH host keys into the base image** (`ssh-keygen -A` at **build** time; at runtime only generate\n if absent). Ephemeral containers all present the **same** host key, so `known_hosts` on `127.0.0.1`\n doesn't churn as the published port rotates across workspaces (otherwise every container's freshly\n generated key collides on `localhost` and trips host-key-changed warnings).\n- The auth image is the Docker equivalent of Phase 3: the **user** runs the agent login **inside** the\n container (you can't drive it — you have no interactive TTY), configures proxy env/config, approves\n hooks, and you commit once they report it's done. On a headless container use the **device-auth** flow\n (§4). Verify login before committing — exit code, or fold stderr and match the exact success line (§4).\n- Do not bind-mount or copy the host's full agent home into the image. Let each container have writable\n agent state; only the committed auth image should carry reusable authenticated state.\n- If committing from an interactive shell, force the runtime entrypoint back to `sshd`:\n `docker commit --change='ENTRYPOINT [\"/usr/local/bin/orca-docker-ssh-entrypoint\"]' …`.\n- `destroy` should read `recipeResult.userData.resourceId` and run `docker rm -f \"$resource_id\"`.\n\nValidation before wiring/live use:\n\n```bash\ndocker image inspect \"$auth_image\" --format '{{json .Config.Entrypoint}}'\ndocker run -d --name \"$name\" -p 127.0.0.1::22 -e \"ORCA_SSH_PUBLIC_KEY=$pubkey\" \"$auth_image\"\ndocker ps -a --filter \"name=$name\"\ndocker logs \"$name\"\nssh -i \"$key\" -p \"$port\" -o IdentitiesOnly=yes user@127.0.0.1 'codex --version'\n```\n\nIf the container exits immediately, inspect logs before the cleanup trap removes it; a committed\ninteractive image with `ENTRYPOINT [\"bash\"]` is a common cause.\n\nAlso confirm the **host key is stable** across containers: the SSH `ssh -i … 127.0.0.1` dial should not\ntrigger a host-key-changed warning when a second container reuses the port. If it does, the host keys\nweren't baked into the base image (see the `ssh-keygen -A` point above).\n\n### 7i. Windows local-side scripts\n\nThe local-side scripts run on the user's desktop. On **Windows**, a bare `.sh` won't execute. Either\nrequire WSL/Git-Bash (and point `orca.yaml` at e.g. `bash ./scripts/orca-vm/.sh` via a `.cmd`\nlauncher), or scaffold PowerShell equivalents. Minimal PowerShell shape:\n\n```powershell\n#requires -Version 5\n$ErrorActionPreference = 'Stop'\n# resolve env→state→fallback; run the provider CLI / ssh the same way;\n# capture provider output; build the result object for the chosen mode and write ONE line of JSON to stdout.\n# Orca-server mode: @{ schemaVersion=1; pairingCode=$pairingCode; projectRoot=$projectRoot; userData=@{...} }\n# SSH mode: @{ schemaVersion=1; connection=@{ type=\"ssh\"; projectRoot=$projectRoot;\n# target=@{ label=$label; host=$host; port=$port; username=$user } } } (see §7g/§7h)\n($result | ConvertTo-Json -Compress -Depth 6)\n# progress/errors → Write-Error / the error stream, never stdout.\n```\n\nThe remote-side commands you run _inside_ the Linux VM stay bash regardless of the desktop OS.\n\n---\n\n## 8. Per-workspace recipe contract (the fast path)\n\nOnce the authenticated snapshot exists, this runs on every workspace create. Define recipes in\n`orca.yaml`:\n\n```yaml\nenvironmentRecipes:\n - id: cloud-sandbox\n name: Cloud Sandbox\n create: ./scripts/orca-vm/cloud-sandbox-create.sh\n suspend: ./scripts/orca-vm/cloud-sandbox-suspend.sh\n resume: ./scripts/orca-vm/cloud-sandbox-resume.sh\n destroy: ./scripts/orca-vm/cloud-sandbox-destroy.sh\n```\n\n`create` runs **locally from the repo root** and prints **one** JSON object to stdout. Its shape depends\non the connection mode chosen in §1:\n\n**Orca-server mode** — boot the env, start `orca serve` in it, and print serve's result:\n\n```json\n{\n \"schemaVersion\": 1,\n \"pairingCode\": \"orca-pairing-code-or-url\",\n \"projectRoot\": \"/absolute/path/to/repo/on/remote\",\n \"userData\": { \"provider\": \"example\", \"resourceId\": \"provider-resource-id\" }\n}\n```\n\nHere `pairingCode` (from `orca serve --recipe-json`) and `projectRoot` are required; `schemaVersion` (`1`)\nand `userData` are optional.\n\n**SSH mode** — do **not** run `orca serve`; print the `connection.type:\"ssh\"` block instead (full shape +\nworked script in §7g). `pairingCode` is **not** used in SSH mode.\n\n**Optional provisioned root** — only for direct SSH and only when explicitly requested. Add\n`checkoutMode: provisioned-root` to the recipe, require `ORCA_RECIPE_RESULT_SCHEMA_VERSION=2`, create\nthe requested `ORCA_REPO_BRANCH` at the pinned `ORCA_REPO_REF_HEAD` commit (use `ORCA_REPO_REF` only\nto fetch that commit) at the returned `projectRoot`, and emit schema version 2 with\n`checkoutMode: \"provisioned-root\"`. All recipes without this field retain the schema-v1 behavior above.\n\nLifecycle hooks (all run locally):\n\n- `create`: required. Prints recipe result JSON.\n- `suspend`: optional. Sleep; reads lifecycle payload on stdin.\n- `resume`: optional. Wake; reads payload on stdin and **prints fresh recipe JSON** (pairing may change).\n- `destroy`: optional unless `destroy: none`. Delete/cleanup; reads payload on stdin.\n\nStart Orca remotely with `orca serve --port \"$PORT\" --project-root \"$ABS_ROOT\" --pairing-address\n\"$EXTERNAL_WSS_URL\" --recipe-json` (exact flags + output in §7c). Set `--pairing-address` to the\nexternally reachable address so the emitted `pairingCode` is reachable; tunneling/port mapping is the\nscript's job.\n\nBackward compatibility: `command`→`create`, `cleanup`→`destroy`, `cleanup: none`→`destroy: none`.\nPrefer the lifecycle names.\n\n---\n\n## 9. Doctor and validation\n\nValidate in two stages — the cheap dry run first, then the live self-test.\n\n### Dry run (free, non-destructive) — always do this first\n\n`orca vm recipe doctor --repo-path --json` validates **static wiring only** — it does\n**not** boot anything. It checks: local-host execution (v1), repo path, recipe id exists,\ncreate/destroy/suspend/resume command paths resolve, suspend/resume are paired, and each script is\nexecutable (POSIX exec bit; skipped on Windows). Fix every failure here before spending any cloud money.\n\n### Live self-test (`--provision`) — diagnose and iterate yourself\n\n`orca vm recipe doctor --repo-path --provision --json` actually runs the recipe end\nto end: it executes `create`, validates the returned recipe JSON, then runs `destroy` to **tear the\nenvironment back down** (so the test leaves nothing running, as long as `destroy` works). It spends real\ncloud money, so get the user's OK **once** before starting — that one approval covers the whole loop\nbelow; do not re-ask before each run.\n\nOn failure, the JSON result includes a `provisionTranscript` with the **complete** captured output of\neach stage so you can self-diagnose without asking the user to relay logs:\n\n```json\n{\n \"ok\": false,\n \"checks\": [{ \"id\": \"recipe.provision\", \"status\": \"fail\", \"message\": \"…\" }],\n \"provisionTranscript\": {\n \"provision\": { \"exitCode\": 0, \"signal\": null, \"stdout\": \"…\", \"stderr\": \"…\", \"parseError\": \"…\" },\n \"destroy\": { \"exitCode\": 0, \"signal\": null, \"stdout\": \"…\", \"stderr\": \"…\" }\n }\n}\n```\n\n**Run it as a loop:** read `provisionTranscript.provision.stderr` / `.stdout` / `.parseError` (and\n`destroy.*`), fix the script, and re-run `--provision` until `ok` is `true` — iterating on your own\nrather than waiting for the user to paste errors. Common reads: a non-empty `stderr` with `exitCode 0`\nplus a `parseError` means `create` ran but printed something other than the single recipe-result JSON on\nstdout (often a stray `echo` — route it to stderr, see §10); a non-zero `exitCode` is a provider/script\nfailure described in `stderr`. Each stream is redacted and capped (head+tail) — large logs keep both the\nsetup context and the failure.\n\nThe self-test cannot see provider-side truth beyond what the scripts print, so still confirm: state has a\npopulated **authenticated** `snapshotId` (Phases 2–3 done), and `destroy` is implemented/tested (or\nexplicitly `none` — in which case the self-test won't tear down, so clean up manually).\n\nFor SSH recipes, also smoke-test the exact emitted target before declaring success: dial the host/port\nwith the identity/proxy settings, run `pwd`, verify the repo path, check the agent binary, and confirm\n`destroy` removes the provider resource/container. For Docker, inspect the auth image entrypoint and do a\nstartup-only `docker run` before the full clone/install path.\n\n---\n\n## 10. Failure modes\n\n- **Build exceeds plan timeout (e.g. Hobby 45m).** Use enough vCPUs and a timeout covering the build;\n else split work or use a higher plan. The cap also limits per-workspace runtime — surface it.\n- **Build exceeds plan RAM.** Build the **headless main only** (drop the renderer) — the biggest fitter.\n- **Private-repo clone hangs/fails.** Wrong/missing token. Use `GIT_ASKPASS` + `GIT_TERMINAL_PROMPT=0`\n so it fails fast instead of prompting.\n- **`GIT_ASKPASS` helper aborts the clone with \"`$1: unbound variable`\".** The `printf`/heredoc that writes\n the helper inside `bash -lc` under `set -u` expanded `$1`/`$GH_TOKEN` at **write** time. Escape them\n (`\\$1`, `\\$GH_TOKEN`) so they land literally and resolve at git-runtime; this also keeps the real token\n out of the file. `rm -f` the helper afterward (§5, §7f).\n- **Agent verified as \"not logged in\" despite a good login.** `codex login status` (and similar) print\n \"Logged in …\" to **stderr**; an stdout-only `grep` misses it. Prefer the status **exit code**; if you\n grep, fold stderr first (`status 2>&1 | grep …`) and match the exact success line — not `grep -qi\n'logged in'`, which also matches \"not logged in\".\n- **Headless agent login hangs.** Plain OAuth `login` starts a loopback callback server on a VM/container\n port the host browser can't reach. Use the **device-auth** flow (`login --device-auth`) — it prints a\n URL + code the user opens on the host.\n- **`known_hosts` host-key churn on local Docker.** Each ephemeral container regenerating its SSH host key\n collides on `127.0.0.1` as the published port rotates. Bake host keys into the base image at build time\n (`ssh-keygen -A`; runtime generates only if absent) so all containers share one stable key (§7h).\n- **Snapshot expired/evicted.** If `create` hits an unknown snapshot id, rerun Phases 2–3 and update\n `snapshotId`.\n- **Agent auth didn't persist.** Confirm `snapshotId` points at the **authenticated** snapshot; re-run\n Phase 3. Warn that short-lived tokens may need periodic re-auth.\n- **Agent auth copied from the host breaks.** Do not bind-mount/copy a full host agent home; sqlite\n files can be unwritable or host-specific, hooks may need approval again, and config may reference\n local-only env vars. Authenticate inside the runtime and snapshot/commit that layer.\n- **Docker auth image exits immediately.** Inspect `docker image inspect … .Config.Entrypoint` and\n `docker logs`. If the image was committed from an interactive shell, reset the entrypoint to the SSH\n entrypoint during `docker commit`.\n- **Leaked paid resource.** Every long script must trap errors and remove the sandbox it created.\n- **`create` emits non-JSON on stdout.** A stray `echo` corrupts the result — stdout is for the final\n JSON only; everything else to stderr. The `--provision` self-test surfaces this as `exitCode 0` + a\n `parseError` with the offending stdout in `provisionTranscript` (§9).\n\n---\n\n## 11. Boundaries\n\n- Don't create accounts, choose plans/regions, or invent scope/project/org/image/billing ids.\n- Don't invent or store credentials; no secrets in `userData`, state, comments, docs, or commits.\n- Don't run paid/long phases (base snapshot, auth, live test) without an explicit OK.\n- Don't hide provider errors behind generic messages — preserve actionable stderr.\n- Don't make Orca own provider lifecycle beyond invoking the configured scripts.\n- Don't commit or create an Orca workspace unless asked.\n" // oxfmt-ignore -const ORCHESTRATION_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Use Orca orchestration for structured multi-agent coordination: threaded\n messages, blocking ask/reply flows, task dispatch, worker_done/escalation\n waits, task DAGs, decision gates, or coordinator loops. Use `orca-cli`\n instead for full ownership handoffs, including requests phrased as \"hand\n off\", \"handoff\", \"handover\", \"give this to another agent\", or \"another\n worktree\" when the user did not explicitly ask to supervise, monitor, wait\n for results, or coordinate a DAG. Use `orca-cli` for terminal control,\n lightweight terminal prompts, shell commands, Orca worktree management,\n reading or waiting on terminals, and the Orca embedded browser. Use Computer\n Use for external browser windows, webviews, Orca app UI, or desktop UI\n outside Orca's embedded browser only when the task requires OS/window-level\n control such as focus, menus, dialogs, coordinates, or screenshots. Use\n `orca-cli` for Orca's embedded pages and a page-automation tool such as\n Playwright or CDP for external pages.\n---\n\n# Orca Inter-Agent Orchestration\n\nOrchestration is Orca's structured coordination layer for agent messages, task ownership, dispatch state, and worker completion tracking.\n\nUse this skill when coordination state matters. For lightweight terminal prompts or basic worktree/terminal/built-in-browser control, use `orca-cli`.\n\n## Tool Boundary\n\nIf a task says to use Orca orchestration, the coordinator must create or bind a Run, create the Task with `orca orchestration task-create`, then attach the worker with either the preferred `orca orchestration worker-start` composition or the low-level `orca orchestration dispatch --inject` path.\n\nDo not substitute non-Orca subagent tools, generic agent-spawn APIs, or chat-only parallel worker features. Those may create useful workers, but they do not create Orca task/dispatch provenance, injected lifecycle preambles, `worker_done` authority, or decision gates.\n\nBefore claiming a worker was orchestrated, verify the task/dispatch exists:\n\n```bash\norca orchestration task-list --json\norca orchestration dispatch-show --task --json\n```\n\nIf the work was accidentally run outside Orca orchestration, say so plainly. To repair provenance, rerun or revalidate the needed work through a fresh Orca terminal plus injected dispatch; do not retroactively describe the external worker as orchestrated.\n\n## When To Use\n\n- Send/reply/ask between agent terminals with persistent messages.\n- Dispatch structured tasks to workers and wait for `worker_done` or `escalation`.\n- Track task DAGs with dependencies.\n- Run coordinator loops or decision gates.\n\nDo not use orchestration merely because the user says \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", or asks for another worktree/agent/model/effort. Those are full ownership transfers unless the user explicitly asks to supervise, monitor, wait for worker completion/results, coordinate a DAG, use decision gates, or keep a blocking ask/reply loop.\n\n## Preconditions\n\n- `orca status --json` should show a running runtime.\n- `orca` must be on PATH (`orca-ide` on Linux).\n- The orchestration experimental feature must be enabled in Settings > Experimental.\n- `orca orchestration` commands are RPC calls to the running Orca runtime.\n\n## Contract Migration\n\nOrca adopts a live pre-update orchestration assignment into an ordinary Run. Adoption preserves the existing agent process, PTY/session, terminal handle, tab/leaf/pane, worktree or folder workspace, Task, and Dispatch; it never restarts or replaces the worker. The retired scheduler is not revived, and a newly created attempt uses the current grammar.\n\nTreat the authority label on injected or formatted messages as definitive:\n\n- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported command printed with the message, using the same CLI executable and arguments that the original prompt supplied.\n- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded, at-least-once cutover replay. Process it idempotently and acknowledge it only through the exact displayed guidance.\n- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or lifecycle action.\n- An unlabeled current message uses the current guide and current grammar.\n\nAn explicitly selected current Run, attested current Run binding, current Dispatch, or federated attachment takes precedence over legacy fallback. A retained adoption record alone never turns a current command into a legacy call.\n\nDatabase provenance, an old-looking terminal, or a legacy Run ID does not prove mutation authority. If the runtime cannot prove liveness, principal ownership, capability, or the exact legacy contract, it degrades to read-only inspection and must not fall back to local execution. Exact recovery may restore the already-live PTY once in its original inactive background tab. It must not spawn, write, signal, stop, switch, focus, split, or inject a terminal. Loss of lifecycle authority does not invalidate the existing assignment, process, or filesystem work.\n\nCompatibility retries have narrow guarantees. A pending ask, a reply, a final Dispatch settlement, and a consuming check have durable recovery identities. A-era heartbeat and escalation calls remain at-least-once across a manual A-to-B retry because identical later signals may be intentional. If an A-era ask may already have been answered, run the exact non-consuming recovery check printed by the runtime first; after its answer is printed and acknowledged, a new invocation with the same question creates a new question. Never guess among multiple identical question threads.\n\nWhen a compatibility or recovery command returns structured next-step arguments, run those exact arguments with the same CLI executable. The arguments intentionally omit the executable name so the guidance works with `orca`, `orca-ide`, `orca-dev`, or another configured Orca CLI command. Do not translate the command from memory, broaden its recipient, or retry it as a current mutation unless the returned guidance explicitly says to.\n\nOn packaged Windows, a legacy ask uses a two-step commit/resume protocol. The initial command durably commits the question, prints its exact `ask --resume ` command, and exits with launcher status `75`; it does not wait for the answer. Run that exact resume command after the launcher or update boundary. Resume is idempotent and read-oriented: it waits for the already-committed question and does not create another one. For a WSL process that received compatibility proof at launch, use the printed executable `orca-ide` WSL resume command so the same distro and packaged launcher authority are preserved; do not substitute a PATH-resolved local CLI. Older WSL processes that never received the hidden launch token remain lifecycle read-only after the update, even while their terminal and filesystem work continue.\n\nLegacy inspection remains available without consuming mail:\n\n```bash\norca orchestration run-list --json\n# run_legacy_local is an empty audit tombstone after adoption.\norca orchestration run-show --id run_legacy_local --json\n# In run-list, find the ordinary Run whose objective is:\n# \"Recovered orchestration work from a contract update\"\norca orchestration run-show --id --json\norca orchestration task-list --run --json\norca orchestration inbox --full --json\norca orchestration check --terminal --peek --format --json\norca terminal read --terminal --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\n```\n\nIf the original coordinator is unavailable or cannot prove its retained authority, a current coordinator may explicitly take over the adopted Run from its own live agent terminal:\n\n```bash\norca orchestration run-use --id --takeover-legacy --json\norca orchestration check --run --json\n```\n\nTakeover fences only the old coordinator, binds the current one, and moves pending worker mail into current Run Delivery. It is bound to the authenticated invoking terminal; `--from` cannot name another coordinator. Live legacy workers keep their original Tasks, Dispatches, processes, filesystems, and old prompt commands; their later questions, escalations, and completion reports route to the current coordinator. Do not use takeover while the original coordinator is still actively coordinating, because its later lifecycle mutations are rejected.\n\nDo not launch a replacement editor merely because the desktop app or runtime was updated. If adoption cannot prove continuing authority, keep the original worker as the only editor until it reaches a stable handoff point, then use a new current Dispatch in a conflict-free placement for any remaining work.\n\n## Ownership\n\nNew orchestration messages and tasks belong to one explicitly bound Run. A Run is only a durable namespace and coordinator inbox; it never schedules or places workers. Lifecycle authority comes from the active Dispatch, and terminal handles remain routing metadata rather than durable identity. Send `worker_done` and `heartbeat` from the worker's own terminal; Orca routes them to that Dispatch's Run.\n\nClassify inherited context before sending lifecycle messages:\n\n- Coordinated subtask: a live coordinator owns the DAG and waits on this dispatch. Follow the preamble exactly, including `worker_done`, heartbeat/status, `ask`, and `escalation`.\n- Full handoff means ownership transfer, not supervised dispatch. The original actor is not monitoring a DAG, so do not create lifecycle obligations unless the user explicitly asks you to supervise.\n- Classify requests containing \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"another agent\", or \"another worktree\" as full handoffs by default, even when the user names a custom model or reasoning effort.\n- Use supervised orchestration only when the user explicitly asks you to \"supervise\", \"monitor\", \"wait\", \"track completion\", \"wait for worker_done\", return results, coordinate a DAG, use a decision gate, or manage ask/reply flow.\n- Do not use `orca orchestration dispatch --inject` for full handoffs. It injects a coordinator preamble that tells the worker to send `worker_done`, heartbeat, and `ask` messages, then end its turn under the original terminal's dispatch lifecycle.\n- Do not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. Do not peek at terminal output after prompt delivery to monitor progress.\n- A review-only `worker_done` reports findings; it does not authorize coordinator file edits. After a review-only completion, synthesize findings, ask a decision gate if ownership is unclear, and dispatch or hand off fixes unless the user explicitly asked the coordinator to own fixes.\n- If the user's plan names a next owner agent (for example, \"then use opencode to create a PR\"), post-review corrections and PR prep belong to that named owner. The coordinator routes, synthesizes, asks decision gates when needed, and supervises; the named owner edits files and creates the PR.\n\nIf unclear, inspect orchestration state before sending lifecycle messages:\n\n```bash\norca orchestration task-list --json\norca terminal list --json\n# If inherited context includes a task id:\norca orchestration dispatch-show --task --json\n```\n\n## Messaging\n\n```bash\norca orchestration send --subject [--to ] [--from ] [--body ] [--type ] [--priority ] [--thread-id ] [--payload ] [--json]\norca orchestration check [--terminal ] [--ack ] [--peek|--all] [--types ] [--format] [--wait] [--timeout-ms ] [--json]\norca orchestration reply --id --body [--from ] [--json]\norca orchestration ask (--question |--resume ) [--options ] [--timeout-ms ] [--from ] [--json]\norca orchestration inbox [--limit ] [--json]\n```\n\nRules:\n\n- Omit `--from` unless impersonating another terminal; Orca auto-resolves it from the current terminal.\n- A coordinator `check` returns the bound Run's oldest FIFO Delivery (up to 50 messages) and replays that exact batch until `--ack `. Process every message before acknowledging; `check --ack --wait` acknowledges, checks, and waits in one operation.\n- Use `--peek` and `--all` only for read-only history/debugging. Type filters decide when a waiter wakes; the returned actionable Delivery is still the oldest full batch.\n- Use `dispatch:` for coordinator guidance to one supervised worker. Orca routes that stable address locally or through the connected-server relay; do not substitute a remote terminal handle.\n- Terminal handles remain appropriate for low-level pre-Dispatch messaging. Prefer `agentTerminalHandle` from the create response, fall back to `startupTerminal.handle` for older runtimes, then re-resolve with `orca terminal list --worktree ... --json` if missing or stale. Continue with the replacement handle only; never dual-send to old and new handles.\n- `terminal list --json` omits `visualLayouts` because handle recovery does not need topology. Add `--include-visual-layouts` only for explicit tab and pane inspection.\n- `orca orchestration check --peek --format --json` returns locally formatted unread mail without consuming it; it never writes to terminal input or remotely wakes another terminal. Use `orchestration dispatch --inject` to deliver a tracked task, or `terminal send` when an existing agent needs a free-form prompt.\n- While supervising workers manually, use `check --wait --types worker_done,escalation,question --timeout-ms ` instead of sleep/poll loops. Process the whole Delivery, reply to `question` messages with `orca orchestration reply --id --body --json`, then acknowledge and keep waiting.\n- `check --json` prints exactly one JSON document on stdout. While `--wait` blocks it also prints keepalive lines (`{\"_keepalive\":true,...}`) to stderr so you can tell the process is alive; those are never on stdout. Do not merge the streams before a parser — `check --wait --json 2>&1 | ` fails with \"Extra data: line 2\". Pipe stdout only.\n- Treat a `check --wait` timeout or `{count:0}` as a checkpoint, not a worker failure. Long coding tasks routinely run 15-60 minutes; keep using rolling waits unless you receive `worker_done`/`escalation`, the terminal exits or disappears, or the user explicitly asks you to stop.\n- Heartbeats and visible terminal activity mean the worker is alive, not done. Do not stop, close, kill, or restart a worker just because it has not produced a completion message yet.\n- Use `ask` when a worker needs a blocking answer from the coordinator; it defaults to the active Dispatch's Run. Timeout or disconnect leaves the question pending, so resume by its original message ID instead of asking again.\n- `check --wait` returns one bounded Delivery, not every future completion. Process every message, acknowledge it, then keep waiting until every expected Dispatch settles.\n- Group addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`, `@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`.\n- Message types include `status`, `dispatch`, `worker_done`, `merge_ready`, `escalation`, `handoff`, `question`, `decision_gate` (legacy/gates), and `heartbeat`.\n- Use group addresses only for messages that are genuinely useful to many terminals, such as `status` broadcasts or intentional fan-out questions. Do not send dispatch lifecycle messages to groups.\n- `worker_done` belongs to the active Dispatch and defaults to its Run mailbox; never target a group.\n- A valid `worker_done` for the active `taskId` + `dispatchId` marks the task and dispatch completed automatically. Do not follow it with `task-update --status completed`; reserve manual updates for explicit recovery or overrides.\n- `heartbeat` is also Dispatch-scoped. Include both IDs and omit `--to` so Orca uses the owning Run; use `status` for broad progress updates.\n\n## Tasks And Dispatch\n\nA Run is the namespace/inbox, a Task is the work item, and a Dispatch assigns one Task attempt to a terminal. Create or bind a Run once before the common loop.\n\n```bash\norca orchestration run-create --objective --json\norca orchestration task-create --spec [--deps ] [--parent ] [--json]\norca orchestration task-list [--status ] [--ready] [--brief] [--json]\norca orchestration task-update --id --status [--result ] [--json]\norca orchestration dispatch --task --to [--from ] [--inject] [--json]\norca orchestration dispatch-show --task [--json]\n```\n\nTask statuses: `pending`, `ready`, `dispatched`, `completed`, `failed`, `blocked`.\n\nDispatch rules:\n\n- `--inject` sends the task spec plus preamble into a recognized agent CLI so it can report `worker_done`.\n- If the target is a bare shell, omit `--inject`, dispatch for tracking if needed, then send the prompt manually with `orca terminal send --terminal --text --enter --json`.\n- After 3 consecutive failures on one task, the dispatch context circuit-breaks and the task is marked failed.\n- Use `task-list --brief --json` for coordinator sweeps; it collapses whitespace and caps each echoed spec at 160 characters (`spec_truncated` marks shortened rows). Omit `--brief` when the full spec is required, or when an older CLI rejects it as an unknown flag.\n\n## How deep workers can nest\n\nA dispatched worker normally cannot dispatch sub-workers. Attempting it fails with\n`nested_worker_depth_exceeded` and a message telling the worker to complete the task\nitself. Do that — do not try to route around it.\n\nThe limit is a number, not an on/off switch. `Settings -> Orchestration -> Nested worker depth`\nsets how many generations are allowed:\n\n- `1` (default): a coordinator dispatches workers; those workers do not dispatch.\n- `2`: workers may dispatch one further generation.\n\nDepth is counted from the terminal that issues the command, not from the Run. Creating a\nnew Run does not reset it — a worker that runs `run-create` then `worker-start` is still a\nworker, and still counted. This is the part that changed: the old behaviour rejected\nsub-dispatch only because a worker's terminal was not bound to a Run, so creating a Run was\nenough to slip past it.\n\nTwo limits worth knowing:\n\n- **It is a guardrail, not a security boundary.** A caller that declares another terminal's\n handle while its own launch evidence is unverifiable (an ordinary restored terminal, for\n example) can be counted as that terminal instead. Orca does not treat workers as hostile.\n- **It applies while a Dispatch is active.** After `worker_done`, or after a coordinator\n settles the task, the terminal is no longer a worker and is counted as a root again. The\n process may still be alive; that is the documented boundary, not an accident.\n\n## Preferred Supervised Worker Loop\n\nUse `worker-start` for the normal supervised path. It composes the existing worktree, terminal, readiness, and dispatch primitives while returning exact created/reused effects. Agents still choose placement and concurrency; Orca does not schedule workers or infer conflicts.\n\nCreate the Run and every independent Task first, then start all independent workers before waiting:\n\n```bash\norca orchestration run-create --objective \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration worker-start --task --worktree current --agent codex --json\norca orchestration worker-start --task --worktree current --agent claude --json\n```\n\n`current` and exact existing worktrees create a fresh agent terminal and do not rerun setup. Reuse an existing agent only with `--terminal `.\n\nFor a per-invocation Claude, Codex, or Cursor launch, pass an opaque provider model id with `--model`; add `--effort` only when that agent/model supports the level. These options apply only to fresh agent terminals, override general agent default arguments, and are reported under `launch.requested` and `launch.effective` in the receipt:\n\n```bash\norca orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`, and neither option can combine with `--terminal`. A connected worker server must advertise launch-preference support before Orca forwards either option.\n\nFor a new worktree, setup runs by default and agent-first creation reuses the returned startup agent terminal:\n\n```bash\norca orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n# Independent/top-level:\norca orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nSetup normally starts alongside the agent. Only a repository explicitly configured with `wait-for-setup` delays agent launch until setup succeeds. Use `--setup skip` or `--setup inherit` only for a concrete reason.\n\nRead the returned receipt before continuing: `ready` plus setup `running` is normal for start-immediately, while wait-for-setup returns setup `succeeded` before accepting task input. A failed or unknown start exits nonzero; inspect its `stage`, `effects`, and `residualResources` instead of guessing or automatically retrying. A wait-for-setup timeout can honestly leave setup `running`, which is not proof of failure.\n\nTo run the worker on another connected Orca server, add `--on `. The Run and Tasks remain authoritative on the current server; later commands route by Dispatch ID, so never repeat `--on`:\n\n```bash\n# Mac Run home -> Windows worker (the reverse is identical from a Windows Run home)\norca orchestration worker-start --task --on windows --worktree new-top-level --repo --name --agent codex --setup run --json\norca orchestration worker-show --dispatch --json\norca orchestration worker-read --dispatch --limit 50 --json\norca orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nRemote `current` and `new-child` are intentionally invalid because those words are ambiguous across servers. Use an exact discovered remote worktree selector or `new-top-level` with an explicit remote repo selector.\n\nThe follow-up is structured inbox mail, not prompt injection. The worker's next\n`orchestration check` receives it even when the Dispatch is on another connected Orca server.\n\n`worker-read` defaults to `--source auto`: Orca returns the exact hook-reported Codex, Claude, OpenClaude, or Grok transcript when it can prove the worker session, otherwise it returns bounded terminal output with `source: \"terminal\"` and a typed `fallbackReason`. Continue with the returned top-level `cursor`; it stays pinned to that exact source. If Orca reports `source_changed`, start a fresh read without the old cursor. Never supply or guess a provider session ID or transcript path.\n\nWait until every expected Dispatch settles, not for a fixed number of batches:\n\n```bash\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n# Process every message. For each accepted worker_done that is not immediately reused:\norca orchestration worker-release --dispatch --json\n# Acknowledge only after every message and required release decision is handled:\norca orchestration check --ack --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\nAfter processing each accepted `worker_done`, choose the terminal's next owner before you acknowledge the Delivery or wait again. If the same exact agent has an immediate follow-up Task, read the `worker.agent_terminal_handle` field of `worker-show --dispatch --json`, then run `orca orchestration worker-start --task --terminal --json` so Orca transfers cleanup ownership to the new Dispatch. Otherwise run `orca orchestration worker-release --dispatch --json`.\n\nRun `worker-release` after both succeeded and failed `worker_done` reports unless the user explicitly asked to keep that worker live. Release is post-completion cleanup, not cancellation: Orca first preserves inspectable output, then closes only the exact agent terminal owned by that settled Dispatch. Reused or pre-existing terminals, setup terminals, coordinators, active workers, user-taken-over terminals, and identities Orca cannot prove are retained. If the user explicitly asks to keep the live terminal for debugging, record that exception with `orca orchestration worker-retain --dispatch --json` instead of silently skipping cleanup. When the user is finished, the same Dispatch can be passed to `worker-release`, which clears the requested retention and releases the terminal.\n\nDo not release a worker because of a timeout, TUI idle state, heartbeat, status, question, escalation, or rejected/stale `worker_done`. If release returns `release_pending` or `release_unknown`, do not substitute `terminal close`; follow the exact recovery action in the receipt. A replayed Delivery may repeat `worker-release` safely.\n\nWorkers report exactly once using the IDs and capability injected by Orca; they do not supply Run/server/terminal identity:\n\n```bash\norca orchestration send --type worker_done --subject \"\" --body \"\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a,path/b\" --json\n# On failure, use --outcome failed; never encode failure only in prose.\n```\n\nA worker question defaults to its owning Run. Timeout leaves it pending:\n\n```bash\norca orchestration ask --question \"\" --options \"yes,no\" --timeout-ms 600000 --json\norca orchestration ask --resume --timeout-ms 600000 --json\n# Coordinator:\norca orchestration reply --id --body \"\" --json\n```\n\nRecovery is conditional, never a fixed destructive sequence:\n\n- The response was lost and named no Dispatch: run `orca orchestration request-show --request --json` first. It is read-only. `completed` means the mutation already took effect. `pending` means the original mutation is still running or Orca restarted before recording its outcome. For either state, replaying the original command with `--retry-request ` reuses the same operation identity so Orca can replay, join, or safely recover it without starting a separate duplicate. `absent` means this runtime holds no receipt under your caller identity and is not proof that nothing happened; inspect the affected state before deciding whether to retry.\n- `worker-show --dispatch ` says `ready`: keep waiting or read bounded output.\n- It proves `failed` or `stopped`: start a replacement with `worker-start --task --retry-of ` plus an explicit `--on`/`--worktree` and `--agent`/`--terminal` choice. Retry does not silently inherit placement.\n- It remains `outcome_unknown`: either `worker-stop --dispatch ` and inspect again, or explicitly `worker-abandon --dispatch ` while accepting that resources may still be live. Abandon performs no remote, process, or filesystem action.\n- `worker-stop` closes only the exact supervised agent terminal. It never deletes the worktree, setup terminal, configured tabs, or unrelated processes.\n\nLow-level `worktree create`, `terminal create`, and `dispatch --inject` remain valid recipes for custom argv or topology that `worker-start` does not express.\n\n`dispatch --inject` deliberately keeps an operator-started terminal unsupervised: it never creates a `worker_dispatches` row and `worker-stop`/`worker-abandon` never close that process. The dispatch context is still authoritative, so `worker-show`, `worker-read`, and `worker-list` report it as `unsupervised`; settled `worker-retain` and `worker-release` report `retained` with `no_owned_resource` and take no process action. Use `worker-start --terminal ` when supervision and worker lifecycle state are required.\n\n## Gates And Legacy Inspection\n\n```bash\norca orchestration gate-create --task --question [--options ] [--json]\norca orchestration gate-resolve --id --resolution [--json]\norca orchestration gate-list [--task ] [--status ] [--json]\n```\n\nUse `ask` for worker-to-coordinator questions; it creates a `question` message that the coordinator answers with `reply`. Use `gate-create` only for coordinator-managed task DAG decisions, not for answering a worker's `ask`.\n\n`coordinator-start`, `coordinator-stop`, `run`, and `run-stop` are retired scheduler commands. They perform no effects and return the current-skill recovery action. They are not aliases for lightweight Run creation or binding.\n\nRecovery only: `orca orchestration reset --tasks|--messages|--all --json` clears the selected local orchestration database state. Do not run it during active coordination unless explicitly abandoning that state.\n\n## Full Handoffs\n\nFor full ownership transfer, use non-lifecycle terminal/worktree commands and then stop monitoring unless the user asks for supervision.\n\nTreat these as full handoff requests by default: \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"send this to another agent\", \"another agent\", \"another worktree\", or \"launch another agent to own this.\" Custom model or reasoning effort words such as `gpt-5.5`, `high`, or `xhigh` do not make the handoff supervised.\n\nSupervised orchestration remains available only when the user explicitly asks for supervision or coordination: \"supervise\", \"monitor\", \"wait for worker_done\", \"wait for results\", \"track completion\", \"DAG\", \"decision gate\", \"ask/reply\", or \"coordinate workers.\"\n\nDo not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. `task-create` is also forbidden because it records coordinator-owned tracking state; if a task row is needed, the user asked for supervised orchestration. Do not create a `taskId`/`dispatchId`, inject a lifecycle preamble, wait for completion, or read the worker terminal after prompt delivery except to avoid losing the initial prompt.\n\nNew top-level worktree handoff:\n\n```bash\norca worktree create --name --no-parent --agent codex --prompt \"\" --setup run --json\n```\n\nBefore creating a new worktree from an active feature branch, decide and state whether the desired Orca lineage is child or top-level. Use child worktree lineage only when the new work is conceptually stacked under or dependent on the active worktree. For independent repo-wide fixes, standalone feature work, or unrelated follow-up tasks, create a top-level worktree with `--no-parent`.\n\nExisting terminal handoff:\n\n```bash\norca terminal send --terminal --text \"\" --enter --json\n```\n\nCustom Codex model/effort handoff:\n\n`orca worktree create --agent codex --prompt ...` launches the known Codex agent but does not accept Codex-specific `--model` or `-c model_reasoning_effort=...` arguments. When the user asks for a specific Codex model or effort, create the independent worktree first, launch Codex with the requested command in that worktree, wait only for TUI readiness if prompt delivery would otherwise race startup, send the prompt, and stop.\n\nThe two-step custom-argv path cannot enforce a repository's explicit `wait-for-setup` startup policy because the later `terminal create` is not the startup owned by `worktree create`. Use it only when the repository starts agents immediately. If the repository requires `wait-for-setup`, use an agent-first configured launcher that can preserve sequencing, or stop and ask rather than silently bypassing the policy.\n\nNote: when no repo default-terminal configuration supplies a primary terminal, bare create opens a fallback shell before `terminal create` adds the agent. Configured default tabs are materialized instead and may run real commands. Prefer `--agent` whenever custom argv is not required. With the two-step path, target only the agent handle; close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell.\n\nUse the exact full `::` worktree id returned by `orca worktree create --json`; a bare repo id cannot target the new worktree.\n\n```bash\norca worktree create --name --no-parent --setup run --json\norca terminal create --worktree id: --title --command 'codex --model gpt-5.5 -c model_reasoning_effort=\"xhigh\"' --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca terminal send --terminal --text \"\" --enter --json\n```\n\nWait only for `tui-idle` when needed to avoid losing the prompt. Do not monitor task completion.\n\n`--no-parent` only controls Orca lineage; it does not choose the Git base. If the work should start from the repo default base, omit `--base-branch` so Orca uses that default, or explicitly pass the repo default base (`origin/main`, `origin/master`, or the `orca repo show --repo --json` value); never base it on the current feature branch unless the user explicitly asks for stacked work or \"branch from current\". Put current-branch context in the prompt instead.\n\n## Worker Terminals\n\nChoose the worker location before creating a terminal. `Fresh worker` means a fresh agent session, not a new git worktree. For parallel work, create one fresh agent terminal per worker in the same required worktree, falling back to the active worktree when none is named. If the task says current worktree only, depends on uncommitted files/artifacts, or must validate/PR the current branch, keep every worker in the active worktree:\n\n```bash\norca terminal create --worktree active --title --command \"codex\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nReuse an idle agent in the required worktree only if the prompt allows reuse; otherwise create a fresh terminal there. Create a new worktree only when the user explicitly requests one or a concrete checkout or filesystem conflict makes sharing unsafe or impossible; if the user did not request it, state that conflict before running `worktree create`. Independent tasks, parallel execution, convenience, or a preference for separate checkouts are not isolation requirements.\n\nWhen a new worktree is allowed, use child lineage for isolated work that is stacked under or dependent on the active worktree, and use `--no-parent` when it is not stacked. Decide the Git base separately: `--no-parent` makes the worktree top-level in Orca, while omitted `--base-branch` uses the repo default base.\n\nFor every new worktree, pass `--setup run` so any configured repository setup hook runs. This does not mean waiting for setup before agent launch: preserve the repository's startup policy, whose default starts setup and the agent side by side. Use `--setup skip` or `--setup inherit` only when there is a concrete task-specific reason, and state that reason before creating the worktree. This rule does not rerun setup for current or existing worktrees.\n\n```bash\norca worktree create --name --agent codex --setup run --json\n# or: --agent claude | omp | pi | grok | ...\n# Read from agentTerminalHandle, falling back to startupTerminal.handle.\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nFor new-worktree workers, read the id and `agentTerminalHandle` from `worktree create`, falling back to `startupTerminal.handle` for older runtimes. Use that as the sole worker handle when present; otherwise use `terminal list` to resolve the agent handle. Omit `--repo` only inside an Orca-managed worktree; otherwise pass `--repo `.\n\n**For an allowed new worktree, use agent-first:** `--agent` reveals the new worktree and launches the selected agent **in its first terminal**, without adding a separate fallback shell for that worker. Pass `--setup run`; repo setup and default-terminal settings may add intentional tabs or splits. Do **not** run bare `worktree create` and then `terminal create --command ` for the same worker when agent-first create is available: without configured default tabs, that two-step path leaves a fallback shell + agent pair. Only use it when custom agent argv is required (for example Codex model/effort flags) or when an older CLI rejects `--agent`; if you must, message only the agent handle. Configured default tabs are intentional surfaces, so close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell. Do not run `worktree create` when the task must stay in the current worktree.\n\nUse `orca worktree create --prompt ...` or `orca terminal send ...` for full handoffs or untracked/lightweight prompts. Those paths do not attach `taskId`/`dispatchId`; the worker should not send lifecycle messages unless the prompt supplies a live orchestration preamble.\n\nSidebar lineage and orchestration lifecycle are related but not identical. A same-worktree worker may appear as a peer under that worktree in the sidebar while remaining a child dispatch in orchestration state; only an actual child worktree creates visible parent/child worktree lineage.\n\nOther terminal commands coordinators often need:\n\n```bash\norca terminal list [--worktree ] [--include-visual-layouts] [--json]\norca terminal create [--worktree ] [--title ] [--command ] [--json]\norca terminal split --terminal [--direction horizontal|vertical] [--command ] [--json]\norca terminal wait --terminal --for tui-idle --timeout-ms --json\norca terminal read --terminal --json\norca terminal send --terminal --text --enter --json\n```\n\nIf an older CLI rejects `worktree create --agent`, create the worktree normally, then run `orca terminal create --worktree --command \"codex\" --json` or `--command \"claude\"`.\n\nWait for `tui-idle` before dispatching. Always pass `--timeout-ms`; real coding tasks can take 15-60 minutes. During supervision, use rolling `check --wait` windows. If a window returns no matching message, inspect `task-list`, `terminal read`, or `terminal wait --for tui-idle` as a liveness checkpoint; if the terminal is still working or producing activity, keep waiting instead of retrying the task.\n\n## Agent Guidance\n\n- Workers with a valid live preamble must send `worker_done` exactly once from their own terminal with an explicit `--outcome succeeded` or `--outcome failed`:\n `orca orchestration send --type worker_done --subject \"\" --body \"<3-sentence summary: what you did, what you found, what's left>\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a\" --report-path \"\" --json`\n- A failed outcome is still a terminal report, but Orca records both the Dispatch and Task as failed. Never encode failure only in the subject/body.\n- After sending `worker_done`, end that dispatched turn and idle at the agent prompt. Do not autonomously start more work, poll, or attempt to close the terminal yourself. A direct user instruction takes precedence and starts ordinary user-owned work: follow it without coordinator approval or a fresh Dispatch, never refuse it because of worker/coordinator roles, and do not reuse the settled Dispatch's lifecycle IDs. A coordinator-supervised follow-up still arrives with a fresh preamble + TASK block.\n- For long tasks, send heartbeat/status only when the preamble asks for it, including both IDs:\n `orca orchestration send --type heartbeat --subject \"alive\" --payload '{\"taskId\":\"\",\"dispatchId\":\"\",\"phase\":\"implementing\"}' --json`\n- If blocked before completion, use `ask`; use `escalation` only when ownership is valid and the coordinator must intervene.\n- Treat preambles inherited through terminal history or full handoffs as stale unless the current prompt explicitly keeps that coordinator in the loop.\n- Coordinators must account for every settled worker terminal before waiting again or ending the turn: immediately reuse the exact worker for a new Dispatch, explicitly retain it at the user's request with `worker-retain`, or run `worker-release`. Do not leave a completed worker live merely to inspect output; released workers remain readable through `worker-read`.\n- Coordinators should use `task-list --ready` as external memory, dispatch parallel waves, and avoid dependency chains deeper than 3-4 steps.\n\n## Example\n\n```bash\norca terminal create --worktree active --title login-css-worker --command \"claude\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration task-create --spec \"Fix the login button CSS\" --json\norca orchestration dispatch --task --to --inject --json\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\n## Next Action\n\nCoordinator: confirm `orca status --json`, create or bind a Run, inspect `task-list`/`dispatch-show` if inheriting state, then use the explicit supervised loop (`task-create` -> `worker-start` -> `check --wait`). Use low-level terminal creation plus `dispatch --inject` only when the composed start does not express the needed topology. After every accepted `worker_done`, either transfer the exact terminal to an immediate follow-up Dispatch or run `worker-release` before the next wait.\n\nWorker: if the current prompt contains a live dispatch preamble, do the task, use `ask` for blocking questions, and send `worker_done` once with the required payload. If the preamble is stale or absent, do not send lifecycle messages; inspect state or treat the prompt as an ordinary handoff.\n" +const ORCHESTRATION_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Use Orca orchestration for structured multi-agent coordination: threaded\n messages, blocking ask/reply flows, task dispatch, worker_done/escalation\n waits, task DAGs, decision gates, or coordinator loops. Use `orca-cli`\n instead for full ownership handoffs, including requests phrased as \"hand\n off\", \"handoff\", \"handover\", \"give this to another agent\", or \"another\n worktree\" when the user did not explicitly ask to supervise, monitor, wait\n for results, or coordinate a DAG. Use `orca-cli` for terminal control,\n lightweight terminal prompts, shell commands, Orca worktree management,\n reading or waiting on terminals, and the Orca embedded browser. Use Computer\n Use for external browser windows, webviews, Orca app UI, or desktop UI\n outside Orca's embedded browser only when the task requires OS/window-level\n control such as focus, menus, dialogs, coordinates, or screenshots. Use\n `orca-cli` for Orca's embedded pages and a page-automation tool such as\n Playwright or CDP for external pages.\n---\n\n# Orca Inter-Agent Orchestration\n\nOrchestration is Orca's structured coordination layer for agent messages, task ownership, dispatch state, and worker completion tracking.\n\nUse this skill when coordination state matters. For lightweight terminal prompts or basic worktree/terminal/built-in-browser control, use `orca-cli`.\n\n## Tool Boundary\n\nIf a task says to use Orca orchestration, the coordinator must create or bind a Run, create the Task with `orca orchestration task-create`, then attach the worker with either the preferred `orca orchestration worker-start` composition or the low-level `orca orchestration dispatch --inject` path.\n\nDo not substitute non-Orca subagent tools, generic agent-spawn APIs, or chat-only parallel worker features. Those may create useful workers, but they do not create Orca task/dispatch provenance, injected lifecycle preambles, `worker_done` authority, or decision gates.\n\nBefore claiming a worker was orchestrated, verify the task/dispatch exists:\n\n```bash\norca orchestration task-list --json\norca orchestration dispatch-show --task --json\n```\n\nIf the work was accidentally run outside Orca orchestration, say so plainly. To repair provenance, rerun or revalidate the needed work through a fresh Orca terminal plus injected dispatch; do not retroactively describe the external worker as orchestrated.\n\n## When To Use\n\n- Send/reply/ask between agent terminals with persistent messages.\n- Dispatch structured tasks to workers and wait for `worker_done` or `escalation`.\n- Track task DAGs with dependencies.\n- Run coordinator loops or decision gates.\n\nDo not use orchestration merely because the user says \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", or asks for another worktree/agent/model/effort. Those are full ownership transfers unless the user explicitly asks to supervise, monitor, wait for worker completion/results, coordinate a DAG, use decision gates, or keep a blocking ask/reply loop.\n\n## Preconditions\n\n- `orca status --json` should show a running runtime.\n- `orca` must be on PATH (`orca-ide` on Linux).\n- The orchestration experimental feature must be enabled in Settings > Experimental.\n- `orca orchestration` commands are RPC calls to the running Orca runtime.\n\n## Contract Migration\n\nOrca adopts a live pre-update orchestration assignment into an ordinary Run. Adoption preserves the existing agent process, PTY/session, terminal handle, tab/leaf/pane, worktree or folder workspace, Task, and Dispatch; it never restarts or replaces the worker. The retired scheduler is not revived, and a newly created attempt uses the current grammar.\n\nTreat the authority label on injected or formatted messages as definitive:\n\n- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported command printed with the message, using the same CLI executable and arguments that the original prompt supplied.\n- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded, at-least-once cutover replay. Process it idempotently and acknowledge it only through the exact displayed guidance.\n- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or lifecycle action.\n- An unlabeled current message uses the current guide and current grammar.\n\nAn explicitly selected current Run, attested current Run binding, current Dispatch, or federated attachment takes precedence over legacy fallback. A retained adoption record alone never turns a current command into a legacy call.\n\nDatabase provenance, an old-looking terminal, or a legacy Run ID does not prove mutation authority. If the runtime cannot prove liveness, principal ownership, capability, or the exact legacy contract, it degrades to read-only inspection and must not fall back to local execution. Exact recovery may restore the already-live PTY once in its original inactive background tab. It must not spawn, write, signal, stop, switch, focus, split, or inject a terminal. Loss of lifecycle authority does not invalidate the existing assignment, process, or filesystem work.\n\nCompatibility retries have narrow guarantees. A pending ask, a reply, a final Dispatch settlement, and a consuming check have durable recovery identities. A-era heartbeat and escalation calls remain at-least-once across a manual A-to-B retry because identical later signals may be intentional. If an A-era ask may already have been answered, run the exact non-consuming recovery check printed by the runtime first; after its answer is printed and acknowledged, a new invocation with the same question creates a new question. Never guess among multiple identical question threads.\n\nWhen a compatibility or recovery command returns structured next-step arguments, run those exact arguments with the same CLI executable. The arguments intentionally omit the executable name so the guidance works with `orca`, `orca-ide`, `orca-dev`, or another configured Orca CLI command. Do not translate the command from memory, broaden its recipient, or retry it as a current mutation unless the returned guidance explicitly says to.\n\nOn packaged Windows, a legacy ask uses a two-step commit/resume protocol. The initial command durably commits the question, prints its exact `ask --resume ` command, and exits with launcher status `75`; it does not wait for the answer. Run that exact resume command after the launcher or update boundary. Resume is idempotent and read-oriented: it waits for the already-committed question and does not create another one. For a WSL process that received compatibility proof at launch, use the printed executable `orca-ide` WSL resume command so the same distro and packaged launcher authority are preserved; do not substitute a PATH-resolved local CLI. Older WSL processes that never received the hidden launch token remain lifecycle read-only after the update, even while their terminal and filesystem work continue.\n\nLegacy inspection remains available without consuming mail:\n\n```bash\norca orchestration run-list --json\n# run_legacy_local is an empty audit tombstone after adoption.\norca orchestration run-show --id run_legacy_local --json\n# In run-list, find the ordinary Run whose objective is:\n# \"Recovered orchestration work from a contract update\"\norca orchestration run-show --id --json\norca orchestration task-list --run --json\norca orchestration inbox --full --json\norca orchestration check --terminal --peek --format --json\norca terminal read --terminal --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\n```\n\nIf the original coordinator is unavailable or cannot prove its retained authority, a current coordinator may explicitly take over the adopted Run from its own live agent terminal:\n\n```bash\norca orchestration run-use --id --takeover-legacy --json\norca orchestration check --run --json\n```\n\nTakeover fences only the old coordinator, binds the current one, and moves pending worker mail into current Run Delivery. It is bound to the authenticated invoking terminal; `--from` cannot name another coordinator. Live legacy workers keep their original Tasks, Dispatches, processes, filesystems, and old prompt commands; their later questions, escalations, and completion reports route to the current coordinator. Do not use takeover while the original coordinator is still actively coordinating, because its later lifecycle mutations are rejected.\n\nDo not launch a replacement editor merely because the desktop app or runtime was updated. If adoption cannot prove continuing authority, keep the original worker as the only editor until it reaches a stable handoff point, then use a new current Dispatch in a conflict-free placement for any remaining work.\n\n## Ownership\n\nNew orchestration messages and tasks belong to one explicitly bound Run. A Run is only a durable namespace and coordinator inbox; it never schedules or places workers. Lifecycle authority comes from the active Dispatch, and terminal handles remain routing metadata rather than durable identity. Send `worker_done` and `heartbeat` from the worker's own terminal; Orca routes them to that Dispatch's Run.\n\nClassify inherited context before sending lifecycle messages:\n\n- Coordinated subtask: a live coordinator owns the DAG and waits on this dispatch. Follow the preamble exactly, including `worker_done`, heartbeat/status, `ask`, and `escalation`.\n- Full handoff means ownership transfer, not supervised dispatch. The original actor is not monitoring a DAG, so do not create lifecycle obligations unless the user explicitly asks you to supervise.\n- Classify requests containing \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"another agent\", or \"another worktree\" as full handoffs by default, even when the user names a custom model or reasoning effort.\n- Use supervised orchestration only when the user explicitly asks you to \"supervise\", \"monitor\", \"wait\", \"track completion\", \"wait for worker_done\", return results, coordinate a DAG, use a decision gate, or manage ask/reply flow.\n- Do not use `orca orchestration dispatch --inject` for full handoffs. It injects a coordinator preamble that tells the worker to send `worker_done`, heartbeat, and `ask` messages, then end its turn under the original terminal's dispatch lifecycle.\n- Do not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. Do not peek at terminal output after prompt delivery to monitor progress.\n- A review-only `worker_done` reports findings; it does not authorize coordinator file edits. After a review-only completion, synthesize findings, ask a decision gate if ownership is unclear, and dispatch or hand off fixes unless the user explicitly asked the coordinator to own fixes.\n- If the user's plan names a next owner agent (for example, \"then use opencode to create a PR\"), post-review corrections and PR prep belong to that named owner. The coordinator routes, synthesizes, asks decision gates when needed, and supervises; the named owner edits files and creates the PR.\n\nIf unclear, inspect orchestration state before sending lifecycle messages:\n\n```bash\norca orchestration task-list --json\norca terminal list --json\n# If inherited context includes a task id:\norca orchestration dispatch-show --task --json\n```\n\n## Messaging\n\n```bash\norca orchestration send --subject [--to ] [--from ] [--body ] [--type ] [--priority ] [--thread-id ] [--payload ] [--json]\norca orchestration check [--terminal ] [--ack ] [--peek|--all] [--types ] [--format] [--wait] [--timeout-ms ] [--json]\norca orchestration reply --id --body [--from ] [--json]\norca orchestration ask (--question |--resume ) [--options ] [--timeout-ms ] [--from ] [--json]\norca orchestration inbox [--limit ] [--json]\n```\n\nRules:\n\n- Omit `--from` unless impersonating another terminal; Orca auto-resolves it from the current terminal.\n- A coordinator `check` returns the bound Run's oldest FIFO Delivery (up to 50 messages) and replays that exact batch until `--ack `. Process every message before acknowledging; `check --ack --wait` acknowledges, checks, and waits in one operation.\n- Use `--peek` and `--all` only for read-only history/debugging. Type filters decide when a waiter wakes; the returned actionable Delivery is still the oldest full batch.\n- Use `dispatch:` for coordinator guidance to one supervised worker. Orca routes that stable address locally or through the connected-server relay; do not substitute a remote terminal handle.\n- Terminal handles remain appropriate for low-level pre-Dispatch messaging. Prefer `agentTerminalHandle` from the create response, fall back to `startupTerminal.handle` for older runtimes, then re-resolve with `orca terminal list --worktree ... --json` if missing or stale. Continue with the replacement handle only; never dual-send to old and new handles.\n- `terminal list --json` omits `visualLayouts` because handle recovery does not need topology. Add `--include-visual-layouts` only for explicit tab and pane inspection.\n- `orca orchestration check --peek --format --json` returns locally formatted unread mail without consuming it; it never writes to terminal input or remotely wakes another terminal. Use `orchestration dispatch --inject` to deliver a tracked task, or `terminal send` when an existing agent needs a free-form prompt.\n- While supervising workers manually, use `check --wait --types worker_done,escalation,question --timeout-ms ` instead of sleep/poll loops. Process the whole Delivery, reply to `question` messages with `orca orchestration reply --id --body --json`, then acknowledge and keep waiting.\n- `check --json` prints exactly one JSON document on stdout. While `--wait` blocks it also prints keepalive lines (`{\"_keepalive\":true,...}`) to stderr so you can tell the process is alive; those are never on stdout. Do not merge the streams before a parser — `check --wait --json 2>&1 | ` fails with \"Extra data: line 2\". Pipe stdout only.\n- Treat a `check --wait` timeout or `{count:0}` as a checkpoint, not a worker failure. Long coding tasks routinely run 15-60 minutes; keep using rolling waits unless you receive `worker_done`/`escalation`, the terminal exits or disappears, or the user explicitly asks you to stop.\n- Heartbeats and visible terminal activity mean the worker is alive, not done. Do not stop, close, kill, or restart a worker just because it has not produced a completion message yet.\n- Use `ask` when a worker needs a blocking answer from the coordinator; it defaults to the active Dispatch's Run. Timeout or disconnect leaves the question pending, so resume by its original message ID instead of asking again.\n- `check --wait` returns one bounded Delivery, not every future completion. Process every message, acknowledge it, then keep waiting until every expected Dispatch settles.\n- Group addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`, `@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`.\n- Message types include `status`, `dispatch`, `worker_done`, `merge_ready`, `escalation`, `handoff`, `question`, `decision_gate` (legacy/gates), and `heartbeat`.\n- Use group addresses only for messages that are genuinely useful to many terminals, such as `status` broadcasts or intentional fan-out questions. Do not send dispatch lifecycle messages to groups.\n- `worker_done` belongs to the active Dispatch and defaults to its Run mailbox; never target a group.\n- A valid `worker_done` for the active `taskId` + `dispatchId` marks the task and dispatch completed automatically. Do not follow it with `task-update --status completed`; reserve manual updates for explicit recovery or overrides.\n- `heartbeat` is also Dispatch-scoped. Include both IDs and omit `--to` so Orca uses the owning Run; use `status` for broad progress updates.\n\n## Tasks And Dispatch\n\nA Run is the namespace/inbox, a Task is the work item, and a Dispatch assigns one Task attempt to a terminal. Create or bind a Run once before the common loop.\n\n```bash\norca orchestration run-create --objective --json\norca orchestration task-create --spec [--deps ] [--parent ] [--json]\norca orchestration task-list [--status ] [--ready] [--brief] [--json]\norca orchestration task-update --id --status [--result ] [--json]\norca orchestration dispatch --task --to [--from ] [--inject] [--json]\norca orchestration dispatch-show --task [--json]\n```\n\nTask statuses: `pending`, `ready`, `dispatched`, `completed`, `failed`, `blocked`.\n\nDispatch rules:\n\n- `--inject` sends the task spec plus preamble into a recognized agent CLI so it can report `worker_done`.\n- If the target is a bare shell, omit `--inject`, dispatch for tracking if needed, then send the prompt manually with `orca terminal send --terminal --text --enter --json`.\n- After 3 consecutive failures on one task, the dispatch context circuit-breaks and the task is marked failed.\n- Use `task-list --brief --json` for coordinator sweeps; it collapses whitespace and caps each echoed spec at 160 characters (`spec_truncated` marks shortened rows). Omit `--brief` when the full spec is required, or when an older CLI rejects it as an unknown flag.\n\n`dispatch` and `worker-start` refuse the following preflight cases with a stable `error.code`; read it before choosing a recovery, and treat `error.data.nextSteps` as the exact recovery text. Older hosts may omit `data`, so treat every field as optional.\n\n| Code | Meaning | Recovery |\n| -------------------- | --------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------ |\n| `task_not_found` | No Task with that id, or not in the bound Run (`data.taskId`, `data.runId`) | Check `task-list --json`; create the Task with `task-create` if it does not exist |\n| `task_not_startable` | Task cannot start now: not `ready`, or invalid `--retry-of` (`data.status`, `data.unmetDependencies`, `data.retryOf`) | Wait for running dependencies with `check --wait`; retry or unblock failed ones; inspect `dispatch-show` if already dispatched |\n| `inject_rejected` | `--inject` refused because no recognized agent runs in the target (`data.terminal`, `data.reason`) | Start a recognized agent there or pick another terminal; or dispatch without `--inject` and use `terminal send` |\n| `runtime_error` | Any other failure, including a target terminal that already owns an active Dispatch | Read the message, inspect state, and do not retry unchanged |\n\n## How deep workers can nest\n\nA dispatched worker normally cannot dispatch sub-workers. Attempting it fails with\n`nested_worker_depth_exceeded` and a message telling the worker to complete the task\nitself. Do that — do not try to route around it.\n\nThe limit is a number, not an on/off switch. `Settings -> Orchestration -> Nested worker depth`\nsets how many generations are allowed:\n\n- `1` (default): a coordinator dispatches workers; those workers do not dispatch.\n- `2`: workers may dispatch one further generation.\n\nDepth is counted from the terminal that issues the command, not from the Run. Creating a\nnew Run does not reset it — a worker that runs `run-create` then `worker-start` is still a\nworker, and still counted. This is the part that changed: the old behaviour rejected\nsub-dispatch only because a worker's terminal was not bound to a Run, so creating a Run was\nenough to slip past it.\n\nTwo limits worth knowing:\n\n- **It is a guardrail, not a security boundary.** A caller that declares another terminal's\n handle while its own launch evidence is unverifiable (an ordinary restored terminal, for\n example) can be counted as that terminal instead. Orca does not treat workers as hostile.\n- **It applies while a Dispatch is active.** After `worker_done`, or after a coordinator\n settles the task, the terminal is no longer a worker and is counted as a root again. The\n process may still be alive; that is the documented boundary, not an accident.\n\n## Preferred Supervised Worker Loop\n\nUse `worker-start` for the normal supervised path. It composes the existing worktree, terminal, readiness, and dispatch primitives while returning exact created/reused effects. Agents still choose placement and concurrency; Orca does not schedule workers or infer conflicts.\n\nCreate the Run and every independent Task first, then start all independent workers before waiting:\n\n```bash\norca orchestration run-create --objective \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration worker-start --task --worktree current --agent codex --json\norca orchestration worker-start --task --worktree current --agent claude --json\n```\n\n`current` and exact existing worktrees create a fresh agent terminal and do not rerun setup. Reuse an existing agent only with `--terminal `.\n\nFor a per-invocation Claude, Codex, or Cursor launch, pass an opaque provider model id with `--model`; add `--effort` only when that agent/model supports the level. These options apply only to fresh agent terminals, override general agent default arguments, and are reported under `launch.requested` and `launch.effective` in the receipt:\n\n```bash\norca orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`, and neither option can combine with `--terminal`. A connected worker server must advertise launch-preference support before Orca forwards either option.\n\nFor a new worktree, setup runs by default and agent-first creation reuses the returned startup agent terminal:\n\n```bash\norca orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n# Independent/top-level:\norca orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nSetup normally starts alongside the agent. Only a repository explicitly configured with `wait-for-setup` delays agent launch until setup succeeds. Use `--setup skip` or `--setup inherit` only for a concrete reason.\n\nRead the returned receipt before continuing: `ready` plus setup `running` is normal for start-immediately, while wait-for-setup returns setup `succeeded` before accepting task input. A failed or unknown start exits nonzero; inspect its `stage`, `effects`, and `residualResources` instead of guessing or automatically retrying. A wait-for-setup timeout can honestly leave setup `running`, which is not proof of failure.\n\nTo run the worker on another connected Orca server, add `--on `. The Run and Tasks remain authoritative on the current server; later commands route by Dispatch ID, so never repeat `--on`:\n\n```bash\n# Mac Run home -> Windows worker (the reverse is identical from a Windows Run home)\norca orchestration worker-start --task --on windows --worktree new-top-level --repo --name --agent codex --setup run --json\norca orchestration worker-show --dispatch --json\norca orchestration worker-read --dispatch --limit 50 --json\norca orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nRemote `current` and `new-child` are intentionally invalid because those words are ambiguous across servers. Use an exact discovered remote worktree selector or `new-top-level` with an explicit remote repo selector.\n\nThe follow-up is structured inbox mail, not prompt injection. The worker's next\n`orchestration check` receives it even when the Dispatch is on another connected Orca server.\n\n`worker-read` defaults to `--source auto`: Orca returns the exact hook-reported Codex, Claude, OpenClaude, or Grok transcript when it can prove the worker session, otherwise it returns bounded terminal output with `source: \"terminal\"` and a typed `fallbackReason`. Continue with the returned top-level `cursor`; it stays pinned to that exact source. If Orca reports `source_changed`, start a fresh read without the old cursor. Never supply or guess a provider session ID or transcript path.\n\nWait until every expected Dispatch settles, not for a fixed number of batches:\n\n```bash\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n# Process every message. For each accepted worker_done that is not immediately reused:\norca orchestration worker-release --dispatch --json\n# Acknowledge only after every message and required release decision is handled:\norca orchestration check --ack --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\nAfter processing each accepted `worker_done`, choose the terminal's next owner before you acknowledge the Delivery or wait again. If the same exact agent has an immediate follow-up Task, read the `worker.agent_terminal_handle` field of `worker-show --dispatch --json`, then run `orca orchestration worker-start --task --terminal --json` so Orca transfers cleanup ownership to the new Dispatch. Otherwise run `orca orchestration worker-release --dispatch --json`.\n\nRun `worker-release` after both succeeded and failed `worker_done` reports unless the user explicitly asked to keep that worker live. Release is post-completion cleanup, not cancellation: Orca first preserves inspectable output, then closes only the exact agent terminal owned by that settled Dispatch. Reused or pre-existing terminals, setup terminals, coordinators, active workers, user-taken-over terminals, and identities Orca cannot prove are retained. If the user explicitly asks to keep the live terminal for debugging, record that exception with `orca orchestration worker-retain --dispatch --json` instead of silently skipping cleanup. When the user is finished, the same Dispatch can be passed to `worker-release`, which clears the requested retention and releases the terminal.\n\nDo not release a worker because of a timeout, TUI idle state, heartbeat, status, question, escalation, or rejected/stale `worker_done`. If release returns `release_pending` or `release_unknown`, do not substitute `terminal close`; follow the exact recovery action in the receipt. A replayed Delivery may repeat `worker-release` safely.\n\nWorkers report exactly once using the IDs and capability injected by Orca; they do not supply Run/server/terminal identity:\n\n```bash\norca orchestration send --type worker_done --subject \"\" --body \"\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a,path/b\" --json\n# On failure, use --outcome failed; never encode failure only in prose.\n```\n\nA worker question defaults to its owning Run. Timeout leaves it pending:\n\n```bash\norca orchestration ask --question \"\" --options \"yes,no\" --timeout-ms 600000 --json\norca orchestration ask --resume --timeout-ms 600000 --json\n# Coordinator:\norca orchestration reply --id --body \"\" --json\n```\n\nRecovery is conditional, never a fixed destructive sequence:\n\n- The response was lost and named no Dispatch: run `orca orchestration request-show --request --json` first. It is read-only. `completed` means the mutation already took effect. `pending` means the original mutation is still running or Orca restarted before recording its outcome. For either state, replaying the original command with `--retry-request ` reuses the same operation identity so Orca can replay, join, or safely recover it without starting a separate duplicate. `absent` means this runtime holds no receipt under your caller identity and is not proof that nothing happened; inspect the affected state before deciding whether to retry.\n- `worker-show --dispatch ` says `ready`: keep waiting or read bounded output.\n- It proves `failed` or `stopped`: start a replacement with `worker-start --task --retry-of ` plus an explicit `--on`/`--worktree` and `--agent`/`--terminal` choice. Retry does not silently inherit placement.\n- It remains `outcome_unknown`: either `worker-stop --dispatch ` and inspect again, or explicitly `worker-abandon --dispatch ` while accepting that resources may still be live. Abandon performs no remote, process, or filesystem action.\n- `worker-stop` closes only the exact supervised agent terminal. It never deletes the worktree, setup terminal, configured tabs, or unrelated processes.\n\nLow-level `worktree create`, `terminal create`, and `dispatch --inject` remain valid recipes for custom argv or topology that `worker-start` does not express.\n\n`dispatch --inject` deliberately keeps an operator-started terminal unsupervised: it never creates a `worker_dispatches` row and `worker-stop`/`worker-abandon` never close that process. The dispatch context is still authoritative, so `worker-show`, `worker-read`, and `worker-list` report it as `unsupervised`; settled `worker-retain` and `worker-release` report `retained` with `no_owned_resource` and take no process action. Use `worker-start --terminal ` when supervision and worker lifecycle state are required.\n\n## Gates And Legacy Inspection\n\n```bash\norca orchestration gate-create --task --question [--options ] [--json]\norca orchestration gate-resolve --id --resolution [--json]\norca orchestration gate-list [--task ] [--status ] [--json]\n```\n\nUse `ask` for worker-to-coordinator questions; it creates a `question` message that the coordinator answers with `reply`. Use `gate-create` only for coordinator-managed task DAG decisions, not for answering a worker's `ask`.\n\n`coordinator-start`, `coordinator-stop`, `run`, and `run-stop` are retired scheduler commands. They perform no effects and return the current-skill recovery action. They are not aliases for lightweight Run creation or binding.\n\nRecovery only: `orca orchestration reset --tasks|--messages|--all --json` clears the selected local orchestration database state. Do not run it during active coordination unless explicitly abandoning that state.\n\n## Full Handoffs\n\nFor full ownership transfer, use non-lifecycle terminal/worktree commands and then stop monitoring unless the user asks for supervision.\n\nTreat these as full handoff requests by default: \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"send this to another agent\", \"another agent\", \"another worktree\", or \"launch another agent to own this.\" Custom model or reasoning effort words such as `gpt-5.5`, `high`, or `xhigh` do not make the handoff supervised.\n\nSupervised orchestration remains available only when the user explicitly asks for supervision or coordination: \"supervise\", \"monitor\", \"wait for worker_done\", \"wait for results\", \"track completion\", \"DAG\", \"decision gate\", \"ask/reply\", or \"coordinate workers.\"\n\nDo not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. `task-create` is also forbidden because it records coordinator-owned tracking state; if a task row is needed, the user asked for supervised orchestration. Do not create a `taskId`/`dispatchId`, inject a lifecycle preamble, wait for completion, or read the worker terminal after prompt delivery except to avoid losing the initial prompt.\n\nNew top-level worktree handoff:\n\n```bash\norca worktree create --name --no-parent --agent codex --prompt \"\" --setup run --json\n```\n\nBefore creating a new worktree from an active feature branch, decide and state whether the desired Orca lineage is child or top-level. Use child worktree lineage only when the new work is conceptually stacked under or dependent on the active worktree. For independent repo-wide fixes, standalone feature work, or unrelated follow-up tasks, create a top-level worktree with `--no-parent`.\n\nExisting terminal handoff:\n\n```bash\norca terminal send --terminal --text \"\" --enter --json\n```\n\nCustom Codex model/effort handoff:\n\n`orca worktree create --agent codex --prompt ...` launches the known Codex agent but does not accept Codex-specific `--model` or `-c model_reasoning_effort=...` arguments. When the user asks for a specific Codex model or effort, create the independent worktree first, launch Codex with the requested command in that worktree, wait only for TUI readiness if prompt delivery would otherwise race startup, send the prompt, and stop.\n\nThe two-step custom-argv path cannot enforce a repository's explicit `wait-for-setup` startup policy because the later `terminal create` is not the startup owned by `worktree create`. Use it only when the repository starts agents immediately. If the repository requires `wait-for-setup`, use an agent-first configured launcher that can preserve sequencing, or stop and ask rather than silently bypassing the policy.\n\nNote: when no repo default-terminal configuration supplies a primary terminal, bare create opens a fallback shell before `terminal create` adds the agent. Configured default tabs are materialized instead and may run real commands. Prefer `--agent` whenever custom argv is not required. With the two-step path, target only the agent handle; close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell.\n\nUse the exact full `::` worktree id returned by `orca worktree create --json`; a bare repo id cannot target the new worktree.\n\n```bash\norca worktree create --name --no-parent --setup run --json\norca terminal create --worktree id: --title --command 'codex --model gpt-5.5 -c model_reasoning_effort=\"xhigh\"' --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca terminal send --terminal --text \"\" --enter --json\n```\n\nWait only for `tui-idle` when needed to avoid losing the prompt. Do not monitor task completion.\n\n`--no-parent` only controls Orca lineage; it does not choose the Git base. If the work should start from the repo default base, omit `--base-branch` so Orca uses that default, or explicitly pass the repo default base (`origin/main`, `origin/master`, or the `orca repo show --repo --json` value); never base it on the current feature branch unless the user explicitly asks for stacked work or \"branch from current\". Put current-branch context in the prompt instead.\n\n## Worker Terminals\n\nChoose the worker location before creating a terminal. `Fresh worker` means a fresh agent session, not a new git worktree. For parallel work, create one fresh agent terminal per worker in the same required worktree, falling back to the active worktree when none is named. If the task says current worktree only, depends on uncommitted files/artifacts, or must validate/PR the current branch, keep every worker in the active worktree:\n\n```bash\norca terminal create --worktree active --title --command \"codex\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nReuse an idle agent in the required worktree only if the prompt allows reuse; otherwise create a fresh terminal there. Create a new worktree only when the user explicitly requests one or a concrete checkout or filesystem conflict makes sharing unsafe or impossible; if the user did not request it, state that conflict before running `worktree create`. Independent tasks, parallel execution, convenience, or a preference for separate checkouts are not isolation requirements.\n\nWhen a new worktree is allowed, use child lineage for isolated work that is stacked under or dependent on the active worktree, and use `--no-parent` when it is not stacked. Decide the Git base separately: `--no-parent` makes the worktree top-level in Orca, while omitted `--base-branch` uses the repo default base.\n\nFor every new worktree, pass `--setup run` so any configured repository setup hook runs. This does not mean waiting for setup before agent launch: preserve the repository's startup policy, whose default starts setup and the agent side by side. Use `--setup skip` or `--setup inherit` only when there is a concrete task-specific reason, and state that reason before creating the worktree. This rule does not rerun setup for current or existing worktrees.\n\n```bash\norca worktree create --name --agent codex --setup run --json\n# or: --agent claude | omp | pi | grok | ...\n# Read from agentTerminalHandle, falling back to startupTerminal.handle.\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nFor new-worktree workers, read the id and `agentTerminalHandle` from `worktree create`, falling back to `startupTerminal.handle` for older runtimes. Use that as the sole worker handle when present; otherwise use `terminal list` to resolve the agent handle. Omit `--repo` only inside an Orca-managed worktree; otherwise pass `--repo `.\n\n**For an allowed new worktree, use agent-first:** `--agent` reveals the new worktree and launches the selected agent **in its first terminal**, without adding a separate fallback shell for that worker. Pass `--setup run`; repo setup and default-terminal settings may add intentional tabs or splits. Do **not** run bare `worktree create` and then `terminal create --command ` for the same worker when agent-first create is available: without configured default tabs, that two-step path leaves a fallback shell + agent pair. Only use it when custom agent argv is required (for example Codex model/effort flags) or when an older CLI rejects `--agent`; if you must, message only the agent handle. Configured default tabs are intentional surfaces, so close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell. Do not run `worktree create` when the task must stay in the current worktree.\n\nUse `orca worktree create --prompt ...` or `orca terminal send ...` for full handoffs or untracked/lightweight prompts. Those paths do not attach `taskId`/`dispatchId`; the worker should not send lifecycle messages unless the prompt supplies a live orchestration preamble.\n\nSidebar lineage and orchestration lifecycle are related but not identical. A same-worktree worker may appear as a peer under that worktree in the sidebar while remaining a child dispatch in orchestration state; only an actual child worktree creates visible parent/child worktree lineage.\n\nOther terminal commands coordinators often need:\n\n```bash\norca terminal list [--worktree ] [--include-visual-layouts] [--json]\norca terminal create [--worktree ] [--title ] [--command ] [--json]\norca terminal split --terminal [--direction horizontal|vertical] [--command ] [--json]\norca terminal wait --terminal --for tui-idle --timeout-ms --json\norca terminal read --terminal --json\norca terminal send --terminal --text --enter --json\n```\n\nIf an older CLI rejects `worktree create --agent`, create the worktree normally, then run `orca terminal create --worktree --command \"codex\" --json` or `--command \"claude\"`.\n\nWait for `tui-idle` before dispatching. Always pass `--timeout-ms`; real coding tasks can take 15-60 minutes. During supervision, use rolling `check --wait` windows. If a window returns no matching message, inspect `task-list`, `terminal read`, or `terminal wait --for tui-idle` as a liveness checkpoint; if the terminal is still working or producing activity, keep waiting instead of retrying the task.\n\n## Agent Guidance\n\n- Workers with a valid live preamble must send `worker_done` exactly once from their own terminal with an explicit `--outcome succeeded` or `--outcome failed`:\n `orca orchestration send --type worker_done --subject \"\" --body \"<3-sentence summary: what you did, what you found, what's left>\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a\" --report-path \"\" --json`\n- A failed outcome is still a terminal report, but Orca records both the Dispatch and Task as failed. Never encode failure only in the subject/body.\n- After sending `worker_done`, end that dispatched turn and idle at the agent prompt. Do not autonomously start more work, poll, or attempt to close the terminal yourself. A direct user instruction takes precedence and starts ordinary user-owned work: follow it without coordinator approval or a fresh Dispatch, never refuse it because of worker/coordinator roles, and do not reuse the settled Dispatch's lifecycle IDs. A coordinator-supervised follow-up still arrives with a fresh preamble + TASK block.\n- For long tasks, send heartbeat/status only when the preamble asks for it, including both IDs:\n `orca orchestration send --type heartbeat --subject \"alive\" --payload '{\"taskId\":\"\",\"dispatchId\":\"\",\"phase\":\"implementing\"}' --json`\n- If blocked before completion, use `ask`; use `escalation` only when ownership is valid and the coordinator must intervene.\n- Treat preambles inherited through terminal history or full handoffs as stale unless the current prompt explicitly keeps that coordinator in the loop.\n- Coordinators must account for every settled worker terminal before waiting again or ending the turn: immediately reuse the exact worker for a new Dispatch, explicitly retain it at the user's request with `worker-retain`, or run `worker-release`. Do not leave a completed worker live merely to inspect output; released workers remain readable through `worker-read`.\n- Coordinators should use `task-list --ready` as external memory, dispatch parallel waves, and avoid dependency chains deeper than 3-4 steps.\n\n## Example\n\n```bash\norca terminal create --worktree active --title login-css-worker --command \"claude\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration task-create --spec \"Fix the login button CSS\" --json\norca orchestration dispatch --task --to --inject --json\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\n## Next Action\n\nCoordinator: confirm `orca status --json`, create or bind a Run, inspect `task-list`/`dispatch-show` if inheriting state, then use the explicit supervised loop (`task-create` -> `worker-start` -> `check --wait`). Use low-level terminal creation plus `dispatch --inject` only when the composed start does not express the needed topology. After every accepted `worker_done`, either transfer the exact terminal to an immediate follow-up Dispatch or run `worker-release` before the next wait.\n\nWorker: if the current prompt contains a live dispatch preamble, do the task, use `ask` for blocking questions, and send `worker_done` once with the required payload. If the preamble is stale or absent, do not send lifecycle messages; inspect state or treat the prompt as an ordinary handoff.\n" // Why: no current guide has bundled reference documents, so --full is byte-identical for now. // oxfmt-ignore diff --git a/src/cli/orchestration-dispatch-refusal-format.test.ts b/src/cli/orchestration-dispatch-refusal-format.test.ts new file mode 100644 index 00000000000..0a4461f25f1 --- /dev/null +++ b/src/cli/orchestration-dispatch-refusal-format.test.ts @@ -0,0 +1,77 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + injectRejectedRefusal, + taskNotFoundRefusal, + taskNotStartableRefusal, + type DispatchRefusalReceipt +} from '../shared/orchestration-dispatch-refusal-contract' +import { formatCliError, reportCliError } from './format' +import { RuntimeRpcFailureError, type RuntimeRpcFailure } from './runtime/types' + +afterEach(() => { + vi.restoreAllMocks() +}) + +// Why: these are the exact envelopes the RPC dispatcher test proved the runtime emits. This +// checkout's formatter never enumerates codes (verified below with a code no build has defined), +// which is what lets a client that predates a new code still print its message and nextSteps. +describe('orchestration dispatch refusals through the CLI error boundary', () => { + it.each([ + { + receipt: taskNotFoundRefusal('Task not found: task_missing', { taskId: 'task_missing' }), + recovery: /task-create|task-list/ + }, + { + receipt: taskNotStartableRefusal( + 'Task task_child is pending; only ready tasks can be dispatched', + { taskId: 'task_child', status: 'pending', unmetDependencies: ['task_parent'] } + ), + recovery: /task_parent/ + }, + { + receipt: injectRejectedRefusal('term_worker', 'no_agent_detected'), + recovery: /without --inject/ + } + ])('prints $receipt.code with its recovery in human and JSON output', ({ receipt, recovery }) => { + const failure = envelope(receipt) + const error = new RuntimeRpcFailureError(failure) + expect(error.code).toBe(receipt.code) + + const human = formatCliError(error, { commandPath: ['orchestration', 'dispatch'] }) + expect(human).toContain(receipt.message) + expect(human).toMatch(recovery) + + const log = vi.spyOn(console, 'log').mockImplementation(() => {}) + reportCliError(error, true, { commandPath: ['orchestration', 'dispatch'] }) + const printed = JSON.parse(log.mock.calls[0]?.[0] as string) as RuntimeRpcFailure + expect(printed.ok).toBe(false) + expect(printed.error).toEqual(receipt) + }) +}) + +// Why: a code this build has never defined stands in for a future host's new code; if the +// formatter ever starts gating on known codes, this is the assertion that catches it. +it('prints an unknown code with its message and nextSteps unchanged', () => { + const failure: RuntimeRpcFailure = { + id: 'rpc_1', + ok: false, + error: { + code: 'code_from_a_newer_host', + message: 'Refused for a reason this CLI has never heard of.', + data: { nextSteps: ['Do the thing the newer host suggested.'] } + }, + _meta: { runtimeId: 'runtime_1' } + } + const error = new RuntimeRpcFailureError(failure) + + expect(formatCliError(error, { commandPath: ['orchestration', 'dispatch'] })).toBe( + 'Refused for a reason this CLI has never heard of.\nNext step: Do the thing the newer host suggested.' + ) + const log = vi.spyOn(console, 'log').mockImplementation(() => {}) + reportCliError(error, true, { commandPath: ['orchestration', 'dispatch'] }) + expect(JSON.parse(log.mock.calls[0]?.[0] as string)).toEqual(failure) +}) + +function envelope(receipt: DispatchRefusalReceipt): RuntimeRpcFailure { + return { id: 'rpc_1', ok: false, error: receipt, _meta: { runtimeId: 'runtime_1' } } +} diff --git a/src/main/runtime/orchestration/db/dispatch-context/dispatch-context-store.ts b/src/main/runtime/orchestration/db/dispatch-context/dispatch-context-store.ts index cd4083acc0a..ee78396292d 100644 --- a/src/main/runtime/orchestration/db/dispatch-context/dispatch-context-store.ts +++ b/src/main/runtime/orchestration/db/dispatch-context/dispatch-context-store.ts @@ -7,6 +7,7 @@ import { paneKeyMatchSuffix } from '../pane-key-match' import { claimDispatchContextRow } from '../dispatch-row-writer' import type { DispatchCreator } from '../dispatch-depth' import type { OrchestrationDb } from '../orchestration-db' +import { taskNotFoundError, taskNotStartableError } from '../../task-dispatch-refusal' export function createDispatchContext( this: OrchestrationDb, @@ -26,10 +27,14 @@ export function createDispatchContext( const depth = this.resolveChildDispatchDepth(params.creator, params.maxDepth) const task = this.getTask(taskId) if (!task) { - throw new Error(`Task not found: ${taskId}`) + throw taskNotFoundError(`Task not found: ${taskId}`, { taskId }) } if (task.status !== 'ready') { - throw new Error(`Task ${taskId} is ${task.status}; only ready tasks can be dispatched`) + throw taskNotStartableError( + this, + `Task ${taskId} is ${task.status}; only ready tasks can be dispatched`, + task + ) } // Why: lock on pane identity too, so a reminted handle can't open a second concurrent dispatch on the same pane. @@ -72,9 +77,12 @@ export function createDispatchContext( `Terminal ${assigneeHandle} already has an active dispatch (${occupied.id} for task ${occupied.task_id})` ) } - throw new Error( - `Task ${taskId} is ${current?.status ?? 'missing'}; only ready tasks can be dispatched` - ) + // Why: the atomic claim lost to a concurrent status change; report it with the same + // typed receipt as the precheck so the loser can recover instead of reading runtime_error. + const message = `Task ${taskId} is ${current?.status ?? 'missing'}; only ready tasks can be dispatched` + throw current + ? taskNotStartableError(this, message, current) + : taskNotFoundError(message, { taskId }) } this.db.prepare("UPDATE tasks SET status = 'dispatched' WHERE id = ?").run(taskId) const dispatch = this.db diff --git a/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-start.ts b/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-start.ts index e26369e7fc1..e472ea1c7e8 100644 --- a/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-start.ts +++ b/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-start.ts @@ -6,6 +6,7 @@ import { generateId } from '../generated-id' import type { OrchestrationDb } from '../orchestration-db' import { insertStartingDispatchContextRow } from '../dispatch-row-writer' import type { DispatchCreator } from '../dispatch-depth' +import { taskNotFoundError, taskNotStartableError } from '../../task-dispatch-refusal' export function createStartingWorkerDispatch( this: OrchestrationDb, @@ -60,7 +61,7 @@ export function createStartingWorkerDispatch( } const task = this.getTask(params.taskId) if (!task) { - throw new OrchestrationError('task_not_found', `Task ${params.taskId} was not found.`) + throw taskNotFoundError(`Task ${params.taskId} was not found.`, { taskId: params.taskId }) } if (params.retryOf) { const prior = this.getDispatchContextById(params.retryOf) @@ -74,15 +75,18 @@ export function createStartingWorkerDispatch( !['failed', 'stopped', 'abandoned'].includes(priorWorker.state) || !['failed', 'blocked'].includes(task.status) ) { - throw new OrchestrationError( - 'task_not_startable', - `Task ${task.id} cannot retry from Dispatch ${params.retryOf}.` + throw taskNotStartableError( + this, + `Task ${task.id} cannot retry from Dispatch ${params.retryOf}.`, + task, + params.retryOf ) } } else if (task.status !== 'ready') { - throw new OrchestrationError( - 'task_not_startable', - `Task ${task.id} is ${task.status}; only a ready Task can start.` + throw taskNotStartableError( + this, + `Task ${task.id} is ${task.status}; only a ready Task can start.`, + task ) } diff --git a/src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts b/src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts index 9e4a2768a1d..156d1427f01 100644 --- a/src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts +++ b/src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts @@ -168,7 +168,12 @@ describe('OrchestrationDb worker Dispatch state', () => { payloadHash: 'payload_hash' } }) - ).toThrow('was not found') + ).toThrowError( + expect.objectContaining({ + code: 'task_not_found', + message: 'Task task_missing was not found.' + }) + ) expect(d.getMutationReceipt('caller_fingerprint', 'invalid_worker_start')).toBeUndefined() }) diff --git a/src/main/runtime/orchestration/task-dispatch-refusal.ts b/src/main/runtime/orchestration/task-dispatch-refusal.ts new file mode 100644 index 00000000000..f3e026d0f73 --- /dev/null +++ b/src/main/runtime/orchestration/task-dispatch-refusal.ts @@ -0,0 +1,61 @@ +import type { OrchestrationDb } from './db' +import { OrchestrationError } from './orchestration-error' +import type { TaskRow } from './types' +import { + injectRejectedRefusal, + taskNotFoundRefusal, + taskNotStartableRefusal, + type DispatchRefusalReceipt, + type InjectRejectionReason +} from '../../../shared/orchestration-dispatch-refusal-contract' + +// Why: each site keeps the exact message it published before; only the code and data are shared. + +export function taskNotFoundError( + message: string, + detail: { taskId: string; runId?: string } +): OrchestrationError { + return toError(taskNotFoundRefusal(message, detail)) +} + +export function taskNotStartableError( + db: OrchestrationDb, + message: string, + task: TaskRow, + retryOf?: string +): OrchestrationError { + return toError( + taskNotStartableRefusal(message, { + taskId: task.id, + status: task.status, + unmetDependencies: unmetTaskDependencies(db, task), + ...(retryOf ? { retryOf } : {}) + }) + ) +} + +export function injectRejectedError( + terminal: string, + reason: InjectRejectionReason +): OrchestrationError { + return toError(injectRejectedRefusal(terminal, reason)) +} + +function toError(receipt: DispatchRefusalReceipt): OrchestrationError { + return new OrchestrationError(receipt.code, receipt.message, receipt.data) +} + +function unmetTaskDependencies(db: OrchestrationDb, task: TaskRow): string[] { + let deps: unknown + try { + deps = JSON.parse(task.deps) + } catch { + return [] + } + if (!Array.isArray(deps)) { + return [] + } + return deps.filter( + (dep): dep is string => typeof dep === 'string' && db.getTask(dep)?.status !== 'completed' + ) +} diff --git a/src/main/runtime/rpc/errors.ts b/src/main/runtime/rpc/errors.ts index f4b4768865b..1e4567f7f6f 100644 --- a/src/main/runtime/rpc/errors.ts +++ b/src/main/runtime/rpc/errors.ts @@ -85,6 +85,7 @@ const STRUCTURED_RUNTIME_PASSTHROUGH_CODES: ReadonlySet = new Set([ 'consumer_fenced', 'task_not_found', 'task_not_startable', + 'inject_rejected', 'dispatch_not_found', 'dispatch_run_mismatch', 'terminal_not_found', diff --git a/src/main/runtime/rpc/methods/orchestration-dispatch-error-codes.test.ts b/src/main/runtime/rpc/methods/orchestration-dispatch-error-codes.test.ts new file mode 100644 index 00000000000..918724e6aaf --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration-dispatch-error-codes.test.ts @@ -0,0 +1,242 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { ORCHESTRATION_CONTRACT_VERSION } from '../../../../shared/protocol-version' +import { + buildInjectRejectionMessage, + injectRejectedRefusal, + taskNotFoundRefusal, + taskNotStartableRefusal +} from '../../../../shared/orchestration-dispatch-refusal-contract' +import { OrcaRuntimeService } from '../../orca-runtime' +import { OrchestrationDb } from '../../orchestration/db' +import type { RpcFailure, RpcRequest, RpcResponse } from '../core' +import { RpcDispatcher } from '../dispatcher' +import { ORCHESTRATION_METHODS } from './orchestration' + +const COORDINATOR_HANDLE = 'term_codes_coordinator' +const COORDINATOR_PANE = 'tab_coord:cccccccc-cccc-4ccc-8ccc-cccccccccccc' +const WORKER_HANDLE = 'term_codes_worker' +const WORKER_PANE = 'tab_worker:dddddddd-dddd-4ddd-8ddd-dddddddddddd' + +type Harness = { db: OrchestrationDb; runtime: OrcaRuntimeService; dispatcher: RpcDispatcher } + +const harnesses: Harness[] = [] +let requestSequence = 0 + +afterEach(() => { + for (const harness of harnesses.splice(0)) { + harness.db.close() + } + vi.restoreAllMocks() +}) + +// Why: an agent reads the receipt code to pick a recovery; every case is driven from the real +// RPC dispatcher and checked against the shared contract the CLI-side test formats. +describe('orchestration dispatch failure codes through RpcDispatcher', () => { + it('reports task_not_found for a task id that does not exist', async () => { + const harness = createHarness() + + const response = await dispatch(harness, { task: 'task_missing', to: WORKER_HANDLE }) + + expect(expectFailure(response).error).toEqual( + taskNotFoundRefusal('Task not found: task_missing', { taskId: 'task_missing' }) + ) + }) + + it('reports task_not_startable with the unmet dependencies for a pending task', async () => { + const harness = createHarness() + const parent = harness.db.createTask({ spec: 'parent' }) + const child = harness.db.createTask({ spec: 'child', deps: [parent.id] }) + + const response = await dispatch(harness, { task: child.id, to: WORKER_HANDLE }) + + expect(expectFailure(response).error).toEqual( + taskNotStartableRefusal(`Task ${child.id} is pending; only ready tasks can be dispatched`, { + taskId: child.id, + status: 'pending', + unmetDependencies: [parent.id] + }) + ) + expect(harness.db.getTask(child.id)?.status).toBe('pending') + }) + + it('reports task_not_startable with the status for a completed task', async () => { + const harness = createHarness() + const task = harness.db.createTask({ spec: 'done' }) + harness.db.updateTaskStatus(task.id, 'completed') + + const response = await dispatch(harness, { task: task.id, to: WORKER_HANDLE }) + + expect(expectFailure(response).error).toEqual( + taskNotStartableRefusal(`Task ${task.id} is completed; only ready tasks can be dispatched`, { + taskId: task.id, + status: 'completed', + unmetDependencies: [] + }) + ) + }) + + it('reports inject_rejected when the target terminal runs no recognized agent', async () => { + const harness = createHarness() + const task = harness.db.createTask({ spec: 'work' }) + vi.spyOn(harness.runtime, 'isTerminalRunningAgent').mockResolvedValue(false) + + const response = await dispatch(harness, { task: task.id, to: WORKER_HANDLE, inject: true }) + + expect(expectFailure(response).error).toEqual( + injectRejectedRefusal(WORKER_HANDLE, 'no_agent_detected') + ) + expect(expectFailure(response).error.message).toBe(buildInjectRejectionMessage(WORKER_HANDLE)) + expect(harness.db.getTask(task.id)?.status).toBe('ready') + expect(harness.db.getDispatchContext(task.id)).toBeUndefined() + }) + + it('reports task_not_startable with dependency detail from worker-start', async () => { + const harness = createHarness() + const parent = harness.db.createTask({ spec: 'parent' }) + const child = harness.db.createTask({ spec: 'child', deps: [parent.id] }) + mockWorkerStartTopology(harness.runtime) + + const response = await harness.dispatcher.dispatch( + request('orchestration.workerStart', { + task: child.id, + from: COORDINATOR_HANDLE, + agent: 'claude' + }) + ) + + expect(expectFailure(response).error).toEqual( + taskNotStartableRefusal(`Task ${child.id} is pending; only a ready Task can start.`, { + taskId: child.id, + status: 'pending', + unmetDependencies: [parent.id] + }) + ) + expect(harness.db.getTask(child.id)?.status).toBe('pending') + }) + + it('reports task_not_startable with retry detail for an invalid --retry-of', async () => { + const harness = createHarness() + const task = harness.db.createTask({ spec: 'work' }) + mockWorkerStartTopology(harness.runtime) + + const response = await harness.dispatcher.dispatch( + request('orchestration.workerStart', { + task: task.id, + from: COORDINATOR_HANDLE, + agent: 'claude', + retryOf: 'ctx_missing' + }) + ) + + expect(expectFailure(response).error).toEqual( + taskNotStartableRefusal(`Task ${task.id} cannot retry from Dispatch ctx_missing.`, { + taskId: task.id, + status: 'ready', + unmetDependencies: [], + retryOf: 'ctx_missing' + }) + ) + }) + + it('types the atomic claim loser when the task changes after the ready precheck', async () => { + const harness = createHarness() + const task = harness.db.createTask({ spec: 'raced' }) + // Why: the pane lookup runs after the ready precheck and before the DB claim, so failing the + // task there is the interleaving a concurrent status change produces. The loser's DB refusal + // must carry the same typed receipt instead of the bare Error it used to throw. + vi.mocked(harness.runtime.getTerminalPaneKey).mockImplementation((handle) => { + if (handle === WORKER_HANDLE) { + harness.db.updateTaskStatus(task.id, 'failed', 'raced out') + return WORKER_PANE + } + return handle === COORDINATOR_HANDLE ? COORDINATOR_PANE : null + }) + + const response = await dispatch(harness, { task: task.id, to: WORKER_HANDLE }) + + expect(expectFailure(response).error).toEqual( + taskNotStartableRefusal(`Task ${task.id} is failed; only ready tasks can be dispatched`, { + taskId: task.id, + status: 'failed', + unmetDependencies: [] + }) + ) + expect(harness.db.getDispatchContext(task.id)).toBeUndefined() + }) + + it('keeps runtime_error for a genuinely unexpected dispatch failure', async () => { + const harness = createHarness() + const task = harness.db.createTask({ spec: 'work' }) + vi.spyOn(harness.runtime, 'isTerminalRunningAgent').mockRejectedValue( + new Error('probe exploded') + ) + + const response = await dispatch(harness, { task: task.id, to: WORKER_HANDLE, inject: true }) + + expect(expectFailure(response).error).toMatchObject({ + code: 'runtime_error', + message: 'probe exploded' + }) + }) +}) + +function expectFailure(response: RpcResponse): RpcFailure { + if (response.ok) { + throw new Error(`Expected a failure, got ${JSON.stringify(response.result)}`) + } + return response +} + +function createHarness(): Harness { + const db = new OrchestrationDb(':memory:') + const runtime = new OrcaRuntimeService() + runtime.setOrchestrationDb(db) + vi.spyOn(runtime, 'getTerminalPaneKey').mockImplementation((handle) => + handle === COORDINATOR_HANDLE ? COORDINATOR_PANE : handle === WORKER_HANDLE ? WORKER_PANE : null + ) + vi.spyOn(runtime, 'getTerminalProcessIncarnation').mockImplementation((handle) => + handle === WORKER_HANDLE ? 'pty-worker:incarnation-1' : null + ) + const runId = db.createRun({ + objective: 'Typed dispatch failures', + coordinatorHandle: COORDINATOR_HANDLE, + coordinatorPaneKey: COORDINATOR_PANE + }).id + const createTask = db.createTask.bind(db) + db.createTask = (task) => createTask({ ...task, runId: task.runId ?? runId }) + const harness = { + db, + runtime, + dispatcher: new RpcDispatcher({ runtime, methods: ORCHESTRATION_METHODS }) + } + harnesses.push(harness) + return harness +} + +function mockWorkerStartTopology(runtime: OrcaRuntimeService): void { + vi.spyOn(runtime, 'validateOrchestrationAgentLauncher').mockImplementation(() => {}) + vi.spyOn(runtime, 'showTerminal').mockImplementation( + async (handle) => ({ handle, worktreeId: 'repo::worktree', status: 'running' }) as never + ) + vi.spyOn(runtime, 'showManagedTerminalWorkspace').mockResolvedValue({ + id: 'repo::worktree' + } as never) +} + +function dispatch(harness: Harness, params: Record): Promise { + return harness.dispatcher.dispatch( + request('orchestration.dispatch', { from: COORDINATOR_HANDLE, ...params }) + ) +} + +function request(method: string, params: Record): RpcRequest { + requestSequence += 1 + return { + id: `rpc_dispatch_error_codes_${requestSequence}`, + authToken: 'test-token', + method, + params, + orchestrationContractVersion: ORCHESTRATION_CONTRACT_VERSION, + orchestrationRequestId: `dispatch_error_codes_${requestSequence}` + } +} diff --git a/src/main/runtime/rpc/methods/orchestration-dispatch-methods.ts b/src/main/runtime/rpc/methods/orchestration-dispatch-methods.ts index ae21a4e5a46..d573d944b2d 100644 --- a/src/main/runtime/rpc/methods/orchestration-dispatch-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration-dispatch-methods.ts @@ -2,7 +2,11 @@ import { defineMethod, type RpcMethod } from '../core' import { OrchestrationError } from '../../orchestration/orchestration-error' import { buildDispatchPreamble } from '../../orchestration/preamble' import { resolveDispatchCreator } from './orchestration-dispatch-creator' -import { buildInjectRejectionMessage } from './orchestration-inject-rejection-message' +import { + injectRejectedError, + taskNotFoundError, + taskNotStartableError +} from '../../orchestration/task-dispatch-refusal' import { resolveRunScope } from './orchestration-run-scope' import { DispatchParams, DispatchShowParams } from './orchestration-schemas' @@ -22,7 +26,7 @@ export const ORCHESTRATION_DISPATCH_METHODS: RpcMethod[] = [ const db = runtime.getOrchestrationDb() const task = db.getTask(params.task) if (!task) { - throw new Error(`Task not found: ${params.task}`) + throw taskNotFoundError(`Task not found: ${params.task}`, { taskId: params.task }) } const run = resolveRunScope(runtime, { runId: params.run, @@ -32,10 +36,10 @@ export const ORCHESTRATION_DISPATCH_METHODS: RpcMethod[] = [ callerEvidence: orchestrationCompatibilityEvidence }) if (task.run_id !== run.id) { - throw new OrchestrationError( - 'task_not_found', - `Task ${task.id} was not found in Run ${run.id}.` - ) + throw taskNotFoundError(`Task ${task.id} was not found in Run ${run.id}.`, { + taskId: task.id, + runId: run.id + }) } // Why: dry-run previews the preamble without mutating state, so it skips the ready-status check and uses a placeholder dispatchId. @@ -66,14 +70,18 @@ export const ORCHESTRATION_DISPATCH_METHODS: RpcMethod[] = [ const to = params.to if (task.status !== 'ready') { - throw new Error(`Task ${params.task} is ${task.status}; only ready tasks can be dispatched`) + throw taskNotStartableError( + db, + `Task ${params.task} is ${task.status}; only ready tasks can be dispatched`, + task + ) } // Why: injecting the preamble into a bare shell dumps it as shell commands (gibberish), so require a detected agent first. if (params.inject) { const hasAgent = await runtime.isTerminalRunningAgent(to) if (!hasAgent) { - throw new Error(buildInjectRejectionMessage(to)) + throw injectRejectedError(to, 'no_agent_detected') } } diff --git a/src/main/runtime/rpc/methods/orchestration-inject-rejection-message.test.ts b/src/main/runtime/rpc/methods/orchestration-inject-rejection-message.test.ts deleted file mode 100644 index a7ae03b00a8..00000000000 --- a/src/main/runtime/rpc/methods/orchestration-inject-rejection-message.test.ts +++ /dev/null @@ -1,31 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { buildInjectRejectionMessage } from './orchestration-inject-rejection-message' -import { TUI_AGENT_CONFIG } from '../../../../shared/tui-agent-config' -import { recognizeAgentProcess } from '../../../../shared/agent-process-recognition' - -describe('buildInjectRejectionMessage', () => { - const message = buildInjectRejectionMessage('term_a') - - it('keeps the substring callers and scripts match on', () => { - expect(message).toContain('Cannot dispatch --inject to terminal term_a') - expect(message).toContain('no recognized agent detected') - }) - - it('names every agent Orca recognizes, including agy', () => { - expect(message).toMatch(/\bagy\b/) - for (const config of Object.values(TUI_AGENT_CONFIG)) { - expect(message).toContain(config.expectedProcess) - } - }) - - it('lists only names detection actually resolves, deduped and sorted', () => { - const listed = (/\(([^)]+)\)/.exec(message)?.[1] ?? '').split(', ') - - expect(listed.length).toBeGreaterThan(0) - expect(new Set(listed).size).toBe(listed.length) - expect([...listed].sort()).toEqual(listed) - for (const name of listed) { - expect(recognizeAgentProcess(name)).not.toBeNull() - } - }) -}) diff --git a/src/main/runtime/rpc/methods/orchestration-inject-rejection-message.ts b/src/main/runtime/rpc/methods/orchestration-inject-rejection-message.ts deleted file mode 100644 index 33d622ca80e..00000000000 --- a/src/main/runtime/rpc/methods/orchestration-inject-rejection-message.ts +++ /dev/null @@ -1,16 +0,0 @@ -import { TUI_AGENT_CONFIG } from '../../../../shared/tui-agent-config' - -// Why: the old five-name example read as an allowlist (#15125); derive from the field detection keys on so it cannot drift. -// Not filtered by `disabledTuiAgents` — that gates Orca's launchers, not detection, so a hand-started disabled agent still injects. -const RECOGNIZED_AGENT_PROCESS_NAMES = [ - ...new Set(Object.values(TUI_AGENT_CONFIG).map((config) => config.expectedProcess)) -].sort() - -export function buildInjectRejectionMessage(terminal: string): string { - return ( - `Cannot dispatch --inject to terminal ${terminal}: no recognized agent detected. ` + - `Orca detects these agent CLIs (${RECOGNIZED_AGENT_PROCESS_NAMES.join(', ')}). ` + - 'Start one in the terminal and let it finish launching, ' + - 'or dispatch without --inject and send the prompt manually.' - ) -} diff --git a/src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts b/src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts index a216b4c7a4d..cd6d79c9fd5 100644 --- a/src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts +++ b/src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts @@ -3,7 +3,7 @@ import type { RpcContext } from '../core' import { createOrchestrationRpcHarness } from './orchestration-rpc-test-harness' import type { OrchestrationDb } from '../../orchestration/db' import type { OrcaRuntimeService } from '../../orca-runtime' -import { buildInjectRejectionMessage } from './orchestration-inject-rejection-message' +import { buildInjectRejectionMessage } from '../../../../shared/orchestration-dispatch-refusal-contract' import { createRootDispatch } from '../../orchestration/db/root-dispatch-test-fixture' describe('orchestration RPC methods', () => { diff --git a/src/main/runtime/rpc/methods/orchestration-workers.ts b/src/main/runtime/rpc/methods/orchestration-workers.ts index 61271525939..632b34cc1b7 100644 --- a/src/main/runtime/rpc/methods/orchestration-workers.ts +++ b/src/main/runtime/rpc/methods/orchestration-workers.ts @@ -21,6 +21,7 @@ import { import { failWorkerStartWithReceipt } from './orchestration-worker-start-receipt' import { prepareLocalWorkerStart } from './orchestration-worker-start-validation' import { resolveDispatchCreator } from './orchestration-dispatch-creator' +import { taskNotFoundError } from '../../orchestration/task-dispatch-refusal' import { resolveOrchestrationCaller } from './orchestration-run-scope' import { isWorkerStartTimeoutWithinTimerLimit, @@ -58,10 +59,10 @@ export const ORCHESTRATION_WORKER_START_METHODS: RpcMethod[] = [ } const task = db.getTask(params.task) if (!task || task.run_id !== run.id) { - throw new OrchestrationError( - 'task_not_found', - `Task ${params.task} was not found in Run ${run.id}.` - ) + throw taskNotFoundError(`Task ${params.task} was not found in Run ${run.id}.`, { + taskId: params.task, + runId: run.id + }) } if (params.on) { diff --git a/src/main/ssh/ssh-remote-cli-dispatch-refusal-passthrough.test.ts b/src/main/ssh/ssh-remote-cli-dispatch-refusal-passthrough.test.ts new file mode 100644 index 00000000000..cd0503348d0 --- /dev/null +++ b/src/main/ssh/ssh-remote-cli-dispatch-refusal-passthrough.test.ts @@ -0,0 +1,64 @@ +import { EventEmitter } from 'node:events' +import { expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ + app: { + isPackaged: false, + getAppPath: () => '/host/app' + } +})) +vi.mock('../persistence', () => ({ + getCanonicalUserDataPath: () => '/host/user-data' +})) + +import { OrcaRuntimeService } from '../runtime/orca-runtime' +import { runRemoteOrcaCli } from './ssh-remote-orca-cli' + +// Why: the SSH bridge captures the host CLI child's stdout and exit code without reparsing; this +// pins that a typed refusal envelope and its nonzero exit reach the remote agent unchanged. +it('relays typed dispatch refusal codes from the host CLI unchanged', async () => { + const child = new EventEmitter() as EventEmitter & { + stdout: EventEmitter + stderr: EventEmitter + stdin: { end: ReturnType; on: ReturnType } + kill: ReturnType + } + child.stdout = new EventEmitter() + child.stderr = new EventEmitter() + child.stdin = { end: vi.fn(), on: vi.fn() } + child.kill = vi.fn() + const spawn = vi.fn(() => child) + const refusal = { + id: 'rpc_1', + ok: false, + error: { + code: 'task_not_startable', + message: 'Task task_1 is pending; only ready tasks can be dispatched', + data: { taskId: 'task_1', status: 'pending', unmetDependencies: ['task_0'] } + }, + _meta: { runtimeId: 'runtime_1' } + } + + const resultPromise = runRemoteOrcaCli( + new OrcaRuntimeService(), + { + argv: ['orchestration', 'dispatch', '--task', 'task_1', '--to', 'term_w', '--json'], + cwd: '/home/alice/repo', + env: { ORCA_TERMINAL_HANDLE: 'term_ssh' } + }, + { + execPath: '/host/electron', + cliEntryPath: '/host/app/out/cli/index.js', + userDataPath: '/host/user-data', + entryExists: () => true, + spawn: spawn as never + } + ) + + const stdout = `${JSON.stringify(refusal, null, 2)}\n` + await Promise.resolve() + child.stdout.emit('data', Buffer.from(stdout)) + child.emit('close', 1) + + expect(await resultPromise).toEqual({ stdout, stderr: '', exitCode: 1 }) +}) diff --git a/src/shared/orchestration-dispatch-refusal-contract.test.ts b/src/shared/orchestration-dispatch-refusal-contract.test.ts new file mode 100644 index 00000000000..4f9fa4bf617 --- /dev/null +++ b/src/shared/orchestration-dispatch-refusal-contract.test.ts @@ -0,0 +1,67 @@ +import { describe, expect, it } from 'vitest' +import { + buildInjectRejectionMessage, + taskNotFoundRefusal, + taskNotStartableRefusal +} from './orchestration-dispatch-refusal-contract' +import { TUI_AGENT_CONFIG } from './tui-agent-config' +import { recognizeAgentProcess } from './agent-process-recognition' + +describe('buildInjectRejectionMessage', () => { + const message = buildInjectRejectionMessage('term_a') + + it('keeps the substring callers and scripts match on', () => { + expect(message).toContain('Cannot dispatch --inject to terminal term_a') + expect(message).toContain('no recognized agent detected') + }) + + it('names every agent Orca recognizes, including agy', () => { + expect(message).toMatch(/\bagy\b/) + for (const config of Object.values(TUI_AGENT_CONFIG)) { + expect(message).toContain(config.expectedProcess) + } + }) + + it('lists only names detection actually resolves, deduped and sorted', () => { + const listed = (/\(([^)]+)\)/.exec(message)?.[1] ?? '').split(', ') + + expect(listed.length).toBeGreaterThan(0) + expect(new Set(listed).size).toBe(listed.length) + expect([...listed].sort()).toEqual(listed) + for (const name of listed) { + expect(recognizeAgentProcess(name)).not.toBeNull() + } + }) +}) + +// Why: these strings are published receipts; they are pinned as literals, independent of the +// builders, so a refactor cannot silently rewrite them together with the expectation. +describe('dispatch refusal receipts keep their published messages', () => { + it('leaves the message exactly as each call site supplies it', () => { + expect(taskNotFoundRefusal('Task not found: task_1', { taskId: 'task_1' }).message).toBe( + 'Task not found: task_1' + ) + expect( + taskNotStartableRefusal('Task task_1 is pending; only a ready Task can start.', { + taskId: 'task_1', + status: 'pending', + unmetDependencies: [] + }).message + ).toBe('Task task_1 is pending; only a ready Task can start.') + }) + + it('tailors nextSteps to retry, dependency, occupancy, and terminal-status refusals', () => { + const base = { taskId: 'task_1', status: 'failed', unmetDependencies: [] } + expect(taskNotStartableRefusal('m', { ...base, retryOf: 'ctx_1' }).data.nextSteps[0]).toMatch( + /--retry-of.*ctx_1/ + ) + expect( + taskNotStartableRefusal('m', { ...base, status: 'pending', unmetDependencies: ['task_0'] }) + .data.nextSteps[0] + ).toMatch(/task_0.*unblock failed/) + expect( + taskNotStartableRefusal('m', { ...base, status: 'dispatched' }).data.nextSteps[0] + ).toMatch(/dispatch-show --task task_1/) + expect(taskNotStartableRefusal('m', base).data.nextSteps[0]).toMatch(/failed Task cannot/) + }) +}) diff --git a/src/shared/orchestration-dispatch-refusal-contract.ts b/src/shared/orchestration-dispatch-refusal-contract.ts new file mode 100644 index 00000000000..b764067402b --- /dev/null +++ b/src/shared/orchestration-dispatch-refusal-contract.ts @@ -0,0 +1,102 @@ +import { TUI_AGENT_CONFIG } from './tui-agent-config' + +// Why: one source for each dispatch refusal's code, message, and data, so the runtime emits and +// the CLI test formats the identical envelope. Messages are supplied per call site because each +// existing string is a published receipt an old consumer may match on. + +export type DispatchRefusalReceipt = { + code: 'task_not_found' | 'task_not_startable' | 'inject_rejected' + message: string + data: Record & { nextSteps: string[] } +} + +export function taskNotFoundRefusal( + message: string, + detail: { taskId: string; runId?: string } +): DispatchRefusalReceipt { + return { + code: 'task_not_found', + message, + data: { + ...detail, + nextSteps: [ + 'Run orca orchestration task-list --json in the bound Run to find the intended Task id.', + 'If the Task does not exist yet, create it with orca orchestration task-create --spec --json.' + ] + } + } +} + +export type TaskNotStartableDetail = { + taskId: string + status: string + unmetDependencies: string[] + retryOf?: string +} + +export function taskNotStartableRefusal( + message: string, + detail: TaskNotStartableDetail +): DispatchRefusalReceipt { + return { + code: 'task_not_startable', + message, + data: { ...detail, nextSteps: taskNotStartableNextSteps(detail) } + } +} + +function taskNotStartableNextSteps(detail: TaskNotStartableDetail): string[] { + if (detail.retryOf) { + return [ + `--retry-of must name the latest settled Dispatch of a failed or blocked Task; check orca orchestration dispatch-show --task ${detail.taskId} --json and orca orchestration worker-show --dispatch ${detail.retryOf} --json.` + ] + } + if (detail.unmetDependencies.length > 0) { + return [ + `Dependencies ${detail.unmetDependencies.join(', ')} are not completed. Wait for running ones with orca orchestration check --wait --json; retry or unblock failed ones before dispatching again.` + ] + } + if (detail.status === 'dispatched') { + return [ + `The Task already has an active Dispatch; inspect it with orca orchestration dispatch-show --task ${detail.taskId} --json.` + ] + } + return [ + `A ${detail.status} Task cannot be dispatched; create a new Task or use worker-start --retry-of for a failed attempt.` + ] +} + +// Why: the old five-name example read as an allowlist (#15125); derive from the field detection keys on so it cannot drift. +// Not filtered by `disabledTuiAgents` — that gates Orca's launchers, not detection, so a hand-started disabled agent still injects. +const RECOGNIZED_AGENT_PROCESS_NAMES = [ + ...new Set(Object.values(TUI_AGENT_CONFIG).map((config) => config.expectedProcess)) +].sort() + +export function buildInjectRejectionMessage(terminal: string): string { + return ( + `Cannot dispatch --inject to terminal ${terminal}: no recognized agent detected. ` + + `Orca detects these agent CLIs (${RECOGNIZED_AGENT_PROCESS_NAMES.join(', ')}). ` + + 'Start one in the terminal and let it finish launching, ' + + 'or dispatch without --inject and send the prompt manually.' + ) +} + +export type InjectRejectionReason = 'no_agent_detected' + +export function injectRejectedRefusal( + terminal: string, + reason: InjectRejectionReason +): DispatchRefusalReceipt { + return { + code: 'inject_rejected', + message: buildInjectRejectionMessage(terminal), + data: { + terminal, + reason, + nextSteps: [ + 'Start a recognized agent CLI in that terminal and wait for it to finish launching, or pick a terminal that already runs one.', + 'Alternatively dispatch without --inject and deliver the prompt with orca terminal send --terminal --text --enter --json.' + ] + } + } +} From 9f0054d89c3e05f9cebc964604a33b214438946d Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 17:35:20 -0700 Subject: [PATCH 14/17] ci: skip idle Mac allocations and redundant native compiler setup (#18954) * ci: avoid idle Mac allocations and cached native toolchain installs * test: anchor artifact fixtures before their fixed expiry --- .../install-node-dependencies/action.yml | 24 +++-- .github/workflows/hourly-mac-build.yml | 90 +++++++++--------- config/scripts/ci-native-toolchain.test.mjs | 69 ++++++++++++++ .../hourly-preflight-workflow.test.mjs | 91 +++++++++++++++++++ docs/reference/ci-runner-efficiency.md | 38 +++++++- .../artifacts/artifact-cloud-recovery.test.ts | 9 +- .../artifact-cloud-service-races.test.ts | 9 +- .../artifacts/artifact-cloud-service.test.ts | 8 +- 8 files changed, 284 insertions(+), 54 deletions(-) create mode 100644 config/scripts/ci-native-toolchain.test.mjs create mode 100644 config/scripts/hourly-preflight-workflow.test.mjs diff --git a/.github/actions/install-node-dependencies/action.yml b/.github/actions/install-node-dependencies/action.yml index e36ec4c65d8..7695d2bec9b 100644 --- a/.github/actions/install-node-dependencies/action.yml +++ b/.github/actions/install-node-dependencies/action.yml @@ -77,14 +77,6 @@ runs: ;; esac - # pnpm's bundled gyp_main.py is not executable on fresh Linux runners. - - name: Use external node-gyp - if: runner.os == 'Linux' && inputs.native-runtime != 'none' - shell: bash - run: | - npm install -g node-gyp@11.5.0 - echo "npm_config_node_gyp=$(npm root -g)/node-gyp/bin/node-gyp.js" >> "$GITHUB_ENV" - - name: Prepare dependency install shell: bash run: | @@ -175,6 +167,22 @@ runs: node_modules/.pnpm/@vscode+windows-process-tree@*/node_modules/@vscode/windows-process-tree/build key: native-modules-${{ runner.os }}-${{ steps.native-cache-scope.outputs.scope }}-${{ runner.arch }}-${{ inputs.native-runtime }}-node${{ steps.requested-node.outputs.node-version || steps.default-node.outputs.node-version }}-${{ hashFiles('pnpm-lock.yaml', '.github/actions/install-node-dependencies/action.yml', 'config/scripts/ensure-native-runtime.mjs', 'config/scripts/rebuild-native-deps.mjs', 'config/patches/node-pty@1.1.0.patch', 'config/patches/@vscode__windows-process-tree@0.8.0.patch') }} + # pnpm's bundled gyp_main.py is not executable on fresh Linux runners. + - name: Use external node-gyp + if: runner.os == 'Linux' && inputs.native-runtime != 'none' + shell: bash + env: + NATIVE_RUNTIME: ${{ inputs.native-runtime }} + NATIVE_CACHE_HIT: ${{ steps.native-cache-restore.outputs.cache-hit || steps.native-cache-restore-only.outputs.cache-hit }} + run: | + # A cache hit can contain unusable addons; probe before skipping the rebuild toolchain. + if [ "$NATIVE_RUNTIME" = node ] && [ "$NATIVE_CACHE_HIT" = true ] && + node config/scripts/ensure-native-runtime.mjs --check-only; then + exit 0 + fi + npm install -g node-gyp@11.5.0 + echo "npm_config_node_gyp=$(npm root -g)/node-gyp/bin/node-gyp.js" >> "$GITHUB_ENV" + - name: Prepare native runtime if: inputs.native-runtime != 'none' shell: bash diff --git a/.github/workflows/hourly-mac-build.yml b/.github/workflows/hourly-mac-build.yml index ac3af92a3bc..c300b2543b8 100644 --- a/.github/workflows/hourly-mac-build.yml +++ b/.github/workflows/hourly-mac-build.yml @@ -26,7 +26,7 @@ name: Hourly macOS Dev Build # HOURLY_RELEASE_APP_ID the App's numeric id # HOURLY_RELEASE_APP_PRIVATE_KEY the App's .pem private key # -# Installation tokens live one hour, which is why this mints twice. Install and +# Installation tokens live one hour, so the build job mints twice. Install and # build need no token at all, and notarization can hold the publish step for tens # of minutes; minting again once the build is done starts the clock at the first # call that actually uses it rather than burning a third of it on `pnpm install`. @@ -60,33 +60,15 @@ env: HOURLY_RETAIN_COUNT: 72 jobs: - build-hourly-mac: + # Avoid occupying the limited Mac pool when main has not moved. + preflight: if: github.repository == 'stablyai/orca' + runs-on: ubuntu-latest + timeout-minutes: 5 outputs: - tag: ${{ steps.release.outputs.tag }} - version: ${{ steps.hourly.outputs.version }} + should_build: ${{ steps.freshness.outputs.should_build }} head_sha: ${{ steps.freshness.outputs.head_sha }} - published: ${{ steps.publish_live.outcome == 'success' && 'true' || 'false' }} - runs-on: blacksmith-6vcpu-macos-15 - # Why 150: it must exceed the worst case the retry budgets below can produce - # (install 3x10 + publish 2x45 = 120, plus ~25 for checkout/build/verify), or - # the job is killed mid-retry and no cleanup step runs at all. A typical run - # is far shorter — this is the notary queue's tail, not its median. - timeout-minutes: 150 - env: - NODE_OPTIONS: --max-old-space-size=4096 steps: - - name: Checkout - uses: actions/checkout@v6 - with: - ref: main - fetch-depth: 0 - # Why: this job only reads stablyai/orca and never pushes; every write - # goes to the hourly repo through a minted App token passed by env. - # Not persisting the checkout credential shrinks the blast radius if a - # build step is compromised (zizmor: artipacked). - persist-credentials: false - - name: Mint hourly repo token id: app_token uses: actions/create-github-app-token@v2 @@ -95,18 +77,19 @@ jobs: private-key: ${{ secrets.HOURLY_RELEASE_APP_PRIVATE_KEY }} owner: stablyai repositories: orca-hourly + permission-contents: read - # Why: main is often idle overnight. Rebuilding an unchanged commit burns a - # runner hour and adds a redundant tag to the retention window. - name: Check whether main moved since the last hourly id: freshness shell: bash env: GH_TOKEN: ${{ steps.app_token.outputs.token }} + MAIN_REPO_TOKEN: ${{ github.token }} FORCED: ${{ github.event_name == 'workflow_dispatch' && inputs.force }} run: | set -euo pipefail - head_sha="$(git rev-parse HEAD)" + head_sha="$(GH_TOKEN="$MAIN_REPO_TOKEN" gh api "repos/$GITHUB_REPOSITORY/commits/main" --jq .sha)" + [[ "$head_sha" =~ ^[0-9a-f]{40}$ ]] || { echo "::error::Could not resolve main"; exit 1; } echo "head_sha=$head_sha" >>"$GITHUB_OUTPUT" if [[ "$FORCED" == "true" ]]; then echo "should_build=true" >>"$GITHUB_OUTPUT" @@ -133,21 +116,55 @@ jobs: echo "main moved to $head_sha (last hourly built $last_sha); building." fi + build-hourly-mac: + needs: preflight + if: needs.preflight.outputs.should_build == 'true' + outputs: + tag: ${{ steps.release.outputs.tag }} + version: ${{ steps.hourly.outputs.version }} + head_sha: ${{ needs.preflight.outputs.head_sha }} + published: ${{ steps.publish_live.outcome == 'success' && 'true' || 'false' }} + runs-on: blacksmith-6vcpu-macos-15 + # Why 150: it must exceed the worst case the retry budgets below can produce + # (install 3x10 + publish 2x45 = 120, plus ~25 for checkout/build/verify), or + # the job is killed mid-retry and no cleanup step runs at all. A typical run + # is far shorter — this is the notary queue's tail, not its median. + timeout-minutes: 150 + env: + NODE_OPTIONS: --max-old-space-size=4096 + steps: + - name: Checkout + uses: actions/checkout@v6 + with: + ref: ${{ needs.preflight.outputs.head_sha }} + fetch-depth: 0 + # Why: this job only reads stablyai/orca and never pushes; every write + # goes to the hourly repo through a minted App token passed by env. + # Not persisting the checkout credential shrinks the blast radius if a + # build step is compromised (zizmor: artipacked). + persist-credentials: false + + - name: Mint hourly repo token + id: app_token + uses: actions/create-github-app-token@v2 + with: + app-id: ${{ secrets.HOURLY_RELEASE_APP_ID }} + private-key: ${{ secrets.HOURLY_RELEASE_APP_PRIVATE_KEY }} + owner: stablyai + repositories: orca-hourly + - name: Setup pnpm - if: steps.freshness.outputs.should_build == 'true' uses: pnpm/setup@v2 with: install: false - name: Setup Node.js - if: steps.freshness.outputs.should_build == 'true' uses: actions/setup-node@v6 with: node-version-file: package.json cache: pnpm - name: Cache electron-builder downloads - if: steps.freshness.outputs.should_build == 'true' uses: actions/cache@v5 with: path: | @@ -158,7 +175,6 @@ jobs: electron-builder-mac- - name: Install dependencies - if: steps.freshness.outputs.should_build == 'true' uses: nick-fields/retry@v4 with: timeout_minutes: 10 @@ -169,7 +185,6 @@ jobs: # Why: signing is what makes an hourly installable over an existing Orca, so # a missing cert must fail here rather than after a 20-minute build. - name: Verify macOS signing environment - if: steps.freshness.outputs.should_build == 'true' run: node config/scripts/verify-macos-release-env.mjs env: CSC_LINK: ${{ secrets.MAC_CERTS }} @@ -180,7 +195,6 @@ jobs: - name: Compute hourly version id: hourly - if: steps.freshness.outputs.should_build == 'true' shell: bash env: GH_TOKEN: ${{ steps.app_token.outputs.token }} @@ -211,7 +225,7 @@ jobs: node config/scripts/hourly-build-version.mjs \ >"$RUNNER_TEMP/hourly-identity.txt" grep -E '^(version|build_number)=' "$RUNNER_TEMP/hourly-identity.txt" - # Why check rather than trust: the checkout above pins `ref: main`, but a + # Why check rather than trust: the checkout above pins the resolved main commit, but a # workflow_dispatch runs this file from whatever branch was dispatched. A # branch that edits this step while main still has the old script yields # an empty name and an untitled release — silent, and only visible once @@ -223,7 +237,6 @@ jobs: cat "$RUNNER_TEMP/hourly-identity.txt" >>"$GITHUB_OUTPUT" - name: Build app - if: steps.freshness.outputs.should_build == 'true' run: pnpm build:release env: NODE_OPTIONS: --max-old-space-size=4096 @@ -239,7 +252,6 @@ jobs: # part the full budget. - name: Re-mint hourly repo token for publish id: app_token_publish - if: steps.freshness.outputs.should_build == 'true' uses: actions/create-github-app-token@v2 with: app-id: ${{ secrets.HOURLY_RELEASE_APP_ID }} @@ -249,13 +261,12 @@ jobs: - name: Create hourly release id: release - if: steps.freshness.outputs.should_build == 'true' shell: bash env: GH_TOKEN: ${{ steps.app_token_publish.outputs.token }} TAG: v${{ steps.hourly.outputs.version }} NAME: ${{ steps.hourly.outputs.name }} - SHA: ${{ steps.freshness.outputs.head_sha }} + SHA: ${{ needs.preflight.outputs.head_sha }} run: | set -euo pipefail # Kept at 12 even though the title shows 7: the freshness check above @@ -291,7 +302,6 @@ jobs: echo "tag=$TAG" >>"$GITHUB_OUTPUT" - name: Publish hourly macOS artifacts - if: steps.freshness.outputs.should_build == 'true' uses: nick-fields/retry@v4 with: # Why 45 like the release pipeline: an attempt is pack + notarize + @@ -322,7 +332,6 @@ jobs: # release missing that manifest is a tag the picker offers and the download # 404s on, so fail loudly instead of leaving a broken entry. - name: Verify update manifest published - if: steps.freshness.outputs.should_build == 'true' shell: bash env: GH_TOKEN: ${{ steps.app_token_publish.outputs.token }} @@ -352,7 +361,6 @@ jobs: # means the picker can never offer a release whose assets are incomplete. - name: Publish the verified release id: publish_live - if: steps.freshness.outputs.should_build == 'true' shell: bash env: GH_TOKEN: ${{ steps.app_token_publish.outputs.token }} diff --git a/config/scripts/ci-native-toolchain.test.mjs b/config/scripts/ci-native-toolchain.test.mjs new file mode 100644 index 00000000000..e35437da77c --- /dev/null +++ b/config/scripts/ci-native-toolchain.test.mjs @@ -0,0 +1,69 @@ +import { execFileSync } from 'node:child_process' +import { chmodSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { parse } from 'yaml' +import { describe, expect, it } from 'vitest' + +const steps = parse(readFileSync('.github/actions/install-node-dependencies/action.yml', 'utf8')) + .runs.steps +const toolchain = steps.find((step) => step.name === 'Use external node-gyp') + +describe('CI native toolchain preparation', () => { + it('probes only after both cache restore variants and before native rebuilding', () => { + const index = steps.indexOf(toolchain) + for (const id of ['native-cache-restore', 'native-cache-restore-only']) { + expect(index).toBeGreaterThan(steps.findIndex((step) => step.id === id)) + expect(toolchain.env.NATIVE_CACHE_HIT).toContain(`steps.${id}.outputs.cache-hit`) + } + expect(index).toBeLessThan(steps.findIndex((step) => step.name === 'Prepare native runtime')) + expect(toolchain.if).toBe("runner.os == 'Linux' && inputs.native-runtime != 'none'") + }) + + // The action's toolchain workaround only runs in Linux Bash. + it.skipIf(process.platform === 'win32').each([ + ['node', 'true', '0', false], + ['node', 'true', '1', true], + ['node', 'false', '0', true], + ['node', '', '0', true], + ['electron', 'true', '0', true], + ['electron', 'false', '0', true] + ])('runtime=%s cache=%s probe=%s installs=%s', (runtime, hit, probeStatus, installs) => { + const directory = mkdtempSync(join(tmpdir(), 'orca-ci-native-toolchain-')) + const log = join(directory, 'commands') + const environment = join(directory, 'github-env') + try { + writeFileSync(log, '') + writeFileSync(environment, '') + for (const [name, source] of [ + ['node', 'echo "node $*" >> "$COMMAND_LOG"\nexit "$PROBE_STATUS"'], + ['npm', 'echo "npm $*" >> "$COMMAND_LOG"\nif [ "$1" = root ]; then echo /global; fi'] + ]) { + const path = join(directory, name) + writeFileSync(path, `#!/bin/sh\n${source}\n`) + chmodSync(path, 0o755) + } + execFileSync('bash', ['-e', '-o', 'pipefail', '-c', toolchain.run], { + env: { + ...process.env, + PATH: `${directory}:${process.env.PATH}`, + NATIVE_RUNTIME: runtime, + NATIVE_CACHE_HIT: hit, + PROBE_STATUS: probeStatus, + COMMAND_LOG: log, + GITHUB_ENV: environment + } + }) + const commands = readFileSync(log, 'utf8') + expect(commands.includes('npm install -g node-gyp@11.5.0')).toBe(installs) + expect(commands.includes('node config/scripts/ensure-native-runtime.mjs --check-only')).toBe( + runtime === 'node' && hit === 'true' + ) + expect(readFileSync(environment, 'utf8')).toBe( + installs ? 'npm_config_node_gyp=/global/node-gyp/bin/node-gyp.js\n' : '' + ) + } finally { + rmSync(directory, { recursive: true, force: true }) + } + }) +}) diff --git a/config/scripts/hourly-preflight-workflow.test.mjs b/config/scripts/hourly-preflight-workflow.test.mjs new file mode 100644 index 00000000000..2bec40b329b --- /dev/null +++ b/config/scripts/hourly-preflight-workflow.test.mjs @@ -0,0 +1,91 @@ +import { mkdtempSync, readFileSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { parse } from 'yaml' +import { runProcess } from '../../src/shared/child-process/run-process' + +const workflow = parse( + readFileSync(new URL('../../.github/workflows/hourly-mac-build.yml', import.meta.url), 'utf8') +) +const preflight = workflow.jobs.preflight +const freshness = preflight.steps.find((step) => step.id === 'freshness') +const head = 'abcdef0123'.repeat(4) + +async function checkFreshness(overrides = {}) { + const directory = mkdtempSync(join(tmpdir(), 'hourly-preflight-')) + const output = join(directory, 'output') + try { + const result = await runProcess({ + program: 'bash', + args: [ + '-c', + `gh() { + case "$1 $2" in + "api "*) printf '%s\\n' "$HEAD_SHA" ;; + "release list") printf '%s\\n' "$LAST_TAG" ;; + "release view") printf '%s\\n' "$LAST_SHA" ;; + *) return 1 ;; + esac + } + ${freshness.run}` + ], + env: { + ...process.env, + GITHUB_OUTPUT: output, + GITHUB_REPOSITORY: 'stablyai/orca', + MAIN_REPO_TOKEN: 'main-token', + HOURLY_REPO: 'stablyai/orca-hourly', + HEAD_SHA: head, + LAST_TAG: 'previous-hourly', + LAST_SHA: head.slice(0, 12), + FORCED: 'false', + ...overrides + } + }) + return { + exitCode: result.code, + stderr: result.stderr, + stdout: result.stdout, + output: result.code === 0 ? readFileSync(output, 'utf8') : '' + } + } finally { + rmSync(directory, { recursive: true, force: true }) + } +} + +describe('hourly build preflight', () => { + it('gates Mac allocation and pins the checkout and downstream identity', () => { + const build = workflow.jobs['build-hourly-mac'] + expect(preflight['runs-on']).toBe('ubuntu-latest') + expect(preflight.steps.some((step) => step.uses?.startsWith('actions/checkout'))).toBe(false) + expect( + preflight.steps.find((step) => step.id === 'app_token').with['permission-contents'] + ).toBe('read') + expect(build.needs).toBe('preflight') + expect(build.if).toBe("needs.preflight.outputs.should_build == 'true'") + expect(build.steps.find((step) => step.name === 'Checkout').with.ref).toBe( + build.outputs.head_sha + ) + expect(build.outputs.head_sha).toBe('${{ needs.preflight.outputs.head_sha }}') + expect(build.steps.find((step) => step.id === 'release').env.SHA).toBe(build.outputs.head_sha) + expect(workflow.concurrency).toEqual({ group: 'hourly-mac-build', 'cancel-in-progress': false }) + }) + + it.each([ + ['unchanged', {}, false], + ['changed', { LAST_SHA: '123456789012' }, true], + ['forced', { FORCED: 'true' }, true], + ['first build', { LAST_TAG: '' }, true], + ['missing prior identity', { LAST_SHA: '' }, true] + ])('%s main selects the expected build decision', async (_name, env, shouldBuild) => { + const result = await checkFreshness(env) + expect(result.exitCode, `${result.stdout} ${result.stderr}`).toBe(0) + expect(result.output).toBe(`head_sha=${head}\nshould_build=${shouldBuild}\n`) + }) + + it('fails closed when main cannot be resolved, even when forced', async () => { + const result = await checkFreshness({ HEAD_SHA: '', FORCED: 'true' }) + expect(result.exitCode).not.toBe(0) + }) +}) diff --git a/docs/reference/ci-runner-efficiency.md b/docs/reference/ci-runner-efficiency.md index e569f749102..a9f644bc435 100644 --- a/docs/reference/ci-runner-efficiency.md +++ b/docs/reference/ci-runner-efficiency.md @@ -21,8 +21,9 @@ billing minutes or queue time. This small sample is not a historical average. default Debian/RPM compression is xz. PR artifacts are inspected on the same runner, so their download size offers no benefit. Keep all AppImage, Debian, RPM, payload, launcher, and shutdown checks. Release compression is unchanged. - Compression savings need a hosted run; do not equate the full packaging step - with removable compression time. + Hosted validation in [33999422341](https://github.com/stablyai/orca/actions/runs/33999422341) + reduced the package-build step to 2m13s and the full Linux job to 6m17s, with + all existing checks passing. This is a small observational sample. - Cancel superseded Mobile Checks and Skill update round-trip PR runs. The skill matrix has 13 jobs. Preserve non-cancelling main/merge-group skill runs, with separate concurrency groups per event. @@ -36,6 +37,21 @@ caching, and changed-spec E2E routing. Increasing shards would increase setup work and simultaneous runner demand. Do not adjust the count without comparing critical-path time and aggregate job time on the same commit. +## Follow-up savings + +- Move the hourly main/release freshness lookup to a five-minute Ubuntu + preflight without a checkout. In unchanged run + [33986205749](https://github.com/stablyai/orca/actions/runs/33986205749), + Blacksmith macOS was occupied for 40 seconds, including a 30-second checkout, + before skipping. The new job-level gate avoids that Mac allocation. Actual + builds gain an Ubuntu scheduling hop; pin the Mac checkout and downstream + Windows identity to the SHA that the preflight checked. +- Avoid global `npm install -g node-gyp` for validated Linux Node-runtime cache + hits. Use the existing native-module load/provenance check before skipping; + misses, broken addons, and Electron jobs still install the rebuild toolchain. + The action file participates in cache keys, so this rollout creates fresh + native caches once. No measured warm-cache seconds are claimed yet. + ## Runner recommendations The repository is **public**, verified using the GitHub API. Standard @@ -66,6 +82,24 @@ See [GitHub Actions billing](https://docs.github.com/en/billing/concepts/product See [pricing](https://ubicloud.com/docs/about/pricing) and [setup](https://ubicloud.com/docs/github-actions-integration/quickstart). +### A bounded Ubicloud candidate + +The Linux leg of `performance-contracts.yml` took 48 seconds in +[33994756657](https://github.com/stablyai/orca/actions/runs/33994756657). +Its daily schedule and 20-minute timeout make it a small candidate: 31 ordinary +scheduled attempts permit at most 620 job-runtime minutes, before runner +startup/cleanup billing. Actual timings on Ubicloud's 2-vCPU hardware still need +measurement; the GitHub timing is only a sizing reference. + +If enabled later, route only the first attempt of the scheduled Linux job to +Ubicloud; keep PRs, manual dispatches, reruns, and macOS/Windows on GitHub. This +avoids spending the allowance on unpredictable PR volume. Check other account +usage and available credit before enabling; a workflow timeout is not an +account-wide billing cap. On September 5, the organization's GitHub App +installation list contained Blacksmith but no Ubicloud installation, so this +follow-up leaves runner selection on GitHub rather than queueing work against +an unprovisioned label. + ## Machines that also run coding agents Do not register the credentialed host directly as a public-PR runner. A PR can diff --git a/src/main/artifacts/artifact-cloud-recovery.test.ts b/src/main/artifacts/artifact-cloud-recovery.test.ts index 5690a37b94c..fea3b73bda0 100644 --- a/src/main/artifacts/artifact-cloud-recovery.test.ts +++ b/src/main/artifacts/artifact-cloud-recovery.test.ts @@ -1,7 +1,7 @@ import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' -import { afterEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' vi.mock('electron', () => ({ app: { isPackaged: false }, @@ -21,7 +21,14 @@ const writeRequest = { authToken: 'token-a' } +beforeEach(() => { + // Keep fixed response expirations independent of the runner's wall clock. + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime('2026-08-07T00:00:00.000Z') +}) + afterEach(async () => { + vi.useRealTimers() vi.unstubAllGlobals() await Promise.all( createdPaths.splice(0).map((path) => rm(path, { recursive: true, force: true })) diff --git a/src/main/artifacts/artifact-cloud-service-races.test.ts b/src/main/artifacts/artifact-cloud-service-races.test.ts index 8606dce4bec..28195bf6965 100644 --- a/src/main/artifacts/artifact-cloud-service-races.test.ts +++ b/src/main/artifacts/artifact-cloud-service-races.test.ts @@ -1,7 +1,7 @@ import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' -import { afterEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' vi.mock('electron', () => ({ app: { isPackaged: false }, @@ -50,7 +50,14 @@ async function setup(): Promise { return new ArtifactCloudService(path, () => true) } +beforeEach(() => { + // Keep fixed response expirations independent of the runner's wall clock. + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime('2026-08-07T00:00:00.000Z') +}) + afterEach(async () => { + vi.useRealTimers() vi.unstubAllGlobals() await Promise.all( createdPaths.splice(0).map((path) => rm(path, { recursive: true, force: true })) diff --git a/src/main/artifacts/artifact-cloud-service.test.ts b/src/main/artifacts/artifact-cloud-service.test.ts index 8a02478feb2..c747e6517aa 100644 --- a/src/main/artifacts/artifact-cloud-service.test.ts +++ b/src/main/artifacts/artifact-cloud-service.test.ts @@ -1,7 +1,7 @@ import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' -import { afterEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' vi.mock('electron', () => ({ app: { isPackaged: false }, @@ -95,6 +95,12 @@ const writeRequest = { authToken: 'token-a' } +beforeEach(() => { + // Keep fixed response expirations independent of the runner's wall clock. + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime('2026-08-07T00:00:00.000Z') +}) + afterEach(async () => { vi.useRealTimers() vi.unstubAllGlobals() From 5238a4d57684786cd32f5edf839f7430510de174 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 17:39:02 -0700 Subject: [PATCH 15/17] test: isolate Source Control generation from shared repository remotes (#18962) --- tests/e2e/helpers/seeded-test-repo.ts | 6 ++++-- .../helpers/source-control-generation-app.ts | 21 +++++++++++++++++++ ...ource-control-pr-generation-switch.spec.ts | 2 +- .../source-control-pr-linked-issue-ai.spec.ts | 2 +- 4 files changed, 27 insertions(+), 4 deletions(-) create mode 100644 tests/e2e/helpers/source-control-generation-app.ts diff --git a/tests/e2e/helpers/seeded-test-repo.ts b/tests/e2e/helpers/seeded-test-repo.ts index dd88351b282..34c4f346714 100644 --- a/tests/e2e/helpers/seeded-test-repo.ts +++ b/tests/e2e/helpers/seeded-test-repo.ts @@ -28,7 +28,7 @@ export function isValidGitRepo(repoPath: string): boolean { } } -export function createSeededTestRepo(): string { +export function createSeededTestRepo(options: { publishPath?: boolean } = {}): string { // Why: realpathSync so the seeded path matches the store's repo.path on // macOS, where os.tmpdir() (/var/...) symlinks to /private/var/... and the // app canonicalizes repo.path via `git rev-parse --show-toplevel` on add. @@ -63,6 +63,8 @@ export function createSeededTestRepo(): string { stdio: 'pipe' }) - writeFileSync(TEST_REPO_PATH_FILE, testRepoDir) + if (options.publishPath !== false) { + writeFileSync(TEST_REPO_PATH_FILE, testRepoDir) + } return testRepoDir } diff --git a/tests/e2e/helpers/source-control-generation-app.ts b/tests/e2e/helpers/source-control-generation-app.ts new file mode 100644 index 00000000000..2a1d00ca390 --- /dev/null +++ b/tests/e2e/helpers/source-control-generation-app.ts @@ -0,0 +1,21 @@ +import { test as base, expect } from './orca-app' +import { createSeededTestRepo } from './seeded-test-repo' +import { cleanupTestRepository } from '../global-teardown' + +export { expect } + +export const test = base.extend({ + testRepoPath: [ + // oxlint-disable-next-line no-empty-pattern -- Playwright requires destructured fixture arguments. + async ({}, provideFixture) => { + // Generation must not fetch external remotes installed by unrelated specs. + const repoPath = createSeededTestRepo({ publishPath: false }) + try { + await provideFixture(repoPath) + } finally { + cleanupTestRepository(repoPath) + } + }, + { scope: 'worker' } + ] +}) diff --git a/tests/e2e/source-control-pr-generation-switch.spec.ts b/tests/e2e/source-control-pr-generation-switch.spec.ts index 58091cd3cf8..47b4acb1b5d 100644 --- a/tests/e2e/source-control-pr-generation-switch.spec.ts +++ b/tests/e2e/source-control-pr-generation-switch.spec.ts @@ -1,7 +1,7 @@ import type { Page, TestInfo } from '@stablyai/playwright-test' import { mkdirSync, readFileSync, writeFileSync } from 'node:fs' import path from 'node:path' -import { test, expect } from './helpers/orca-app' +import { test, expect } from './helpers/source-control-generation-app' import { waitForActiveWorktree, waitForSessionReady } from './helpers/store' import { createBranchCommit, diff --git a/tests/e2e/source-control-pr-linked-issue-ai.spec.ts b/tests/e2e/source-control-pr-linked-issue-ai.spec.ts index 625d19d58b8..be6966b7f3a 100644 --- a/tests/e2e/source-control-pr-linked-issue-ai.spec.ts +++ b/tests/e2e/source-control-pr-linked-issue-ai.spec.ts @@ -1,7 +1,7 @@ import { rmSync } from 'node:fs' import os from 'node:os' import path from 'node:path' -import { test, expect } from './helpers/orca-app' +import { test, expect } from './helpers/source-control-generation-app' import { createBranchCommit, openSourceControl, From 61b09b7a0257e77563f51d16fd9d78b55e64ba2e Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:47:27 -0400 Subject: [PATCH 16/17] fix(relay): abandon dead client accepts, jitter and lengthen the control lease, fail direct probes fast (#18959) * fix(relay): abandon a client accept once the phone hangs up; jitter the control lease The accept runs several serialized Postgres calls behind the contended cell-inventory lock, and phones bound their dial. Finishing that work for a phone that had already left acquired (and leaked for 90s) an activity lease and then failed at bind with host_data_reservation_already_bound. Check the client socket between the DB steps and unwind what was taken, reporting the stage on orca_relay_client_accept_abandoned. Jitter the control lease grant so a cohort that reconnected in the same minute (a cell recreate dumps hundreds at once) walks apart instead of rebinding together every cycle. On the phone, treat a probe session that enters 'reconnecting' as a failed probe: it is the direct client's own backoff after a dead-LAN 1006, and waiting it out held the supervisor's operation mutex for the full 12s bound. * perf(relay): lengthen the control lease to 6h The lease bounds how long a host lingers on a cell after a missed drain, and rebinding it is the only passive rebalancing we have, so it stays finite. 6h keeps both properties while cutting control-activation traffic on the contended cell-inventory lock ~6x. The relay JWT (5 min, refreshed by the desktop) and the 75s silence watchdog are enforced separately, so the longer grant authorizes nothing extra. The jitter widens with it, to +/-30 min. * fix(relay): let one flap recover the direct probe; correct the leak window 'reconnecting' is published on any socket close, so rejecting on it outright turned a single access-point flap into a booked direct failure and a 60s cooldown. Give the first 'reconnecting' a 2s grace in which a 'connected' transition still resolves; a dead LAN still fails in ~2s rather than holding the supervisor's operation mutex for the 12s bound. The abandoned accept held its activity lease for the 10s attach deadline, not 90s -- the attach timer is armed before bind throws and already unwinds it. Also cover the assignment-stage check that guards reserveCredential, and drop a spread assertion the two exact-value assertions above already imply. * fix(relay): extend the probe grace once on a handshake; pin the lease band top The redial fires at 500ms but 'connected' waits on the Noise handshake and a capability RPC, so one 2s window is too tight for real work. A 'handshaking' transition is evidence the peer answered, so extend the grace once; a stalled handshake still fails at ~3.5s, far inside the 12s bound. The longest-lease case only had an upper bound, which a jitter clamped to one side would satisfy. Pin it to the exact top of the band instead, and assert the assignment resolve ran so the third-guard test cannot pass vacuously. --- .../src/host-session-client-accept.test.ts | 382 ++++++++++++++++++ cloud/apps/relay/src/host-session-registry.ts | 62 ++- .../relay/src/relay-observability.test.ts | 11 +- cloud/apps/relay/src/relay-observability.ts | 18 + cloud/apps/relay/src/relay-server.ts | 4 +- .../mobile-direct-endpoint-probe.test.ts | 110 +++++ .../transport/mobile-direct-endpoint-probe.ts | 33 +- ...e-endpoint-supervisor-direct-probe.test.ts | 51 +++ 8 files changed, 660 insertions(+), 11 deletions(-) create mode 100644 cloud/apps/relay/src/host-session-client-accept.test.ts create mode 100644 mobile/src/transport/mobile-endpoint-supervisor-direct-probe.test.ts diff --git a/cloud/apps/relay/src/host-session-client-accept.test.ts b/cloud/apps/relay/src/host-session-client-accept.test.ts new file mode 100644 index 00000000000..83b6c21f997 --- /dev/null +++ b/cloud/apps/relay/src/host-session-client-accept.test.ts @@ -0,0 +1,382 @@ +import { EventEmitter } from 'node:events' +import { RELAY_CLOSE_CODE } from '@orca-cloud/relay-contract' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type WebSocket from 'ws' +import type { RelayAssignmentStore } from './assignment-store.js' +import type { RelayConfig } from './config.js' +import type { CredentialReservation, RelayCredentialStore } from './credential-store.js' +import { + CONTROL_LEASE_JITTER_MS, + CONTROL_LEASE_MS, + HostSessionRegistry +} from './host-session-registry.js' +import type { RelayRuntimeObserver } from './relay-observability.js' +import type { RelayTokenClaims } from './relay-token-verifier.js' +import { ProcessQueuedByteBudget } from './splice-forwarder.js' + +// Incident 2026-09-04 ~01:05Z: the phone's dial bound ran out while the cell was +// still inside acceptClient's serialized Postgres phase (cell-inventory lock +// contention). The cell then finished the work for a socket nobody held, holding +// an activity lease for the 10s attach deadline before its timer unwound it, and +// logged `host_data_reservation_already_bound`. + +class FakeSocket extends EventEmitter { + readonly OPEN = 1 + readonly CLOSING = 2 + readonly CLOSED = 3 + readyState = this.OPEN + readonly send = vi.fn() + readonly close = vi.fn((code?: number, reason?: string) => { + this.readyState = this.CLOSED + this.emit('close', code, Buffer.from(reason ?? '')) + }) + readonly terminate = vi.fn(() => { + this.readyState = this.CLOSED + this.emit('close') + }) +} + +const config = { + port: 8080, + publicUrl: 'https://relay-c3.example.com', + cellUrl: 'https://relay-c3.example.com', + authIssuer: 'https://auth.example.com', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.com/jwks', + assignmentSigningKey: new Uint8Array(32), + role: 'cell', + cellId: 'production-gce-c3', + cells: [{ id: 'production-gce-c3', url: 'https://relay-c3.example.com', capacityRequests: 4_000 }], + adminAudience: 'https://relay-c3.example.com/v1/admin/drain', + deployServiceAccount: 'deploy@example.com', + runtimeServiceAccount: 'runtime@example.com', + adminJwksUrl: 'https://auth.example.com/admin-jwks', + databasePoolMax: 10, + publicAssignmentsEnabled: true, + publicAssignmentConcurrency: 2, + publicAssignmentQueueMax: 128, + publicAssignmentWaitMs: 4_000, + publicResolveConcurrency: 1, + publicResolveWaitMs: 5_000, + publicAssignmentRetryAfterSeconds: 5, + dataDir: './test-data' +} satisfies RelayConfig + +const identity = { + sub: 'user-1', + prof: 'profile-1', + relayHostId: 'abcdefghijklmnop', + purpose: 'host-control', + exp: 4_102_444_800 +} satisfies RelayTokenClaims + +function deferred(): { promise: Promise; resolve: (value: T) => void } { + let resolve!: (value: T) => void + const promise = new Promise((next) => (resolve = next)) + return { promise, resolve } +} + +const reservation: CredentialReservation = { + userId: identity.sub, + relayHostId: identity.relayHostId, + credentialKind: 'resume', + relayDeviceId: 'device-1', + tokenHash: 'hash', + reservationId: 'reservation-1', + leaseExpiresAt: Date.now() + 60_000, + acceptedCredentialVersion: 2, + acceptedAs: 'current' +} + +function harness(options: { random?: () => number; now?: () => number } = {}) { + const acquireActivity = vi.fn().mockResolvedValue(undefined) + const releaseActivity = vi.fn().mockResolvedValue(true) + const assignments = { + activateControl: vi.fn().mockResolvedValue('control:production-gce-c3:1'), + markMigrationTargetRegistered: vi.fn().mockResolvedValue(undefined), + resolve: vi.fn().mockResolvedValue({ cellId: config.cellId }), + acquireActivity, + renewControlActivity: vi.fn().mockResolvedValue(undefined), + releaseActivity + } as unknown as RelayAssignmentStore + const store = { + resolveResume: vi.fn().mockResolvedValue({ userId: identity.sub }), + reserveCredential: vi.fn().mockResolvedValue(reservation), + failReservation: vi.fn().mockResolvedValue(undefined) + } + const observer = { + recordAuth: vi.fn(), + recordForwardedBytes: vi.fn(), + recordHttp: vi.fn(), + recordReconnect: vi.fn(), + recordSql: vi.fn(), + recordClientAcceptAbandoned: vi.fn() + } satisfies RelayRuntimeObserver + const registry = new HostSessionRegistry( + config, + vi.fn(), + store as unknown as RelayCredentialStore, + assignments, + new ProcessQueuedByteBudget(), + observer, + options.now, + options.random + ) + const activate = ( + registry as unknown as { + activate: ( + socket: WebSocket, + identity: RelayTokenClaims, + existing: null, + generation: number, + rebind: boolean, + assignmentEpoch: number, + appVersion: string + ) => Promise + } + ).activate.bind(registry) + return { registry, store, assignments, acquireActivity, releaseActivity, observer, activate } +} + +async function activeHost(h: ReturnType): Promise { + const control = new FakeSocket() + await h.activate(control as unknown as WebSocket, identity, null, 1, false, 1, '1.4.197') + return control +} + +describe('client accept abandoned mid-DB-phase', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => { + vi.clearAllTimers() + vi.useRealTimers() + }) + + it('stops after a slow activity acquire when the phone already hung up', async () => { + const h = harness() + const control = await activeHost(h) + const slowAcquire = deferred() + h.acquireActivity.mockReturnValueOnce(slowAcquire.promise) + const capacity = { bind: vi.fn(), release: vi.fn() } + const client = new FakeSocket() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + try { + const accepting = h.registry.acceptClient( + client as unknown as WebSocket, + identity.relayHostId, + 'credential', + capacity + ) + await vi.advanceTimersByTimeAsync(0) + expect(h.acquireActivity).toHaveBeenCalledOnce() + // The phone's 12s bound fires while the cell still waits on Postgres. + client.close(1000, 'client bound') + capacity.release() + slowAcquire.resolve() + await accepting + + // No conn-open reached the desktop; nothing pending; the lease it just took is + // released instead of leaking to expiry cleanup; bind never throws. + expect(control.send).not.toHaveBeenCalledWith(expect.stringContaining('conn-open')) + expect(capacity.bind).not.toHaveBeenCalled() + const session = h.registry.get({ userId: identity.sub, relayHostId: identity.relayHostId }) + expect(session?.pendingConns.size).toBe(0) + expect(h.store.failReservation).toHaveBeenCalledWith(reservation) + expect(h.releaseActivity).toHaveBeenCalledWith( + { userId: identity.sub, relayHostId: identity.relayHostId }, + expect.stringMatching(/^confirmation:/) + ) + expect(h.observer.recordClientAcceptAbandoned).toHaveBeenCalledWith( + 'activity', + expect.any(Number) + ) + const line = warn.mock.calls.map((call) => String(call[0])).find((entry) => + entry.includes('orca_relay_client_accept_abandoned') + ) + expect(line).toBeDefined() + expect(JSON.parse(line!)).toMatchObject({ stage: 'activity' }) + expect(line).not.toContain(identity.relayHostId) + } finally { + warn.mockRestore() + h.registry.drain(0) + vi.advanceTimersByTime(0) + } + }) + + it('stops after a slow credential reservation without acquiring an activity lease', async () => { + const h = harness() + await activeHost(h) + const slowReserve = deferred() + h.store.reserveCredential.mockReturnValueOnce(slowReserve.promise) + const client = new FakeSocket() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + try { + const accepting = h.registry.acceptClient( + client as unknown as WebSocket, + identity.relayHostId, + 'credential' + ) + await vi.advanceTimersByTimeAsync(0) + client.close(1000, 'client bound') + slowReserve.resolve(reservation) + await accepting + + expect(h.acquireActivity).not.toHaveBeenCalled() + expect(h.store.failReservation).toHaveBeenCalledWith(reservation) + expect(h.observer.recordClientAcceptAbandoned).toHaveBeenCalledWith( + 'credential', + expect.any(Number) + ) + } finally { + warn.mockRestore() + h.registry.drain(0) + vi.advanceTimersByTime(0) + } + }) + + it('stops after a slow resume lookup before starting the invite and assignment lookups', async () => { + const h = harness() + await activeHost(h) + const store = h.store as typeof h.store & { resolveInviteForMove: ReturnType } + store.resolveInviteForMove = vi.fn().mockResolvedValue(null) + const slowResume = deferred() + h.store.resolveResume.mockReturnValueOnce(slowResume.promise) + const resolveAssignment = (h.assignments as unknown as { resolve: ReturnType }) + .resolve + resolveAssignment.mockClear() + const client = new FakeSocket() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + try { + const accepting = h.registry.acceptClient( + client as unknown as WebSocket, + identity.relayHostId, + 'credential' + ) + await vi.advanceTimersByTimeAsync(0) + client.close(1000, 'client bound') + slowResume.resolve(null) + await accepting + + expect(store.resolveInviteForMove).not.toHaveBeenCalled() + expect(resolveAssignment).not.toHaveBeenCalled() + expect(h.store.reserveCredential).not.toHaveBeenCalled() + expect(h.observer.recordClientAcceptAbandoned).toHaveBeenCalledWith( + 'assignment', + expect.any(Number) + ) + } finally { + warn.mockRestore() + h.registry.drain(0) + vi.advanceTimersByTime(0) + } + }) + + it('stops after a slow same-cell assignment resolve, before reserving a credential', async () => { + const h = harness() + await activeHost(h) + const resolveAssignment = (h.assignments as unknown as { resolve: ReturnType }) + .resolve + const slowResolve = deferred<{ cellId: string }>() + resolveAssignment.mockReturnValueOnce(slowResolve.promise) + const client = new FakeSocket() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + try { + const accepting = h.registry.acceptClient( + client as unknown as WebSocket, + identity.relayHostId, + 'credential' + ) + await vi.advanceTimersByTimeAsync(0) + client.close(1000, 'client bound') + // A correct, same-cell assignment: only the closed socket stops the accept. + slowResolve.resolve({ cellId: config.cellId }) + await accepting + + // Proves the accept reached the third guard, not the first. + expect(resolveAssignment).toHaveBeenCalled() + expect(h.store.reserveCredential).not.toHaveBeenCalled() + expect(h.observer.recordClientAcceptAbandoned).toHaveBeenCalledWith( + 'assignment', + expect.any(Number) + ) + } finally { + warn.mockRestore() + h.registry.drain(0) + vi.advanceTimersByTime(0) + } + }) + + it('still opens the connection when the phone is holding on', async () => { + const h = harness() + const control = await activeHost(h) + const capacity = { bind: vi.fn(), release: vi.fn() } + const client = new FakeSocket() + await h.registry.acceptClient( + client as unknown as WebSocket, + identity.relayHostId, + 'credential', + capacity + ) + expect(control.send).toHaveBeenCalledWith(expect.stringContaining('"type":"conn-open"')) + expect(capacity.bind).toHaveBeenCalledOnce() + expect(h.observer.recordClientAcceptAbandoned).not.toHaveBeenCalled() + expect(client.close).not.toHaveBeenCalled() + h.registry.drain(0) + vi.advanceTimersByTime(0) + }) +}) + +describe('control lease jitter', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => { + vi.clearAllTimers() + vi.useRealTimers() + }) + + it('grants a lease uniformly around its mean so cohorts drift apart at the same mean rate', async () => { + const now = 1_700_000_000_000 + const helloAck = (socket: FakeSocket) => + JSON.parse( + String(socket.send.mock.calls.find((call) => String(call[0]).includes('host-hello-ack'))![0]) + ) as { leaseExpiresAt: number } + + const shortest = harness({ now: () => now, random: () => 0 }) + const shortestAck = helloAck(await activeHost(shortest)) + const centered = harness({ now: () => now, random: () => 0.5 }) + const centeredAck = helloAck(await activeHost(centered)) + const longestRoll = 0.999999 + const longest = harness({ now: () => now, random: () => longestRoll }) + const longestAck = helloAck(await activeHost(longest)) + + // Pinned, not bounded: a jitter clamped to one side still satisfies an upper + // bound, so only the exact top of the band proves it is symmetric. + const longestOffset = Math.floor((longestRoll * 2 - 1) * CONTROL_LEASE_JITTER_MS) + expect(shortestAck.leaseExpiresAt).toBe(now + CONTROL_LEASE_MS - CONTROL_LEASE_JITTER_MS) + expect(centeredAck.leaseExpiresAt).toBe(now + CONTROL_LEASE_MS) + expect(longestAck.leaseExpiresAt).toBe(now + CONTROL_LEASE_MS + longestOffset) + shortest.registry.drain(0) + centered.registry.drain(0) + longest.registry.drain(0) + vi.advanceTimersByTime(0) + }) + + it('rebinds re-roll the jitter instead of pinning the cohort phase', async () => { + const now = 1_700_000_000_000 + let roll = 0 + const h = harness({ now: () => now, random: () => roll }) + const first = await activeHost(h) + const session = h.registry.get({ userId: identity.sub, relayHostId: identity.relayHostId })! + const firstLease = session.leaseExpiresAt + roll = 0.75 + const rebind = new FakeSocket() + await ( + h.registry as unknown as { + activate: (...args: unknown[]) => Promise + } + ).activate(rebind as unknown as WebSocket, identity, session, 1, true, 1, '1.4.197') + expect(session.leaseExpiresAt).toBe(now + CONTROL_LEASE_MS + CONTROL_LEASE_JITTER_MS / 2) + expect(session.leaseExpiresAt).not.toBe(firstLease) + expect(first.close).toHaveBeenCalledWith(RELAY_CLOSE_CODE.PEER_DROPPED, 'control rebound') + h.registry.drain(0) + vi.advanceTimersByTime(0) + }) +}) diff --git a/cloud/apps/relay/src/host-session-registry.ts b/cloud/apps/relay/src/host-session-registry.ts index 5c53041e7ff..1b7ed3df4af 100644 --- a/cloud/apps/relay/src/host-session-registry.ts +++ b/cloud/apps/relay/src/host-session-registry.ts @@ -29,7 +29,7 @@ import { import { HostCloseReasonMemory } from './host-close-reason-memory.js' import { relayHostLogDigest } from './relay-host-log-digest.js' import type { RelayTokenClaims } from './relay-token-verifier.js' -import type { RelayRuntimeObserver } from './relay-observability.js' +import type { RelayClientAcceptStage, RelayRuntimeObserver } from './relay-observability.js' import type { PendingHostDataReservation } from './relay-connection-ledger.js' import { closeRelayWebSocket } from './relay-websocket-close.js' import { ProcessQueuedByteBudget, wireSplice } from './splice-forwarder.js' @@ -129,6 +129,16 @@ function send(socket: WebSocket, type: string, message: object): void { // stalled predecessor only accumulates doomed sockets. const ACTIVATION_QUEUE_WAIT_MS = 30_000 +// Why: this lease bounds how long a host lingers on a cell after a missed drain, +// and rebinding it is the only passive rebalancing we have, so it has to stay +// finite. 6h keeps both properties while cutting control-activation traffic on +// the contended cell-inventory lock ~6x; the relay JWT (5 min, refreshed by the +// desktop) and the 75s silence watchdog are enforced separately, so a longer +// grant authorizes nothing extra. Symmetric jitter walks same-minute reconnect +// cohorts apart across cycles without changing the mean rebind rate. +export const CONTROL_LEASE_MS = 6 * 60 * 60 * 1000 +export const CONTROL_LEASE_JITTER_MS = 30 * 60 * 1000 + export class HostSessionRegistry { private readonly sessions = new Map() private readonly activationQueues = new Map>() @@ -145,9 +155,16 @@ export class HostSessionRegistry { private readonly assignments: RelayAssignmentStore, private readonly queuedByteBudget: ProcessQueuedByteBudget, private readonly observer: RelayRuntimeObserver, - private readonly now: () => number = Date.now + private readonly now: () => number = Date.now, + private readonly random: () => number = Math.random ) {} + // Uniform over [CONTROL_LEASE_MS - jitter, CONTROL_LEASE_MS + jitter). + private controlLeaseExpiresAt(): number { + const offset = Math.floor((this.random() * 2 - 1) * CONTROL_LEASE_JITTER_MS) + return this.now() + CONTROL_LEASE_MS + offset + } + async acceptClient( socket: WebSocket, hostId: string, @@ -159,10 +176,31 @@ export class HostSessionRegistry { this.rejectClient(socket, RELAY_CLOSE_CODE.DRAINING) return } + // Why: the accept runs several serialized Postgres calls behind the contended + // cell-inventory lock, and phones bound their dial. Finishing the work for a + // phone that already hung up took an activity lease held for the 10s attach + // deadline, then failed at bind with host_data_reservation_already_bound. + const acceptStartedAt = this.now() + const abandonedByClient = (stage: RelayClientAcceptStage, cleanup?: () => void): boolean => { + if (socket.readyState === socket.OPEN) return false + capacityReservation?.release() + cleanup?.() + const elapsedMs = this.now() - acceptStartedAt + this.observer.recordClientAcceptAbandoned?.(stage, elapsedMs) + console.warn( + JSON.stringify({ event: 'orca_relay_client_accept_abandoned', stage, elapsedMs }) + ) + return true + } if (this.config.role === 'cell') { - const outerIdentity = - (await this.store.resolveResume(hostId, credential)) ?? - (await this.store.resolveInviteForMove(hostId, credential)) + // Each lookup is its own pooled round trip; stop between them once the phone + // has left instead of running the rest of the chain for nobody. + let outerIdentity = await this.store.resolveResume(hostId, credential) + if (abandonedByClient('assignment')) return + if (!outerIdentity) { + outerIdentity = await this.store.resolveInviteForMove(hostId, credential) + if (abandonedByClient('assignment')) return + } const assignment = outerIdentity ? await this.assignments.resolve({ userId: outerIdentity.userId, relayHostId: hostId }) : null @@ -172,6 +210,7 @@ export class HostSessionRegistry { this.rejectClient(socket, RELAY_CLOSE_CODE.WRONG_CELL) return } + if (abandonedByClient('assignment')) return } const reservation = await this.store.reserveCredential(hostId, credential) if (!reservation) { @@ -181,6 +220,7 @@ export class HostSessionRegistry { return } this.observer.recordAuth(true) + if (abandonedByClient('credential', () => this.failReservationBestEffort(reservation))) return const sessionKey = this.key(reservation.userId, hostId) const session = this.sessions.get(sessionKey) if ( @@ -227,6 +267,14 @@ export class HostSessionRegistry { return } } + if ( + abandonedByClient('activity', () => { + this.failReservationBestEffort(reservation) + if (credentialActivityId) this.releaseActivityBestEffort(identity, credentialActivityId) + }) + ) { + return + } const attachTimer = setTimeout(() => { session.pendingConns.delete(connId) capacityReservation?.release() @@ -740,7 +788,7 @@ export class HostSessionRegistry { existing.socket = socket existing.state = existing.regionalDrainAttemptId ? 'drain-only' : 'active' existing.appVersion = appVersion - existing.leaseExpiresAt = this.now() + 55 * 60 * 1000 + existing.leaseExpiresAt = this.controlLeaseExpiresAt() existing.lastPongAt = this.now() existing.activityRenewalDueAt = this.now() + RELAY_PROTOCOL_LIMITS.controlPingIntervalMs @@ -791,7 +839,7 @@ export class HostSessionRegistry { appVersion, state: 'active', socket, - leaseExpiresAt: this.now() + 55 * 60 * 1000, + leaseExpiresAt: this.controlLeaseExpiresAt(), orphanTimer: null, heartbeatTimer: null, lastPongAt: this.now(), diff --git a/cloud/apps/relay/src/relay-observability.test.ts b/cloud/apps/relay/src/relay-observability.test.ts index 2b9ceb0b72a..fc8a4fcb4af 100644 --- a/cloud/apps/relay/src/relay-observability.test.ts +++ b/cloud/apps/relay/src/relay-observability.test.ts @@ -195,16 +195,23 @@ describe('relay observability', () => { observability.recordControlClose(4402) observability.recordSpliceClose('host-oversize-frame') observability.recordSpliceClose('queue-limit') + observability.recordClientAcceptAbandoned('activity', 14_250.4) + observability.recordClientAcceptAbandoned('activity', 2_000) + observability.recordClientAcceptAbandoned('credential', 3_000) observability.flush(counts) observability.flush(counts) expect(entries[0]).toMatchObject({ controlClosesByCodeDelta: { 1006: 2, 4402: 1 }, - spliceClosesByTriggerDelta: { 'host-oversize-frame': 1, 'queue-limit': 1 } + spliceClosesByTriggerDelta: { 'host-oversize-frame': 1, 'queue-limit': 1 }, + clientAcceptsAbandonedByStageDelta: { activity: 2, credential: 1 }, + clientAcceptAbandonedMsMax: 14_250.4 }) expect(entries[1]).toMatchObject({ controlClosesByCodeDelta: {}, - spliceClosesByTriggerDelta: {} + spliceClosesByTriggerDelta: {}, + clientAcceptsAbandonedByStageDelta: {}, + clientAcceptAbandonedMsMax: 0 }) }) diff --git a/cloud/apps/relay/src/relay-observability.ts b/cloud/apps/relay/src/relay-observability.ts index 2266217d607..59ff437e40b 100644 --- a/cloud/apps/relay/src/relay-observability.ts +++ b/cloud/apps/relay/src/relay-observability.ts @@ -64,8 +64,12 @@ export interface RelayRuntimeObserver { }): void recordControlClose?(code: number): void recordSpliceClose?(trigger: string): void + recordClientAcceptAbandoned?(stage: RelayClientAcceptStage, elapsedMs: number): void } +// Which serialized accept step the phone had already hung up behind. +export type RelayClientAcceptStage = 'assignment' | 'credential' | 'activity' + type RelayMetricDeltas = { forwardedBytes: number authSuccesses: number @@ -87,6 +91,8 @@ type RelayMetricDeltas = { unavailableRegions: Record controlClosesByCode: Record spliceClosesByTrigger: Record + clientAcceptsAbandonedByStage: Record + clientAcceptAbandonedMsMax: number controlRenewalLatenciesMs: number[] controlRenewalsByOutcome: Record controlActivityRecoveries: number @@ -116,6 +122,8 @@ const emptyDeltas = (): RelayMetricDeltas => ({ unavailableRegions: {}, controlClosesByCode: {}, spliceClosesByTrigger: {}, + clientAcceptsAbandonedByStage: {}, + clientAcceptAbandonedMsMax: 0, controlRenewalLatenciesMs: [], controlRenewalsByOutcome: {}, controlActivityRecoveries: 0, @@ -228,6 +236,14 @@ export class RelayObservability implements RelayRuntimeObserver { (this.deltas.spliceClosesByTrigger[trigger] ?? 0) + 1 } + recordClientAcceptAbandoned(stage: RelayClientAcceptStage, elapsedMs: number): void { + increment(this.deltas.clientAcceptsAbandonedByStage, stage) + this.deltas.clientAcceptAbandonedMsMax = Math.max( + this.deltas.clientAcceptAbandonedMsMax, + elapsedMs + ) + } + start(readCounts: () => RelayProcessCounts, intervalMs = 30_000): void { if (this.timer) return this.eventLoop.enable() @@ -289,6 +305,8 @@ export class RelayObservability implements RelayRuntimeObserver { unavailableRegionsDelta: deltas.unavailableRegions, controlClosesByCodeDelta: deltas.controlClosesByCode, spliceClosesByTriggerDelta: deltas.spliceClosesByTrigger, + clientAcceptsAbandonedByStageDelta: deltas.clientAcceptsAbandonedByStage, + clientAcceptAbandonedMsMax: Number(deltas.clientAcceptAbandonedMsMax.toFixed(3)), sqlQueriesDelta: deltas.sqlQueries, sqlFailuresDelta: deltas.sqlFailures, sqlLatencyMsMax: Number(deltas.sqlLatencyMsMax.toFixed(3)), diff --git a/cloud/apps/relay/src/relay-server.ts b/cloud/apps/relay/src/relay-server.ts index 32a83962d81..6331b584b1b 100644 --- a/cloud/apps/relay/src/relay-server.ts +++ b/cloud/apps/relay/src/relay-server.ts @@ -88,6 +88,7 @@ export function createRelayServer( database: RelayDatabase, options: { now?: () => number + random?: () => number connectionLedgerLimits?: { hardCap: number; controlReserve: number } cellIncarnation?: string } = {} @@ -123,7 +124,8 @@ export function createRelayServer( assignments, queuedBytes, observability, - options.now + options.now, + options.random ) const app = createRelayApp(config, { store, diff --git a/mobile/src/transport/mobile-direct-endpoint-probe.test.ts b/mobile/src/transport/mobile-direct-endpoint-probe.test.ts index 69fe8a2ab8a..4049b4b073b 100644 --- a/mobile/src/transport/mobile-direct-endpoint-probe.test.ts +++ b/mobile/src/transport/mobile-direct-endpoint-probe.test.ts @@ -71,4 +71,114 @@ describe('mobile direct endpoint probe', () => { expect(clients.get(host.endpoint)?.close).toHaveBeenCalledOnce() expect(result?.client.close).not.toHaveBeenCalled() }) + + it('fails a whole dead LAN in seconds instead of holding the 12s bound', async () => { + // Incident 2026-09-04: foregrounding on a dead LAN produced an instant 1006 and + // the direct client's 500/1000/2000ms redials, while the probe sat on the + // 'connecting' phase and held the supervisor mutex for the whole 12s bound. + const clients: FakeClient[] = [] + const openDirect = vi.fn(() => { + const client = new FakeClient('connecting') + clients.push(client) + setTimeout(() => client.publishState('reconnecting'), 20) + return client + }) + + const probing = openAuthenticatedDirectEndpoint(host, openDirect, 12_000) + await vi.advanceTimersByTimeAsync(20) + await vi.advanceTimersByTimeAsync(2_000) + await expect(probing).resolves.toBeNull() + + expect(clients).toHaveLength(2) + for (const client of clients) { + expect(client.close).toHaveBeenCalledOnce() + } + // No 12s timer is left behind to fire into a settled probe. + expect(vi.getTimerCount()).toBe(0) + }) + + it('rides out one access-point flap that the first redial recovers', async () => { + // 'reconnecting' is published on any socket close, so a single RST on the first + // dial must not book a direct failure and its 60s cooldown. + const openDirect = vi.fn((endpoint: string) => { + const client = new FakeClient('connecting') + if (endpoint.includes('100.64.0.2')) { + setTimeout(() => client.publishState('reconnecting'), 20) + setTimeout(() => client.publishState('connected'), 600) + } + return client + }) + + const probing = openAuthenticatedDirectEndpoint(host, openDirect, 12_000) + await vi.advanceTimersByTimeAsync(600) + const result = await probing + + expect(result?.path).toBe('tailscale') + expect(result?.client.close).not.toHaveBeenCalled() + }) + + it('extends the grace once when the redial reaches a handshake', async () => { + // The redial fires at 500ms, but 'connected' waits on the Noise handshake and a + // capability RPC, so real work needs more than one grace window. + const openDirect = vi.fn((endpoint: string) => { + const client = new FakeClient('connecting') + if (endpoint.includes('100.64.0.2')) { + setTimeout(() => client.publishState('reconnecting'), 20) + setTimeout(() => client.publishState('handshaking'), 1_500) + // Past the first grace window: only the re-arm keeps this probe alive. + setTimeout(() => client.publishState('connected'), 3_000) + } + return client + }) + + const probing = openAuthenticatedDirectEndpoint(host, openDirect, 12_000) + await vi.advanceTimersByTimeAsync(3_000) + + expect((await probing)?.path).toBe('tailscale') + }) + + it('fails a handshake that stalls, one grace after it started', async () => { + const openDirect = vi.fn(() => { + const client = new FakeClient('connecting') + setTimeout(() => client.publishState('reconnecting'), 20) + setTimeout(() => client.publishState('handshaking'), 1_500) + // A restarted handshake must not buy a second extension. + setTimeout(() => client.publishState('handshaking'), 2_500) + return client + }) + + const probing = openAuthenticatedDirectEndpoint(host, openDirect, 12_000) + await vi.advanceTimersByTimeAsync(3_499) + let settled = false + void probing.then(() => { + settled = true + }) + await vi.advanceTimersByTimeAsync(0) + expect(settled).toBe(false) + + await vi.advanceTimersByTimeAsync(1) + await expect(probing).resolves.toBeNull() + expect(vi.getTimerCount()).toBe(0) + }) + + it('gives up at the grace window when the redial never lands', async () => { + const openDirect = vi.fn(() => { + const client = new FakeClient('connecting') + setTimeout(() => client.publishState('reconnecting'), 20) + return client + }) + + const probing = openAuthenticatedDirectEndpoint(host, openDirect, 12_000) + await vi.advanceTimersByTimeAsync(2_019) + let settled = false + void probing.then(() => { + settled = true + }) + await vi.advanceTimersByTimeAsync(0) + expect(settled).toBe(false) + + await vi.advanceTimersByTimeAsync(1) + await expect(probing).resolves.toBeNull() + expect(vi.getTimerCount()).toBe(0) + }) }) diff --git a/mobile/src/transport/mobile-direct-endpoint-probe.ts b/mobile/src/transport/mobile-direct-endpoint-probe.ts index 114a4f29130..03264d5f7e5 100644 --- a/mobile/src/transport/mobile-direct-endpoint-probe.ts +++ b/mobile/src/transport/mobile-direct-endpoint-probe.ts @@ -25,17 +25,45 @@ export function directPathForEndpoint( return 'lan' } +// Why: 'reconnecting' is published on any socket close, so it cannot tell a dead +// LAN (instant 1006, then doomed redials) from one access-point flap that the +// first redial recovers. One redial fits here; a dead LAN still fails in ~2s +// instead of holding the supervisor's operation mutex for the full outer bound. +const RECONNECT_GRACE_MS = 2_000 + function waitForAuthenticatedSession(session: RpcClient, timeoutMs: number): Promise { if (session.getState() === 'connected') { return Promise.resolve() } return new Promise((resolve, reject) => { let timer: ReturnType | null = null + let graceTimer: ReturnType | null = null + let graceExtended = false + const armGrace = (): ReturnType => + setTimeout(() => { + finish() + reject(new Error('probe session reconnecting')) + }, RECONNECT_GRACE_MS) const unsubscribe = session.onStateChange((state) => { if (state === 'connected') { finish() resolve() - } else if (state === 'disconnected' || state === 'auth-failed') { + return + } + if (state === 'reconnecting' && !graceTimer) { + graceTimer = armGrace() + return + } + // Why: the redial fires at 500ms but 'connected' waits on the Noise handshake + // and a capability RPC. 'handshaking' is proof the peer answered, so extend + // once; a dead handshake still fails at ~4s, far inside the outer bound. + if (state === 'handshaking' && graceTimer && !graceExtended) { + graceExtended = true + clearTimeout(graceTimer) + graceTimer = armGrace() + return + } + if (state === 'disconnected' || state === 'auth-failed' || state === 'reconnecting') { finish() reject(new Error(`probe session ${state}`)) } @@ -48,6 +76,9 @@ function waitForAuthenticatedSession(session: RpcClient, timeoutMs: number): Pro if (timer) { clearTimeout(timer) } + if (graceTimer) { + clearTimeout(graceTimer) + } unsubscribe() } }) diff --git a/mobile/src/transport/mobile-endpoint-supervisor-direct-probe.test.ts b/mobile/src/transport/mobile-endpoint-supervisor-direct-probe.test.ts new file mode 100644 index 00000000000..3ee52fc7ddf --- /dev/null +++ b/mobile/src/transport/mobile-endpoint-supervisor-direct-probe.test.ts @@ -0,0 +1,51 @@ +import { beforeEach, afterEach, describe, expect, it, vi } from 'vitest' +import { MobileEndpointSupervisor } from './mobile-endpoint-supervisor' +import { + dependencies, + FakeLogicalClient, + FakeRelaySession, + FakeSession, + host +} from './mobile-endpoint-supervisor-test-fakes' + +vi.mock('react-native', () => ({ Platform: { OS: 'ios' } })) +vi.mock('expo-secure-store', () => ({ WHEN_UNLOCKED_THIS_DEVICE_ONLY: 'when-unlocked' })) +vi.mock('expo-crypto', () => ({ getRandomBytes: (length: number) => new Uint8Array(length) })) + +describe('mobile endpoint supervisor direct probe', () => { + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(new Date('2026-07-13T12:00:00Z')) + }) + + afterEach(() => { + vi.useRealTimers() + }) + + it('does not block relay recovery behind a direct probe stuck in its redial loop', async () => { + const logical = new FakeLogicalClient('connected', 'relay') + const direct = new FakeSession('connecting') + const openRelay = vi.fn(() => new FakeRelaySession('connected')) + const deps = dependencies({ openDirect: vi.fn(() => direct), openRelay }) + const supervisor = new MobileEndpointSupervisor(logical, host, deps) + await supervisor.start() + + // Foreground return: the probe dials direct at once, the dead LAN answers with + // an instant 1006, and the direct client enters its 500/1000/2000ms backoff. + supervisor.setForeground(false) + supervisor.setForeground(true) + await vi.advanceTimersByTimeAsync(0) + expect(deps.openDirect).toHaveBeenCalledOnce() + direct.publishState('reconnecting') + logical.publishState('disconnected') + + // Relay recovery must not wait out the probe's 12s bound; the probe gives up + // one grace window after the redial fails to land. + await vi.advanceTimersByTimeAsync(2_000) + expect(openRelay).toHaveBeenCalledOnce() + expect(direct.close).toHaveBeenCalled() + expect(logical.getState()).toBe('connected') + expect(logical.getActivePath()).toBe('relay') + supervisor.stop() + }) +}) From b6ca8dad99ac5ab8b8174ad33f7a1cd7ac34b068 Mon Sep 17 00:00:00 2001 From: OrcaWin Date: Sat, 5 Sep 2026 17:50:33 -0700 Subject: [PATCH 17/17] fix(hooks): register the Claude hook script directly on Windows (#18875) (#18905) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(hooks): register the Claude hook script directly on Windows (#18875) The Windows Claude Code lifecycle hook was registered as `powershell.exe -NoProfile -EncodedCommand <...>` whose entire decoded payload was a `Test-Path` and a call to `~/.orca/agent-hooks/claude-hook.cmd`. Every hook event paid a full PowerShell start-up to reach a script that exits at its first `ORCA_PANE_KEY` guard, so sessions outside Orca paid it to do nothing. Register the script path itself instead, with `|| echo {}` for the neutral-JSON-when-missing contract (#14818). Measured on Windows 11, invoked as Claude Code invokes it (`printf payload | bash -c -l ""`): idle (n=12) baseline 177ms | before 471ms | after 213ms 10-way conc (n=40) -- | before 656ms | after 296ms p95 under load -- | before 696ms | after 337ms It also drops an interpreter from the chain the hook's timeout kill must tear down. Killing the hook does not kill its PowerShell grandchild, which still holds the stdout handle the agent reads to EOF -- measured, EOF arrived 352ms AFTER the kill, when the orphan exited by itself. msys2 creates children suspended and resumes them after, so a kill landing in that window strands one that never exits and EOF never comes; that is the reported frozen session. The encoded launcher stays as the fallback for profile paths the shells cannot carry bare (space, `%`, `^`, `&`, non-ASCII) and for hosts where Git Bash is not resolvable, because PowerShell 5.1 rejects `||`. Every other agent's hook is untouched, as is the remote/SSH path. Not adopted from the report: `cmd.exe /d /c ` (MSYS rewrites the `/c` under Git Bash -- measured, the invocation fails), and raising the 10s timeout (the orphan survives the kill regardless; the fast path puts the hook 30x under the budget so the kill effectively stops firing). * fix(build): list the new hook launcher modules in the CLI tsconfig project config/tsconfig.cli.json enumerates its files explicitly, so the two new imports reached by src/main/claude/hook-settings.ts failed tc:cli with TS6307. src/main/git-bash.ts pulls in only node:fs, node:path and a shared constant, so it adds nothing heavy to the CLI project. * fix(hooks): address review of the direct Windows Claude hook launcher - Make the Windows hook suites host-independent. A box with a cmd.exe AutoRun (HKCU\...\Command Processor\AutoRun) failed them at HEAD too: the tests redirect USERPROFILE, the AutoRun target vanishes, and MSYS spawns a .cmd without /d so AutoRun runs and lands on the hook's stderr. Seed an empty target, including under the deliberately-absent profile. - Note in managed-hook-stdin-lifecycle why the "missing managed script" case no longer exercises the fallback for the direct shape (it carries an absolute path, so a redirected profile changes nothing); that path is covered live in windows-direct-cmd-hook-command.test.ts. - Keep the direct shape off UNC profiles: WINDOWS_CMD_SAFE_PATH admits them, but //server/share/... is not a command cmd.exe reliably starts. - Correct the comments: `|| echo {}` also fires when cmd.exe itself exits non-zero (failing AutoRun), printing {} twice. The encoded launcher exited 1 on that same box, so neither shape is clean there. - Test the contract that replaced runtime %USERPROFILE% resolution (STA-3348): a stale absolute path reports not_installed and is rewritten on install. - Record the standing unmeasured assumption in windows-edr-posture.md: `||` does not parse in Windows PowerShell 5.1, so a compat consumer that hosts hook strings there would fail closed. Measure before widening to another agent. - Trim the launcher comments per AGENTS.md; the numbers live in the doc. * test(win32): register the new Windows-gated hook test in the CI lane win32-test-lane-registration guards against exactly this: a Windows-gated file that self-skips on ubuntu and reports success, so it runs on no machine. The new windows-direct-cmd-hook-command.test.ts needs both entries — WINDOWS_PACKAGE_TESTS decides whether package_windows runs for a diff, and the workflow argv decides whether the file runs once that job started. * test(win32): remove the hook temp tree through the retrying helper windows-lane-tree-removal-boundary scans exactly the specs in the Windows CI lane, so registering windows-direct-cmd-hook-command.test.ts subjected it to the rule: cmd.exe and bash have just exited in that tree, and a raw recursive rm throws EPERM on Windows while their handles drain, turning a green spec into a lane failure. Use removeTreeSync, which carries the repo's maxRetries policy. --------- Co-authored-by: Orca Worker --- .github/workflows/pr.yml | 1 + config/scripts/pr-code-change-scope.mjs | 1 + config/tsconfig.cli.json | 2 + docs/reference/windows-edr-posture.md | 38 +++- .../managed-hook-stdin-lifecycle.test.ts | 24 ++- .../windows-direct-cmd-hook-command.test.ts | 143 +++++++++++++ .../windows-direct-cmd-hook-command.ts | 30 +++ .../windows-hook-payload-delivery.test.ts | 18 +- .../windows-powershell-hook-launcher.ts | 5 + src/main/claude/hook-service.test.ts | 200 ++++++++++++++++-- src/main/claude/hook-settings.ts | 31 ++- 11 files changed, 470 insertions(+), 23 deletions(-) create mode 100644 src/main/agent-hooks/windows-direct-cmd-hook-command.test.ts create mode 100644 src/main/agent-hooks/windows-direct-cmd-hook-command.ts diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index ba2eaf83192..93bc4c0afc8 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -844,6 +844,7 @@ jobs: src/main/providers/pty-repaint-wide-char-buffer.node-pty.test.ts src/shared/child-process/windows-command-line.win32.test.ts src/main/agent-hooks/windows-hook-payload-delivery.test.ts + src/main/agent-hooks/windows-direct-cmd-hook-command.test.ts src/main/windows/windows-pty-job.win32.test.ts src/main/windows/windows-host-job.win32.test.ts src/main/windows-live-tree-kill.win32.test.ts diff --git a/config/scripts/pr-code-change-scope.mjs b/config/scripts/pr-code-change-scope.mjs index 15ded915c67..fd36a803bb9 100644 --- a/config/scripts/pr-code-change-scope.mjs +++ b/config/scripts/pr-code-change-scope.mjs @@ -217,6 +217,7 @@ const WINDOWS_PACKAGE_TESTS = [ 'src/main/providers/pty-repaint-wide-char-buffer.node-pty.test.ts', 'src/shared/child-process/windows-command-line.win32.test.ts', 'src/main/agent-hooks/windows-hook-payload-delivery.test.ts', + 'src/main/agent-hooks/windows-direct-cmd-hook-command.test.ts', 'src/main/windows/windows-pty-job.win32.test.ts', 'src/main/windows/windows-host-job.win32.test.ts', 'src/main/windows-live-tree-kill.win32.test.ts', diff --git a/config/tsconfig.cli.json b/config/tsconfig.cli.json index 1b9600188f2..2423647577b 100644 --- a/config/tsconfig.cli.json +++ b/config/tsconfig.cli.json @@ -16,6 +16,7 @@ "../src/main/agent-hooks/managed-hook-script-refresh.ts", "../src/main/agent-hooks/posix-hook-command.ts", "../src/main/agent-hooks/runtime-home-hook-command.ts", + "../src/main/agent-hooks/windows-direct-cmd-hook-command.ts", "../src/main/agent-hooks/windows-powershell-hook-launcher.ts", "../src/main/amp/agent-status-plugin-source.ts", "../src/main/amp/hook-service.ts", @@ -117,6 +118,7 @@ "../src/main/hermes/hermes-home-filesystem.ts", "../src/main/hermes/hermes-managed-plugin-source.ts", "../src/main/hermes/hook-service.ts", + "../src/main/git-bash.ts", "../src/main/in-flight-run-dedupe.ts", "../src/main/kimi/hook-service.ts", "../src/main/kimi/kimi-hook-config-toml.ts", diff --git a/docs/reference/windows-edr-posture.md b/docs/reference/windows-edr-posture.md index 06eb2d5ff9b..65287ac0459 100644 --- a/docs/reference/windows-edr-posture.md +++ b/docs/reference/windows-edr-posture.md @@ -166,7 +166,8 @@ What remains is `-EncodedCommand` without the bypass: the PTY bootstraps `src/main/providers/windows-shell-args.ts`), the hook wrappers (`src/main/agent-hooks/windows-powershell-hook-launcher.ts` and its callers `src/main/agent-hooks/runtime-home-hook-command.ts`, -`src/main/agent-hooks/installer-utils.ts`, `src/main/claude/hook-settings.ts`), +`src/main/agent-hooks/installer-utils.ts`, and `src/main/claude/hook-settings.ts` +— that last one only as a *fallback* since #18875, see below), `src/main/runtime/windows-default-route-interfaces.ts`, `src/main/runtime/orchestration/setup-completion-signal.ts`, `src/shared/hermes-startup-query.ts`, and the four ex-bypass sites above. @@ -242,6 +243,41 @@ breadth: every interpreter hop between Orca and the thing the user asked for add a scored edge, which is why the shipped doctrine of #15520 and #15595 is to *shorten the interpreter chain* rather than to hide a window. +#18875 is a worked example of that doctrine. The Claude Code lifecycle hook was +registered as `powershell.exe -NoProfile -EncodedCommand <...>` whose entire +decoded payload was a `Test-Path` and a call to `~/.orca/agent-hooks/claude-hook.cmd`. +It now registers the script path itself (` || echo {}`), so `bash -> +powershell -> cmd -> curl` became `bash -> cmd -> curl` and one +`powershell.exe -EncodedCommand` per hook event — a first-class Defender alert +title — leaves the tree. The reporting box fired ~6 900 of them in five days, +70% from Claude sessions that were not running under Orca at all and whose hook +exits at its first `ORCA_PANE_KEY` guard. + +What is measured is latency and the hop count, nothing else: median 471 ms -> +213 ms per event idle, and 656 ms -> 296 ms (p95 696 ms -> 337 ms) under 10-way +concurrency, invoked as Claude Code invokes it. **No EDR verdict on either tree +was measured**, so claim the removed `-EncodedCommand` spelling and the shorter +chain, not a score. `cmd.exe` remains in the tree, spelled by MSYS's own `.cmd` +spawn rather than by us — the doc's one "unavoidable for `.cmd`/`.bat`" case, +carrying an absolute path and two literal tokens, with no caret escaping, no +encoding and no free text. The encoded launcher is still the shape for profile +paths the shells cannot carry bare (a space, `%`, `^`, `&`, non-ASCII, a UNC +profile) and for hosts where Git Bash is not resolvable, because PowerShell 5.1 +rejects `||` (measured: parse error, exit 1). + +That last clause is the standing assumption of this change, and it is worth +stating plainly because it is **not** measured. `||` parses in Git Bash, cmd.exe +and pwsh, but not in Windows PowerShell 5.1, so the direct shape is correct for +any host that is one of the first three. Claude Code itself is a Git Bash host on +native Windows. What no one here has verified is which host a *compat consumer* +uses: cursor-agent and Devin import `~/.claude/settings.json` and run `command` +through their own launcher (the managed `.cmd` carries a `DEVIN_PROJECT_DIR` skip +for exactly that). If one of them spawns hook strings through Windows PowerShell +5.1, its imported Claude events become a parse error with empty stdout, which is +the fail-closed case #14818 exists to prevent. The encoded launcher had no such +assumption — it was a `powershell.exe` invocation and therefore parsed anywhere. +Before widening the direct shape to another agent, measure that consumer's host. + ### Computer use: screen capture, synthetic input, runtime-compiled MSIL `native/computer-use-windows/runtime.ps1` is a large PowerShell script. diff --git a/src/main/agent-hooks/managed-hook-stdin-lifecycle.test.ts b/src/main/agent-hooks/managed-hook-stdin-lifecycle.test.ts index 40237141d24..af70f6f54f0 100644 --- a/src/main/agent-hooks/managed-hook-stdin-lifecycle.test.ts +++ b/src/main/agent-hooks/managed-hook-stdin-lifecycle.test.ts @@ -4,7 +4,7 @@ // missing-Orca-env path, so their writer may break there. import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { spawn } from 'node:child_process' -import { mkdtempSync, readFileSync, readdirSync, rmSync, writeFileSync } from 'node:fs' +import { mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, writeFileSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' import type { SFTPWrapper } from 'ssh2' @@ -60,6 +60,7 @@ import { DroidHookService } from '../droid/hook-service' import { GeminiHookService } from '../gemini/hook-service' import { GrokHookService } from '../grok/hook-service' import { KimiHookService } from '../kimi/hook-service' + import { openClaudeHookService } from '../openclaude/hook-service' import { wrapPosixHookCommand, wrapWindowsHookCommand } from './installer-utils' import { POSIX_HOOK_STDIN_READER } from './hook-stdin-contract' @@ -69,6 +70,16 @@ import { findGitBash } from './windows-git-bash-path.test-fixture' const REMOTE_HOME = '/home/dev' const LARGE_PAYLOAD = Buffer.alloc(1_000_000, 'x') + +// Why: a developer box may set HKCU\...\Command Processor\AutoRun, which cmd.exe runs before any +// .cmd — and MSYS spawns a .cmd without `/d`, so it fires on the Git Bash legs. Redirecting the +// profile makes the usual `%USERPROFILE%\.cmd_aliases.cmd` target vanish, putting cmd's "not +// recognized" on the hook's stderr. Seed an empty target so these suites measure the launcher +// rather than the host's shell configuration. +function seedCmdAutoRunTarget(profileDir: string): void { + mkdirSync(profileDir, { recursive: true }) + writeFileSync(join(profileDir, '.cmd_aliases.cmd'), '@echo off\r\n', 'utf8') +} const REMOTE_INSTALLERS = [ { agent: 'antigravity', @@ -228,6 +239,7 @@ describe('Windows managed hook stdin structure', () => { it('exits immediately when Orca env is missing and keeps drain for other failures', async () => { const home = mkdtempSync(join(tmpdir(), 'orca-hook-stdin-windows-')) homedirMock.mockReturnValue(home) + seedCmdAutoRunTarget(home) const previousGrokHome = process.env.GROK_HOME const previousKimiHome = process.env.KIMI_CODE_HOME delete process.env.GROK_HOME @@ -317,6 +329,7 @@ describe('Windows managed hook stdin structure', () => { async () => { const home = mkdtempSync(join(tmpdir(), 'orca-hook-stdin-windows-live-')) homedirMock.mockReturnValue(home) + seedCmdAutoRunTarget(home) try { const gitBash = findGitBash() for (const entry of LOCAL_INSTALLERS) { @@ -399,6 +412,9 @@ describe('Windows managed hook stdin structure', () => { async () => { const home = mkdtempSync(join(tmpdir(), 'orca-hook-stdout-json-')) homedirMock.mockReturnValue(home) + const absentProfile = join(home, 'absent') + seedCmdAutoRunTarget(home) + seedCmdAutoRunTarget(absentProfile) try { expect(new ClaudeHookService().install().state).toBe('installed') const settings = JSON.parse( @@ -426,8 +442,12 @@ describe('Windows managed hook stdin structure', () => { }) }, { + // Why: the encoded launcher resolves %USERPROFILE% at run time, so redirecting it is + // what makes the script vanish for that shape. The direct launcher (#18875) carries + // an absolute path, so here it asserts only that a bogus profile changes nothing; its + // missing-script fallback is covered live in windows-direct-cmd-hook-command.test.ts. name: 'missing managed script', - env: hookEnvironment({ USERPROFILE: join(home, 'absent') }) + env: hookEnvironment({ USERPROFILE: absentProfile }) } ] for (const shell of shells) { diff --git a/src/main/agent-hooks/windows-direct-cmd-hook-command.test.ts b/src/main/agent-hooks/windows-direct-cmd-hook-command.test.ts new file mode 100644 index 00000000000..acb4bf2d46d --- /dev/null +++ b/src/main/agent-hooks/windows-direct-cmd-hook-command.test.ts @@ -0,0 +1,143 @@ +// Why (#18875): the registered Windows Claude hook is now the script path itself, so this file +// pins the two things that make that safe — the shape carries nothing MSYS or cmd.exe rewrites, +// and it still answers with neutral JSON when the script is gone. The live legs run the string +// through BOTH hosts Claude Code can pick, because the shape has to parse in either. +import { describe, expect, it } from 'vitest' +import { execFileSync } from 'node:child_process' +import { existsSync, mkdtempSync, readdirSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { removeTreeSync } from '../../shared/windows-transient-lock-removal' +import { WINDOWS_CMD_SAFE_PATH } from './installer-utils' +import { wrapWindowsDirectCmdHookCommand } from './windows-direct-cmd-hook-command' +import { findGitBash } from './windows-git-bash-path.test-fixture' + +const SAFE_PATH = 'C:\\Users\\alice\\.orca\\agent-hooks\\claude-hook.cmd' + +describe('wrapWindowsDirectCmdHookCommand', () => { + it('emits the script path with forward slashes and a neutral-JSON fallback', () => { + expect(wrapWindowsDirectCmdHookCommand(SAFE_PATH)).toBe( + 'C:/Users/alice/.orca/agent-hooks/claude-hook.cmd || echo {}' + ) + }) + + it('spells nothing either shell would rewrite or reinterpret', () => { + const command = wrapWindowsDirectCmdHookCommand(SAFE_PATH)! + + // Why: MSYS rewrites `/c`-shaped tokens into drive paths — a literal `cmd.exe /d /c ` + // does not survive Git Bash (measured), which is why no interpreter is spelled at all. + expect(command).not.toMatch(/ \/[a-zA-Z]+( |$)/) + expect(command).not.toMatch(/\\/) + expect(command).not.toMatch(/["']/) + expect(command).not.toMatch(/powershell|cmd\.exe|conhost/i) + // Why: `2>nul` writes a literal file named `nul` into the cwd under MSYS (measured), and no + // stderr sink parses in both hosts. The missing-script line is left on stderr deliberately. + expect(command).not.toContain('2>') + }) + + it('declines any path the shells cannot carry bare', () => { + for (const path of [ + 'C:\\Users\\Bob Smith\\.orca\\agent-hooks\\claude-hook.cmd', + 'C:\\Users\\%name%\\.orca\\agent-hooks\\claude-hook.cmd', + 'C:\\Users\\a^b\\.orca\\agent-hooks\\claude-hook.cmd', + 'C:\\Users\\a&b\\.orca\\agent-hooks\\claude-hook.cmd', + 'C:\\Users\\a(b)\\.orca\\agent-hooks\\claude-hook.cmd', + 'C:\\Users\\rené\\.orca\\agent-hooks\\claude-hook.cmd', + '/home/alice/.orca/agent-hooks/claude-hook.sh', + // Why: WINDOWS_CMD_SAFE_PATH admits a UNC profile, but `//server/share/...` is not a + // command cmd.exe reliably starts — keep those on the encoded launcher. + '\\\\server\\share\\alice\\.orca\\agent-hooks\\claude-hook.cmd' + ]) { + expect(wrapWindowsDirectCmdHookCommand(path), path).toBeNull() + } + }) +}) + +describe.skipIf(process.platform !== 'win32')('direct hook command, run by both hook hosts', () => { + // Why: the fixture throws when Git Bash is absent, and that is a skip here, not a failure — + // a box without it never gets this command shape in the first place. + const gitBash = ((): string | null => { + try { + return findGitBash() + } catch { + return null + } + })() + + function runInCmd(command: string, cwd: string): { stdout: string; status: number } { + return runCapture('cmd.exe', ['/d', '/c', command], cwd) + } + + function runInBash(command: string, cwd: string): { stdout: string; status: number } { + return runCapture(gitBash!, ['-c', command], cwd) + } + + function runCapture(file: string, args: string[], cwd: string) { + try { + const stdout = execFileSync(file, args, { + cwd, + input: '{"hook_event_name":"PreToolUse"}', + encoding: 'utf8', + stdio: ['pipe', 'pipe', 'pipe'] + }) + return { stdout, status: 0 } + } catch (error) { + const failure = error as { stdout?: string; status?: number } + return { stdout: failure.stdout ?? '', status: failure.status ?? 1 } + } + } + + // Why: a runner whose TEMP sits under a profile with a space is the encoded-launcher case, + // so these legs skip rather than assert a contract that shape never claimed. + const tempIsCmdSafe = WINDOWS_CMD_SAFE_PATH.test(join(tmpdir(), 'orca-direct-hook-x', 'x.cmd')) + const canRunLive = Boolean(gitBash) && tempIsCmdSafe + + function withTempDir(run: (dir: string, scriptPath: string, command: string) => void): void { + const dir = mkdtempSync(join(tmpdir(), 'orca-direct-hook-')) + try { + const scriptPath = join(dir, 'claude-hook.cmd') + const command = wrapWindowsDirectCmdHookCommand(scriptPath) + expect(command, 'precondition: temp path must be cmd-safe').not.toBeNull() + run(dir, scriptPath, command!) + } finally { + // Why: cmd.exe/bash have just exited in this tree; a raw recursive rm throws EPERM on + // Windows while their handles drain. + removeTreeSync(dir) + } + } + + it.skipIf(!canRunLive)('answers {} and exit 0 in both hosts when the script exists', () => { + withTempDir((dir, scriptPath, command) => { + writeFileSync(scriptPath, '@echo off\r\necho {}\r\nexit /b 0\r\n', 'utf8') + for (const result of [runInCmd(command, dir), runInBash(command, dir)]) { + expect(result.stdout.trim()).toBe('{}') + expect(result.status).toBe(0) + } + }) + }) + + it.skipIf(!canRunLive)( + 'still answers {} and exit 0 in both hosts when the script is gone', + () => { + // Why: compat consumers require neutral JSON even with no managed script (#14818). The + // encoded launcher did this with a Test-Path; `|| echo {}` does it with no interpreter. + withTempDir((dir, scriptPath, command) => { + expect(existsSync(scriptPath)).toBe(false) + for (const result of [runInCmd(command, dir), runInBash(command, dir)]) { + expect(result.stdout.trim()).toBe('{}') + expect(result.status).toBe(0) + } + }) + } + ) + + it.skipIf(!canRunLive)('leaves no stray `nul` file behind in the working directory', () => { + // Why this is worth a test: adding `2>nul` to silence the missing-script line looks like + // tidy-up, but under MSYS it creates a real file named `nul` in the cwd — which is the + // user's repo. Measured on Windows 11. Keep stderr unredirected. + withTempDir((dir, _scriptPath, command) => { + runInBash(command, dir) + expect(readdirSync(dir)).not.toContain('nul') + }) + }) +}) diff --git a/src/main/agent-hooks/windows-direct-cmd-hook-command.ts b/src/main/agent-hooks/windows-direct-cmd-hook-command.ts new file mode 100644 index 00000000000..f7646f19578 --- /dev/null +++ b/src/main/agent-hooks/windows-direct-cmd-hook-command.ts @@ -0,0 +1,30 @@ +import { WINDOWS_CMD_SAFE_PATH } from './installer-utils' + +// Why: a drive-letter path only. WINDOWS_CMD_SAFE_PATH also admits a UNC profile, and +// `//server/share/...` is not a command cmd.exe reliably starts. +const WINDOWS_DRIVE_LETTER_PATH = /^[A-Za-z]:\\/ + +/** + * Shortest launcher for a managed Windows `.cmd` hook: the script path itself (#18875). + * + * The encoded PowerShell launcher spent a full interpreter start-up per hook event to reach a + * script that exits at its first `ORCA_PANE_KEY` guard, and left a stdout-holding orphan behind + * when the hook's timeout kill landed. Measurements and the EDR trade are in + * `docs/reference/windows-edr-posture.md`. + * + * Returns null when the caller must keep the encoded launcher: a path either shell would mangle. + */ +export function wrapWindowsDirectCmdHookCommand(scriptPath: string): string | null { + if (!WINDOWS_CMD_SAFE_PATH.test(scriptPath) || !WINDOWS_DRIVE_LETTER_PATH.test(scriptPath)) { + return null + } + // Why: forward slashes are the one separator both hosts read, and no token here is a switch + // MSYS can rewrite — a literal `cmd.exe /d /c ` does not survive Git Bash (measured). + const invocation = scriptPath.replaceAll('\\', '/') + // Why: neutral JSON when the script is missing (#14818), with no interpreter to Test-Path with. + // Valid in bash and cmd.exe; PowerShell 5.1 rejects `||`, which is what gates this on Git Bash. + // It also fires when cmd.exe itself exits non-zero (a failing AutoRun), printing `{}` twice — + // on that same box the encoded launcher exited 1 instead, so neither shape is clean there. + // Stderr stays unredirected: `2>nul` writes a literal `nul` file into the cwd under MSYS. + return `${invocation} || echo {}` +} diff --git a/src/main/agent-hooks/windows-hook-payload-delivery.test.ts b/src/main/agent-hooks/windows-hook-payload-delivery.test.ts index 22103b178ff..79d9f4f60ae 100644 --- a/src/main/agent-hooks/windows-hook-payload-delivery.test.ts +++ b/src/main/agent-hooks/windows-hook-payload-delivery.test.ts @@ -7,7 +7,7 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import { spawn } from 'node:child_process' import { createServer, type Server } from 'node:http' -import { mkdtempSync, readFileSync } from 'node:fs' +import { mkdtempSync, readFileSync, writeFileSync } from 'node:fs' import { removeTreeSync } from '../../shared/windows-transient-lock-removal' import { tmpdir } from 'node:os' import { join } from 'node:path' @@ -32,6 +32,7 @@ vi.mock('os', async (importOriginal) => { }) import { ClaudeHookService } from '../claude/hook-service' +import { WINDOWS_CMD_SAFE_PATH } from './installer-utils' import { getConfigPath, getWindowsManagedLifecycleHook } from '../claude/hook-settings' import { findGitBash } from './windows-git-bash-path.test-fixture' @@ -122,6 +123,15 @@ function runHookCommand( }) } +// Why: a developer box may set HKCU\...\Command Processor\AutoRun, which cmd.exe runs before +// any .cmd — and MSYS spawns a .cmd without `/d`, so it fires on the Git Bash leg. Redirecting +// USERPROFILE to a temp home makes the usual `%USERPROFILE%\.cmd_aliases.cmd` target vanish, and +// cmd's "not recognized" lands on the hook's stderr. Seed an empty target so this suite measures +// the launcher rather than the host's shell configuration. +function seedCmdAutoRunTarget(home: string): void { + writeFileSync(join(home, '.cmd_aliases.cmd'), '@echo off\r\n', 'utf8') +} + function hookEnvironment(extra: NodeJS.ProcessEnv): NodeJS.ProcessEnv { const base = Object.fromEntries( Object.entries(process.env).filter(([key]) => !key.startsWith('ORCA_')) @@ -158,6 +168,7 @@ describe.skipIf(process.platform !== 'win32')('Windows managed hook payload deli it('delivers the piped payload to the hook listener through cmd.exe and Git Bash', async () => { home = mkdtempSync(join(tmpdir(), 'orca-hook-payload-')) homedirMock.mockReturnValue(home) + seedCmdAutoRunTarget(home) expect(new ClaudeHookService().install().state).toBe('installed') const settings = JSON.parse(readFileSync(getConfigPath(), 'utf8')) as { @@ -166,6 +177,11 @@ describe.skipIf(process.platform !== 'win32')('Windows managed hook payload deli // Why: assert nothing about the launcher's shape here — this test's whole value is // that it fails for any launcher that loses the payload, named conhost or not. const registeredCommand = settings.hooks.PreToolUse[0].hooks[0].command + // ...with one exception: a cmd-safe profile must reach the script with no interpreter in + // front of it, or #18875's per-event PowerShell start-up has quietly come back. + if (WINDOWS_CMD_SAFE_PATH.test(join(home, '.orca', 'agent-hooks', 'claude-hook.cmd'))) { + expect(registeredCommand).not.toMatch(/powershell|-EncodedCommand/i) + } const listener = await startHookListener() server = listener.server diff --git a/src/main/agent-hooks/windows-powershell-hook-launcher.ts b/src/main/agent-hooks/windows-powershell-hook-launcher.ts index b9a9f6dd208..2cdb8c0f3fa 100644 --- a/src/main/agent-hooks/windows-powershell-hook-launcher.ts +++ b/src/main/agent-hooks/windows-powershell-hook-launcher.ts @@ -39,6 +39,11 @@ export function getWindowsPowerShellExecutablePath(): string { * Do not restore the flag to fix a console report. That trades every hook on an * AV host for a flicker. The answer is to shorten the interpreter chain — the * shipped doctrine of #15520 and #15595 — or a launcher that owns no console. + * + * #18875 took that answer for the Claude lifecycle hook, which now registers the + * managed `.cmd` path directly (`windows-direct-cmd-hook-command.ts`) and reaches + * this launcher only when the profile path is not cmd-safe or Git Bash is not + * resolvable. Every other caller still comes through here on every event. */ export const WINDOWS_POWERSHELL_HOOK_SWITCHES = '-NoProfile' diff --git a/src/main/claude/hook-service.test.ts b/src/main/claude/hook-service.test.ts index e2937015bea..a4e48c98120 100644 --- a/src/main/claude/hook-service.test.ts +++ b/src/main/claude/hook-service.test.ts @@ -7,6 +7,7 @@ import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync import { tmpdir } from 'node:os' import { join } from 'node:path' import { vi, describe, expect, it } from 'vitest' +import type * as GitBashModule from '../git-bash' vi.mock('electron', () => ({ app: { @@ -14,11 +15,23 @@ vi.mock('electron', () => ({ } })) +// Why: the installed hook shape depends on whether Git Bash is resolvable on the host, so the +// install assertions below have to state which host they describe rather than inherit the box's. +const { gitBashAvailableMock } = vi.hoisted(() => ({ gitBashAvailableMock: { value: true } })) +vi.mock('../git-bash', async (importOriginal) => ({ + ...(await importOriginal()), + isGitBashAvailable: () => gitBashAvailableMock.value +})) + import type { SFTPWrapper } from 'ssh2' -import { createManagedCommandMatcher } from '../agent-hooks/installer-utils' +import { createManagedCommandMatcher, WINDOWS_CMD_SAFE_PATH } from '../agent-hooks/installer-utils' import { WINDOWS_HOOK_STDIN_DRAIN_LABEL } from '../agent-hooks/hook-stdin-contract' import { ClaudeHookService } from './hook-service' -import { getWindowsManagedLifecycleHook, OPENCLAUDE_HOOK_SETTINGS } from './hook-settings' +import { + CLAUDE_EVENTS, + getWindowsManagedLifecycleHook, + OPENCLAUDE_HOOK_SETTINGS +} from './hook-settings' const CLAUDE_SCRIPT_FILE_NAME = process.platform === 'win32' ? 'claude-hook.cmd' : 'claude-hook.sh' const STATUSLINE_SCRIPT_FILE_NAME = @@ -35,14 +48,29 @@ function hasManagedCommand(hook: TestHook, matcher: (command: string | undefined } describe('getWindowsManagedLifecycleHook', () => { - it('resolves the managed script from the runtime Windows profile, as a single command string', () => { - const scriptPath = 'C:\\Users\\%name%\\a^b&c\\.orca\\agent-hooks\\claude-hook.cmd' - const hook = getWindowsManagedLifecycleHook(scriptPath) + const SAFE_SCRIPT_PATH = 'C:\\Users\\alice\\.orca\\agent-hooks\\claude-hook.cmd' + const UNSAFE_SCRIPT_PATH = 'C:\\Users\\%name%\\a^b&c\\.orca\\agent-hooks\\claude-hook.cmd' + + it('registers the script itself, with no interpreter in front of it (#18875)', () => { + // Why this is the whole point: the encoded launcher spent a PowerShell start-up per hook + // event (471ms vs 201ms measured) before the .cmd could reach its ORCA_PANE_KEY guard, and + // its orphan outlived the hook's timeout kill still holding the stdout the agent reads. + const hook = getWindowsManagedLifecycleHook(SAFE_SCRIPT_PATH, { gitBashAvailable: true }) + + expect(hook.args).toBeUndefined() + expect(hook.command).toBe('C:/Users/alice/.orca/agent-hooks/claude-hook.cmd || echo {}') + expect(hook.command).not.toMatch(/powershell|-EncodedCommand|conhost/i) + // Why: Git Bash/MSYS mangles backslash paths and rewrites slash-prefixed switches. + expect(hook.command).not.toMatch(/\\/) + expect(hook.command).not.toMatch(/ \/[a-zA-Z]+( |$)/) + }) + + it('falls back to the encoded launcher when the profile path is not cmd-safe', () => { + const hook = getWindowsManagedLifecycleHook(UNSAFE_SCRIPT_PATH, { gitBashAvailable: true }) expect(hook.args).toBeUndefined() expect(hook.command).toMatch(/\/powershell\.exe -NoProfile -EncodedCommand /) - expect(hook.command).not.toContain(scriptPath) - // Why: Git Bash/MSYS mangles backslash paths and slash-prefixed switches. + expect(hook.command).not.toContain(UNSAFE_SCRIPT_PATH) expect(hook.command.replace(/-EncodedCommand \S+$/, '')).not.toMatch(/\\| \/[a-zA-Z]+( |$)/) const encoded = hook.command.match(/-EncodedCommand (\S+)$/)?.[1] @@ -51,10 +79,21 @@ describe('getWindowsManagedLifecycleHook', () => { expect(decoded).toContain('.orca\\agent-hooks\\claude-hook.cmd') }) + it('falls back to the encoded launcher when Git Bash is not resolvable', () => { + // Why: without Git Bash, Claude Code hosts the hook in PowerShell, and PowerShell 5.1 + // rejects `||` as a statement separator (measured) — every event would be a parse error. + const hook = getWindowsManagedLifecycleHook(SAFE_SCRIPT_PATH, { gitBashAvailable: false }) + + expect(hook.command).toMatch(/\/powershell\.exe -NoProfile -EncodedCommand /) + }) + it('is still recognized as managed by createManagedCommandMatcher (#14825)', () => { - const scriptPath = 'C:\\Users\\alice\\.orca\\agent-hooks\\claude-hook.cmd' - const hook = getWindowsManagedLifecycleHook(scriptPath) - expect(isClaudeManagedCommand(hook.command)).toBe(true) + for (const hook of [ + getWindowsManagedLifecycleHook(SAFE_SCRIPT_PATH, { gitBashAvailable: true }), + getWindowsManagedLifecycleHook(SAFE_SCRIPT_PATH, { gitBashAvailable: false }) + ]) { + expect(isClaudeManagedCommand(hook.command)).toBe(true) + } }) }) @@ -200,7 +239,13 @@ describe('ClaudeHookService.install', () => { const managedHook = legacyHooks.find((hook: TestHook) => hasManagedCommand(hook, isClaudeManagedCommand) ) - expect(JSON.stringify(managedHook)).not.toContain(tmpHome.replaceAll('\\', '/')) + // Why: POSIX resolves the profile at runtime (`${HOME-}`, STA-3348). Windows cannot — + // no single token expands in both Git Bash and cmd.exe — so it registers the absolute + // path, as Codex/Grok/Devin/Antigravity already do (#18875). A moved profile is caught + // by getStatus's exact match and rewritten, and `|| echo {}` keeps a stale entry neutral. + if (process.platform !== 'win32') { + expect(JSON.stringify(managedHook)).not.toContain(tmpHome.replaceAll('\\', '/')) + } expect( legacyHooks.some((hook: TestHook) => hasManagedCommand(hook, isClaudeManagedCommand)) ).toBe(true) @@ -365,7 +410,7 @@ describe('ClaudeHookService.install', () => { }) it.skipIf(process.platform !== 'win32')( - 'runs portable managed hooks through a single headless command string', + 'pins the encoded-launcher fallback for a profile path the shells cannot carry bare', () => { const tmpHome = mkdtempSync(join(tmpdir(), 'orca claude home with spaces ')) vi.stubEnv('HOME', tmpHome) @@ -397,6 +442,137 @@ describe('ClaudeHookService.install', () => { } ) + it.skipIf(process.platform !== 'win32')( + 'installs the bare script path on every event when the profile path is cmd-safe (#18875)', + () => { + const tmpHome = mkdtempSync(join(tmpdir(), 'orca-claude-direct-')) + vi.stubEnv('HOME', tmpHome) + vi.stubEnv('USERPROFILE', tmpHome) + const scriptPath = join(tmpHome, '.orca', 'agent-hooks', CLAUDE_SCRIPT_FILE_NAME) + // Why: a runner whose tmpdir carries a space (a profile-scoped TEMP) belongs to the + // fallback case above, not this one; skip rather than assert the wrong contract. + if (!WINDOWS_CMD_SAFE_PATH.test(scriptPath)) { + vi.unstubAllEnvs() + rmSync(tmpHome, { recursive: true, force: true }) + return + } + try { + expect(new ClaudeHookService().install().state).toBe('installed') + + const settings = JSON.parse( + readFileSync(join(tmpHome, '.claude', 'settings.json'), 'utf-8') + ) as { hooks: Record } + + const expected = `${scriptPath.replaceAll('\\', '/')} || echo {}` + for (const { eventName } of CLAUDE_EVENTS) { + const hook = settings.hooks[eventName]?.[0]?.hooks?.[0] + expect(hook?.args, eventName).toBeUndefined() + expect(hook?.command, eventName).toBe(expected) + } + // Why: the whole point of #18875 — no interpreter is started to reach the script. + expect(JSON.stringify(settings.hooks)).not.toMatch(/powershell|EncodedCommand/i) + expect(new ClaudeHookService().getStatus().state).toBe('installed') + } finally { + vi.unstubAllEnvs() + rmSync(tmpHome, { recursive: true, force: true }) + } + } + ) + + it.skipIf(process.platform !== 'win32')( + 'sweeps a previously installed encoded launcher on reinstall, keeping user hooks', + () => { + const tmpHome = mkdtempSync(join(tmpdir(), 'orca-claude-migrate-')) + vi.stubEnv('HOME', tmpHome) + vi.stubEnv('USERPROFILE', tmpHome) + const scriptPath = join(tmpHome, '.orca', 'agent-hooks', CLAUDE_SCRIPT_FILE_NAME) + if (!WINDOWS_CMD_SAFE_PATH.test(scriptPath)) { + vi.unstubAllEnvs() + rmSync(tmpHome, { recursive: true, force: true }) + return + } + try { + const settingsPath = join(tmpHome, '.claude', 'settings.json') + mkdirSync(join(tmpHome, '.claude'), { recursive: true }) + const stale = getWindowsManagedLifecycleHook(scriptPath, { gitBashAvailable: false }) + writeFileSync( + settingsPath, + JSON.stringify({ + hooks: { + Stop: [{ hooks: [stale] }], + PreToolUse: [{ matcher: '*', hooks: [stale] }], + UserPromptSubmit: [{ hooks: [{ type: 'command', command: 'echo mine' }] }] + } + }), + 'utf-8' + ) + + expect(new ClaudeHookService().install().state).toBe('installed') + + const settings = JSON.parse(readFileSync(settingsPath, 'utf-8')) as { + hooks: Record + } + expect(JSON.stringify(settings.hooks)).not.toContain('-EncodedCommand') + expect( + settings.hooks.UserPromptSubmit.some((definition) => + definition.hooks.some((hook) => hook.command === 'echo mine') + ) + ).toBe(true) + } finally { + vi.unstubAllEnvs() + rmSync(tmpHome, { recursive: true, force: true }) + } + } + ) + + it.skipIf(process.platform !== 'win32')( + 'reports a stale absolute path as not_installed and rewrites it on install (#18875)', + () => { + // Why: the direct shape bakes the profile path in, where the encoded launcher resolved + // %USERPROFILE% at run time (STA-3348). That is only safe because a moved profile is + // caught here and rewritten, so this is the test that carries the replaced contract. + const tmpHome = mkdtempSync(join(tmpdir(), 'orca-claude-moved-')) + vi.stubEnv('HOME', tmpHome) + vi.stubEnv('USERPROFILE', tmpHome) + const scriptPath = join(tmpHome, '.orca', 'agent-hooks', CLAUDE_SCRIPT_FILE_NAME) + if (!WINDOWS_CMD_SAFE_PATH.test(scriptPath)) { + vi.unstubAllEnvs() + rmSync(tmpHome, { recursive: true, force: true }) + return + } + try { + const settingsPath = join(tmpHome, '.claude', 'settings.json') + mkdirSync(join(tmpHome, '.claude'), { recursive: true }) + const staleCommand = 'C:/Users/someone-else/.orca/agent-hooks/claude-hook.cmd || echo {}' + const stale = { type: 'command', command: staleCommand, timeout: 10 } + writeFileSync( + settingsPath, + JSON.stringify({ + hooks: Object.fromEntries( + CLAUDE_EVENTS.map(({ eventName }) => [eventName, [{ hooks: [stale] }]]) + ) + }), + 'utf-8' + ) + + expect(new ClaudeHookService().getStatus().state).toBe('not_installed') + expect(new ClaudeHookService().install().state).toBe('installed') + + const settings = JSON.parse(readFileSync(settingsPath, 'utf-8')) as { + hooks: Record + } + expect(JSON.stringify(settings.hooks)).not.toContain('someone-else') + expect(settings.hooks.PreToolUse[0].hooks[0].command).toBe( + `${scriptPath.replaceAll('\\', '/')} || echo {}` + ) + expect(new ClaudeHookService().getStatus().state).toBe('installed') + } finally { + vi.unstubAllEnvs() + rmSync(tmpHome, { recursive: true, force: true }) + } + } + ) + it.skipIf(process.platform !== 'win32')( 'posts from the managed .cmd via curl.exe, not a second PowerShell', () => { diff --git a/src/main/claude/hook-settings.ts b/src/main/claude/hook-settings.ts index c6cf3a9b53c..047fcbb26b6 100644 --- a/src/main/claude/hook-settings.ts +++ b/src/main/claude/hook-settings.ts @@ -14,23 +14,25 @@ import { type HooksConfig } from '../agent-hooks/installer-utils' import { wrapRuntimeHomeHookCommand } from '../agent-hooks/runtime-home-hook-command' +import { wrapWindowsDirectCmdHookCommand } from '../agent-hooks/windows-direct-cmd-hook-command' +import { isGitBashAvailable } from '../git-bash' export type ClaudeCompatibleHookSettings = { configDirName: '.claude' | '.openclaude' scriptBaseName: 'claude-hook' | 'openclaude-hook' - usesWindowsPowerShellLauncher: boolean + usesWindowsCompatLauncher: boolean } export const CLAUDE_HOOK_SETTINGS: ClaudeCompatibleHookSettings = { configDirName: '.claude', scriptBaseName: 'claude-hook', - usesWindowsPowerShellLauncher: true + usesWindowsCompatLauncher: true } export const OPENCLAUDE_HOOK_SETTINGS: ClaudeCompatibleHookSettings = { configDirName: '.openclaude', scriptBaseName: 'openclaude-hook', - usesWindowsPowerShellLauncher: false + usesWindowsCompatLauncher: false } export const CLAUDE_EVENTS = [ @@ -153,16 +155,31 @@ export function getManagedCommand( export function getManagedLifecycleHook( scriptPath: string, - settings = CLAUDE_HOOK_SETTINGS + settings = CLAUDE_HOOK_SETTINGS, + options: WindowsManagedLifecycleHookOptions = {} ): HookCommandConfig { - if (process.platform !== 'win32' || !settings.usesWindowsPowerShellLauncher) { + if (process.platform !== 'win32' || !settings.usesWindowsCompatLauncher) { return buildManagedCommandHook(getManagedCommand(scriptPath, { neutralJsonWhenMissing: true })) } - return getWindowsManagedLifecycleHook(scriptPath) + return getWindowsManagedLifecycleHook(scriptPath, options) } +export type WindowsManagedLifecycleHookOptions = { gitBashAvailable?: boolean } + // Why: some Claude-compatible consumers ignore `args`, so the invocation must be self-contained. -export function getWindowsManagedLifecycleHook(scriptPath: string): HookCommandConfig { +export function getWindowsManagedLifecycleHook( + scriptPath: string, + options: WindowsManagedLifecycleHookOptions = {} +): HookCommandConfig { + // Why (#18875): the encoded launcher cost a PowerShell start-up per hook event. Take the direct + // path only where the host can parse `||` — Git Bash can, Windows PowerShell 5.1 cannot. + const directCommand = + (options.gitBashAvailable ?? isGitBashAvailable()) + ? wrapWindowsDirectCmdHookCommand(scriptPath) + : null + if (directCommand) { + return { type: 'command', command: directCommand, timeout: MANAGED_HOOK_TIMEOUT_SECONDS } + } const scriptFileName = win32.basename(scriptPath) // Why: runtime profile resolution keeps the managed entry portable across users (STA-3348). const quotedRelativePath = quotePowerShellString(`.orca\\agent-hooks\\${scriptFileName}`)