From 7bec98466bb314aa213c7ed7302e40451ea304dd Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 15:04:52 -0700 Subject: [PATCH 01/23] test: canonicalize setup fixture paths before worktree lookup (#18912) --- tests/e2e/setup-script-import.spec.ts | 4 ++-- ...script-prompt-unreadable-orca-yaml.spec.ts | 19 ++++++++----------- 2 files changed, 10 insertions(+), 13 deletions(-) diff --git a/tests/e2e/setup-script-import.spec.ts b/tests/e2e/setup-script-import.spec.ts index 180b34f85aa..340258c884b 100644 --- a/tests/e2e/setup-script-import.spec.ts +++ b/tests/e2e/setup-script-import.spec.ts @@ -1,5 +1,5 @@ import { execFileSync } from 'node:child_process' -import { mkdirSync, rmSync, writeFileSync } from 'node:fs' +import { mkdirSync, realpathSync, rmSync, writeFileSync } from 'node:fs' import path from 'node:path' import type { Locator, Page } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' @@ -108,7 +108,7 @@ async function addAndActivateRepo(page: Page, repoPath: string): Promise state.setActiveWorktree(worktree.id) state.setSidebarOpen(true) return addedRepo.id - }, repoPath) + }, realpathSync.native(repoPath)) } async function openRepoSettings(page: Page, repoId: string): Promise { diff --git a/tests/e2e/setup-script-prompt-unreadable-orca-yaml.spec.ts b/tests/e2e/setup-script-prompt-unreadable-orca-yaml.spec.ts index 151a4b02bbc..199694b961a 100644 --- a/tests/e2e/setup-script-prompt-unreadable-orca-yaml.spec.ts +++ b/tests/e2e/setup-script-prompt-unreadable-orca-yaml.spec.ts @@ -1,5 +1,5 @@ import { execFileSync } from 'node:child_process' -import { mkdirSync, rmSync, writeFileSync } from 'node:fs' +import { mkdirSync, realpathSync, rmSync, writeFileSync } from 'node:fs' import path from 'node:path' import type { ElectronApplication, Page } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' @@ -112,17 +112,10 @@ async function addRepoAndActivateMainWorktree( if (!store) { throw new Error('window.__store is not available') } - const normalize = (value: string): string => - value.startsWith('/private/var/') ? value.slice('/private'.length) : value - const state = store.getState() const worktrees = state.worktreesByRepo[targetRepoId] ?? [] - const mainWorktree = worktrees.find( - (entry) => normalize(entry.path) === normalize(targetRepoPath) - ) - const featureWorktree = worktrees.find( - (entry) => normalize(entry.path) === normalize(targetFeaturePath) - ) + const mainWorktree = worktrees.find((entry) => entry.path === targetRepoPath) + const featureWorktree = worktrees.find((entry) => entry.path === targetFeaturePath) if (!mainWorktree || !featureWorktree) { throw new Error( `Missing worktrees for ${targetRepoPath}: ${worktrees.map((entry) => entry.path).join(', ')}` @@ -145,7 +138,11 @@ async function addRepoAndActivateMainWorktree( featureWorktreeId: featureWorktree.id } }, - { targetRepoId: repoId, targetRepoPath: repoPath, targetFeaturePath: featureWorktreePath } + { + targetRepoId: repoId, + targetRepoPath: realpathSync.native(repoPath), + targetFeaturePath: realpathSync.native(featureWorktreePath) + } ) } From d7767fb1960507a7ae3fc47d5858b06f8887bd6b Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 15:08:22 -0700 Subject: [PATCH 02/23] perf(worktree): remove redundant creation and terminal startup work (#18793) * perf(worktree): remove redundant creation and terminal startup work * test(worktree): cover optimized creation call signatures Preserve explicit branch adoption, WSL callback routing and sparse cleanup expectations. * perf: preserve user Git checkout worker settings * perf(git): skip malformed remote base probes * perf(cli): avoid loading other agent hooks for Codex preflight * fix(build): retain Codex preflight entry for packaged CLI * test(ssh): wait for replacement PTY before lease recovery input * test(ssh): verify recovered shell execution and lease ownership * test(electron): reap isolated macOS crash reporters on teardown * test: allow either observed self-exit snapshot ordering * test: capture frozen-host input recovery evidence --- config/reliability-gates.jsonc | 96 ++++++++++++++++++ electron.vite.config.ts | 3 + src/cli/handlers/agent-hooks.test.ts | 5 +- src/cli/handlers/agent-hooks.ts | 10 +- .../claude-stream-json-connection.test.ts | 11 ++- src/main/git/repo-branch-conflict.test.ts | 74 +++++++++++++- src/main/git/repo-branch-conflict.ts | 34 +++++-- src/main/git/runner-wsl-direct-read.test.ts | 28 ++++++ .../git/worktree-add-creation-config.test.ts | 4 +- .../worktree-add-local-base-refresh.test.ts | 5 +- ...worktree-add-local-base-suggestion.test.ts | 5 +- src/main/git/worktree-add.ts | 15 ++- ...rktree-create-preparation-real-wsl.test.ts | 77 +++++++++++++++ src/main/git/worktree-create-preparation.ts | 41 +++++--- .../git/worktree-preparation-base-oid.test.ts | 98 +++++++++++++++++++ src/main/ipc/worktree-logic-wsl.test.ts | 21 ++++ src/main/ipc/worktree-logic.ts | 9 +- src/main/ipc/worktree-remote.ts | 42 +++++--- .../ipc/worktrees-local-create-flow.test.ts | 36 +++++-- src/main/ipc/worktrees-test-module-mocks.ts | 23 +++-- src/main/ipc/worktrees-test-runtime-stub.ts | 2 + .../local-pty-provider-spawn-session.test.ts | 7 +- src/main/providers/local-pty-spawn-state.ts | 2 + src/main/runtime/fetch-remote-cache.test.ts | 13 ++- ...orca-runtime-refresh-repo-worktree-scan.ts | 5 + .../local-worktree-creation-part-02.spec.ts | 12 ++- .../local-worktree-creation.spec.ts | 6 +- ...orktree-removal-and-reconciliation.spec.ts | 13 ++- ...runtime-local-worktree-create-candidate.ts | 26 +++-- .../runtime-remote-fetch-controller.ts | 2 +- ...rktree-scan-admin-fingerprint-gate.test.ts | 16 +++ .../terminal-pane/ipc-pty-connect-result.ts | 4 + ...tion-deferred-reattach-live-output.test.ts | 35 +++++++ .../pty-connection/apply-reattach-payload.ts | 7 ++ .../pty-transport-connect-spawn.test.ts | 18 ++++ .../terminal-pane-manager-options.ts | 6 ++ .../lib/pane-manager/pane-lifecycle.test.ts | 28 +++++- .../src/lib/pane-manager/pane-lifecycle.ts | 6 +- .../pane-manager-pane-creation.ts | 2 +- .../lib/pane-manager/pane-manager-types.ts | 1 + .../src/lib/pane-manager/pane-split-close.ts | 2 +- src/shared/git-binary-compatibility.test.ts | 27 +++++ .../e2e/helpers/electron-crashpad-cleanup.ts | 46 +++++++++ .../electron-crashpad-cleanup.unit.test.ts | 45 +++++++++ .../e2e/helpers/electron-process-shutdown.ts | 2 + .../helpers/ssh-recovery-input-observation.ts | 53 ++++++++++ ...ssh-docker-transport-drop-recovery.spec.ts | 62 +++++++++--- 47 files changed, 971 insertions(+), 114 deletions(-) create mode 100644 src/main/git/worktree-create-preparation-real-wsl.test.ts create mode 100644 src/main/git/worktree-preparation-base-oid.test.ts create mode 100644 tests/e2e/helpers/electron-crashpad-cleanup.ts create mode 100644 tests/e2e/helpers/electron-crashpad-cleanup.unit.test.ts create mode 100644 tests/e2e/helpers/ssh-recovery-input-observation.ts diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc index 2dd39ad7c3f..d48a239354f 100644 --- a/config/reliability-gates.jsonc +++ b/config/reliability-gates.jsonc @@ -10,6 +10,102 @@ } }, "gates": [ + { + "id": "terminal-output.prestarted-shell-snapshot-adoption", + "title": "Prestarted shell adoption paints covered output once", + "maturity": "experimental", + "protection": "partial", + "owner": "terminal-runtime", + "layer": "renderer-transport-and-live-electron", + "surfaces": [ + "backend-created first terminal", + "daemon snapshot adoption", + "deferred live output" + ], + "platforms": ["macos", "linux", "windows"], + "providers": ["local", "daemon", "wsl", "ssh", "remote-runtime"], + "coveredPlatforms": ["macos", "linux", "windows"], + "coveredProviders": ["local", "daemon", "wsl"], + "coverageNotes": "macOS daemon-backed Electron journey verifies same PID and terminal identity plus rendered output. Focused renderer contracts pass on Linux, Windows and WSL. Neighboring SSH model and replay contracts pass locally; no new live SSH or paired-runtime journey.", + "motivatingLinks": [ + "https://github.com/user-attachments/assets/e8c6d1dc-6150-4c3d-b55a-3d12efefdd04", + "https://github.com/user-attachments/assets/b0328f88-34ac-4d51-8119-9efe17072435" + ], + "invariant": "Adopting a prestarted terminal preserves its existing process and paints snapshot-covered startup output once while retaining subsequent live output. Missing sequence proof or blank snapshots must not authorize dropping output.", + "oracle": "Pass snapshot sequence and proven zero keyboard flags through real IPC transport projection. Deliver snapshot-covered and newer output before reattach resolves; drain replay parse callbacks and require one startup marker and the newer output. Repeat with no sequence and blank snapshot to retain unproven bytes. In Electron select a prestarted workspace, type a generated marker and compare PID and stable terminal identities before and after.", + "commands": [ + "pnpm test src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts", + "pnpm test src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts src/renderer/src/components/terminal-pane/pty-connection-hidden-snapshot-live-overlap.test.ts src/renderer/src/components/terminal-pane/pty-connection-replay-payload-handling.test.ts src/renderer/src/components/terminal-pane/pty-connection/reattach-payload-ssh-reconnect-model-paint.test.ts", + "pnpm test src/renderer/src/components/terminal-pane/pty-connection src/renderer/src/components/terminal-pane/pty-transport" + ], + "testFiles": [ + "src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts", + "src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts" + ], + "assertionRefs": [ + { + "file": "src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts", + "assertions": [ + "zero and nonzero snapshot sequence and proven zero keyboard flags survive IPC projection" + ] + }, + { + "file": "src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts", + "assertions": [ + "startup output covered by the snapshot is painted once", + "new output remains visible", + "legacy unsequenced and blank snapshots retain bytes" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-04", + "runner": "local", + "platform": "macos", + "command": "pnpm test src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts src/renderer/src/components/terminal-pane/pty-connection-hidden-snapshot-live-overlap.test.ts src/renderer/src/components/terminal-pane/pty-connection-replay-payload-handling.test.ts src/renderer/src/components/terminal-pane/pty-connection/reattach-payload-ssh-reconnect-model-paint.test.ts", + "result": "passed", + "durationSeconds": 2.34, + "summary": "5 suites / 48 tests pass. Focused 2-suite runs independently pass 27 tests on Linux, Windows and WSL." + }, + { + "date": "2026-09-04", + "runner": "local", + "platform": "macos", + "command": "pnpm test src/renderer/src/components/terminal-pane/pty-connection src/renderer/src/components/terminal-pane/pty-transport", + "result": "passed", + "durationSeconds": 5.67, + "summary": "Broader connection/transport gate: 77 files and 813 tests passed, including neighboring restore, reconnect, input and replay behavior. Log: artifacts/worktree-create/orca-draft-replay-broader-gate.log." + } + ], + "runtimeBudget": { + "p95Seconds": 15, + "scope": "focused renderer transport and deferred-adoption contracts" + }, + "flakeHistory": { + "status": "unknown", + "evidence": "Focused local and remote runs pass; no CI soak history." + }, + "redGreenEvidence": { + "status": "partial", + "evidence": "Metadata tests fail before forwarding. Corrected parse-draining regression observes two startup markers when the baseline installation is removed, and one after restoration. Initial missing-live-output failure was a harness parse-drain omission and is not red proof. Before/fixed Electron screenshots show duplicate/single startup output." + }, + "performanceBudget": { + "required": true, + "evidence": "Reuses existing snapshot baseline reconciliation with no new scan, timer or subprocess. Corrected daemon-backed rendered trial reaches replay at 116.6 ms and generated keyboard output at 177 ms after selecting the prestarted workspace. This measures selection/adoption, not ordinary composer creation." + }, + "promotionCriteria": [ + "Meet manifest CI and soak policy.", + "Retain intentional-break and rendered identity/output proof.", + "Exercise live SSH and paired-runtime snapshot adoption before claiming full provider coverage." + ], + "knownGaps": [ + "Composer draft creation and cancellation are not implemented by this gate.", + "No new live SSH, Windows or WSL UI run; remote evidence is focused contract tests.", + "Mixed-version snapshots without sequence proof intentionally retain legacy behavior." + ], + "demotionRule": "Keep experimental or demote if adoption duplicates covered output, drops newer or unproven output, changes terminal ownership, or flakes without explanation." + }, { "id": "cmd-j-tabs.host-qualified-candidate-ownership", "title": "Cmd-J tab candidates retain execution-host ownership", diff --git a/electron.vite.config.ts b/electron.vite.config.ts index 4ed4641cde1..90dc637c204 100644 --- a/electron.vite.config.ts +++ b/electron.vite.config.ts @@ -253,6 +253,9 @@ export const electronViteConfig: UserConfig = { 'agent-hooks/managed-agent-hook-controls': resolve( 'src/main/agent-hooks/managed-agent-hook-controls.ts' ), + 'codex/managed-home-shell-preflight': resolve( + 'src/main/codex/managed-home-shell-preflight.ts' + ), // Why: account import mutates the user's macOS Keychain from the CLI. 'claude-accounts/keychain': resolve('src/main/claude-accounts/keychain.ts') }, diff --git a/src/cli/handlers/agent-hooks.test.ts b/src/cli/handlers/agent-hooks.test.ts index 4fcc186b0d8..279a8900bec 100644 --- a/src/cli/handlers/agent-hooks.test.ts +++ b/src/cli/handlers/agent-hooks.test.ts @@ -56,7 +56,10 @@ vi.mock('../runtime-client', () => { vi.mock('../../main/agent-hooks/managed-agent-hook-controls', () => ({ applyAgentStatusHooksEnabled: applyAgentStatusHooksEnabledMock, - getManagedAgentHookStatuses: getManagedAgentHookStatusesMock, + getManagedAgentHookStatuses: getManagedAgentHookStatusesMock +})) + +vi.mock('../../main/codex/managed-home-shell-preflight', () => ({ prepareManagedCodexHomeBeforeShellLaunch: prepareManagedCodexHomeBeforeShellLaunchMock })) diff --git a/src/cli/handlers/agent-hooks.ts b/src/cli/handlers/agent-hooks.ts index bf44211b4a7..4fcfe64f9b7 100644 --- a/src/cli/handlers/agent-hooks.ts +++ b/src/cli/handlers/agent-hooks.ts @@ -15,11 +15,7 @@ import { getDefaultPersistedState } from '../../shared/constants' import { normalizeDisabledTuiAgents } from '../../shared/tui-agent-selection' import type { GlobalSettings } from '../../shared/global-settings-types' import type { PersistedState } from '../../shared/persisted-state-types' -import { - applyAgentStatusHooksEnabled, - getManagedAgentHookStatuses, - prepareManagedCodexHomeBeforeShellLaunch -} from '../../main/agent-hooks/managed-agent-hook-controls' +import { prepareManagedCodexHomeBeforeShellLaunch } from '../../main/codex/managed-home-shell-preflight' type AgentHookCommandResult = { enabled: boolean @@ -194,6 +190,8 @@ async function setAgentHooksEnabled( client: RuntimeClient, enabled: boolean ): Promise { + const { applyAgentStatusHooksEnabled, getManagedAgentHookStatuses } = + await import('../../main/agent-hooks/managed-agent-hook-controls.js') const updatedRuntime = await updateRunningRuntime(client, enabled) const offlineUpdate = updatedRuntime ? null : updateEnabledOnDisk(enabled) const settingsPath = offlineUpdate?.settingsPath ?? getDataPath() @@ -234,6 +232,8 @@ export const AGENT_HOOK_HANDLERS: Record = { }) }, 'agent hooks status': async ({ json }) => { + const { getManagedAgentHookStatuses } = + await import('../../main/agent-hooks/managed-agent-hook-controls.js') const result: AgentHookCommandResult = { enabled: readHookSettingsFromDisk().agentStatusHooksEnabled, settingsPath: getDataPath(), diff --git a/src/main/claude/claude-stream-json-connection.test.ts b/src/main/claude/claude-stream-json-connection.test.ts index c4f1f9a6fca..eb69a66a897 100644 --- a/src/main/claude/claude-stream-json-connection.test.ts +++ b/src/main/claude/claude-stream-json-connection.test.ts @@ -566,7 +566,7 @@ describe('Claude stream-json connection', () => { ) }) - it('reports a self-exit with its status and stderr, and leaves its tree unverifiable', async () => { + it('reports a self-exit with its status, stderr, and observed tree verdict', async () => { const scenario = scriptScenario([{ stderr: 'claude: not signed in\n' }, { exit: 1 }]) let exit: Error | null = null const connection = await open(launchFor(scenario), { @@ -579,10 +579,11 @@ describe('Claude stream-json connection', () => { // The status and stderr are the only diagnostic a refused start leaves behind. expect((exit as unknown as Error).message).toMatch(/exited \(code 1\): claude: not signed in/) expect(connection.closed).toBe(true) - // The root's exit is first-hand, but it left before a descendant snapshot - // could be armed, so close() has no tree proof to offer and says so. - await expect(connection.close()).resolves.toBe(false) - expect(connection.exitVerdict).toEqual({ root: 'exited', tree: 'unverifiable' }) + // Stderr-triggered capture can win or lose the race with this real child's exit. + const closed = await connection.close() + expect(connection.exitVerdict.root).toBe('exited') + expect(['exited', 'unverifiable']).toContain(connection.exitVerdict.tree) + expect(closed).toBe(connection.exitVerdict.tree === 'exited') }) it.runIf(process.platform !== 'win32')( diff --git a/src/main/git/repo-branch-conflict.test.ts b/src/main/git/repo-branch-conflict.test.ts index 873c8bd3340..82bd1e65c9d 100644 --- a/src/main/git/repo-branch-conflict.test.ts +++ b/src/main/git/repo-branch-conflict.test.ts @@ -21,7 +21,7 @@ describe('getBranchConflictKindViaExec', () => { await expect(getBranchConflictKindViaExec(exec, 'feature/fix')).resolves.toBe('remote') expect(calls).toEqual([ - ['rev-parse', '--verify', 'refs/heads/feature/fix'], + ['rev-parse', '--verify', '--quiet', 'refs/heads/feature/fix'], ['remote'], ['show-ref', '--verify', '--quiet', '--', 'refs/remotes/foo/bar/feature/fix'], ['show-ref', '--verify', '--quiet', '--', 'refs/remotes/origin/feature/fix'] @@ -41,7 +41,10 @@ describe('getBranchConflictKindViaExec', () => { await expect( getBranchConflictKindViaExec(exec, 'feature/fix', 'origin/feature/fix') ).resolves.toBeNull() - expect(calls).toEqual([['rev-parse', '--verify', 'refs/heads/feature/fix'], ['remote']]) + expect(calls).toEqual([ + ['rev-parse', '--verify', '--quiet', 'refs/heads/feature/fix'], + ['remote'] + ]) }) it('keeps longest configured remote-name matching semantics', async () => { @@ -176,7 +179,7 @@ describe('getBranchConflictKindViaExec batched remote probe', () => { getBranchConflictKindViaExec(exec, 'feature', undefined, {}, batched) ).resolves.toBeNull() expect(calls).toEqual([ - ['rev-parse', '--verify', 'refs/heads/feature'], + ['rev-parse', '--verify', '--quiet', 'refs/heads/feature'], ['remote'], ['cat-file', '--batch-check'] ]) @@ -254,3 +257,68 @@ describe('getBranchConflictKindViaExec batched remote probe', () => { expect(calls.filter((argv) => argv[0] === 'show-ref')).toHaveLength(3) }) }) + +describe('branch conflict with existing-branch adoption', () => { + const absent = () => Object.assign(new Error('missing'), { code: 1, stderr: '' }) + + it('skips adoption and its commit probe for a proven missing local ref', async () => { + const exec = vi.fn(async (argv: string[]) => { + if (argv[0] === 'rev-parse') { + throw absent() + } + return { stdout: '' } + }) + const adopt = vi.fn(async () => false) + await expect( + getBranchConflictKindViaExec(exec, 'new', undefined, {}, undefined, adopt) + ).resolves.toBeNull() + expect(adopt).not.toHaveBeenCalled() + expect(exec).toHaveBeenCalledTimes(2) + }) + + it('allows an existing branch without querying remote refs', async () => { + const exec = vi.fn(async () => ({ stdout: 'a'.repeat(40) })) + const adopt = vi.fn(async () => true) + await expect( + getBranchConflictKindViaExec(exec, 'existing', undefined, {}, undefined, adopt) + ).resolves.toBeNull() + expect(adopt).toHaveBeenCalledOnce() + expect(exec).toHaveBeenCalledOnce() + }) + + it('retains conflicts for refs whose objects cannot be adopted as commits', async () => { + const exec = vi.fn(async () => ({ stdout: 'a'.repeat(40) })) + const adopt = vi.fn(async () => false) + await expect( + getBranchConflictKindViaExec(exec, 'dangling', undefined, {}, undefined, adopt) + ).resolves.toBe('local') + expect(adopt).toHaveBeenCalledOnce() + expect(exec).toHaveBeenCalledTimes(2) + }) + + it.each([ + Object.assign(new Error('transport'), { code: 1, stderr: 'transport failed' }), + Object.assign(new Error('timeout'), { code: 'ETIMEDOUT' }) + ])('still attempts adoption after an undecided ref probe: %s', async (error) => { + const exec = vi.fn(async () => { + throw error + }) + const adopt = vi.fn(async () => true) + await expect( + getBranchConflictKindViaExec(exec, 'existing', undefined, {}, undefined, adopt) + ).resolves.toBeNull() + expect(adopt).toHaveBeenCalledOnce() + }) + + it('rechecks a ref that disappeared while adoption was running', async () => { + const exec = vi + .fn() + .mockResolvedValueOnce({ stdout: 'a'.repeat(40) }) + .mockRejectedValueOnce(absent()) + .mockResolvedValueOnce({ stdout: '' }) + await expect( + getBranchConflictKindViaExec(exec, 'removed', undefined, {}, undefined, async () => false) + ).resolves.toBeNull() + expect(exec).toHaveBeenCalledTimes(3) + }) +}) diff --git a/src/main/git/repo-branch-conflict.ts b/src/main/git/repo-branch-conflict.ts index 5c6e03b94b0..f22fd97f99e 100644 --- a/src/main/git/repo-branch-conflict.ts +++ b/src/main/git/repo-branch-conflict.ts @@ -3,6 +3,7 @@ import { gitExecOptions, type LocalGitExecOptions } from './repo-default-base-re import { gitExecFileAsync } from './runner' import { isSafeGitRefName } from '../../shared/git-status-upstream-ref' import { + isShowRefNoMatchError, probeAnyExactRef, probeAnyExactRefBatched, type ExactRefProbeExec, @@ -29,16 +30,17 @@ function canQueryRemoteBranchName(branchName: string): boolean { return !branchName.startsWith('-') && isSafeGitRefName(`refs/heads/${branchName}`) } -async function hasGitRefAsync( +async function probeLocalBranchRef( exec: ExactRefProbeExec, ref: string, options: ExactRefProbeExecOptions -): Promise { +): Promise<'present' | 'absent' | 'unknown'> { try { - const { stdout } = await runGit(exec, ['rev-parse', '--verify', ref], options) - return stdout.trim().length > 0 - } catch { - return false + // Quiet absence avoids retrying the WSL probe through a login shell. + const { stdout } = await runGit(exec, ['rev-parse', '--verify', '--quiet', ref], options) + return stdout.trim().length > 0 ? 'present' : 'unknown' + } catch (error) { + return isShowRefNoMatchError(error) ? 'absent' : 'unknown' } } @@ -105,7 +107,8 @@ export async function getBranchConflictKindViaExec( branchName: string, allowedBaseRef?: string, options: ExactRefProbeExecOptions = {}, - batchedExec?: ExactRefProbeStdinExec + batchedExec?: ExactRefProbeStdinExec, + allowLocalBranch?: () => Promise ): Promise { if (!canQueryRemoteBranchName(branchName)) { return null @@ -114,7 +117,16 @@ export async function getBranchConflictKindViaExec( // are quiet, so introducing a smaller implicit cap would only make a large // remote configuration look like a missing conflict. const probeOptions: ExactRefProbeExecOptions = options - if (await hasGitRefAsync(exec, `refs/heads/${branchName}`, probeOptions)) { + const localRef = `refs/heads/${branchName}` + let presence = await probeLocalBranchRef(exec, localRef, probeOptions) + if (allowLocalBranch && presence !== 'absent') { + if (await allowLocalBranch()) { + return null + } + // Adoption can span ref changes; preserve the fresh conflict check after it fails. + presence = await probeLocalBranchRef(exec, localRef, probeOptions) + } + if (presence === 'present') { return 'local' } @@ -142,7 +154,8 @@ export function getBranchConflictKind( path: string, branchName: string, allowedBaseRef?: string, - options: LocalGitExecOptions = {} + options: LocalGitExecOptions = {}, + allowLocalBranch?: () => Promise ): Promise { const execOptions = gitExecOptions(path, options) const runLocalGit = ( @@ -168,7 +181,8 @@ export function getBranchConflictKind( // one `show-ref` subprocess per remote -- the exact cost the batch exists to remove. // `show-ref --verify --quiet` prints nothing and is read by exit code, so it needs // no fence; the capture wrapper preserves the payload's exit status either way. - (argv, commandOptions) => runLocalGit(argv, commandOptions, true) + (argv, commandOptions) => runLocalGit(argv, commandOptions, true), + allowLocalBranch ) } diff --git a/src/main/git/runner-wsl-direct-read.test.ts b/src/main/git/runner-wsl-direct-read.test.ts index 284cda55718..e7ea423f78e 100644 --- a/src/main/git/runner-wsl-direct-read.test.ts +++ b/src/main/git/runner-wsl-direct-read.test.ts @@ -17,6 +17,7 @@ vi.mock('../observability/instrumentation', () => ({ })) vi.mock('../diagnostics/main-thread-churn-probe', () => ({ recordSubprocessSpawn: vi.fn() })) +import { getBranchConflictKind } from './repo-branch-conflict' import { pendingWslDirectGitReadEnvironment } from './command-runner/git-command-resolution' import { gitExecFileAsync, gitSpawn, gitStreamStdout } from './runner' import { @@ -584,6 +585,33 @@ describe('WSL direct Git reads', () => { }) }) + it('checks a missing branch conflict without retrying through a login shell', async () => { + await withPlatform('win32', async () => { + seedWslGitReadEnvironmentForTests(DISTRO, LOGIN_ENVIRONMENT) + execFileMock.mockImplementation((_command, args: string[], _options, callback) => { + const child = createMockChild() + queueMicrotask(() => { + const missingRef = args.join(' ').includes('rev-parse') + const quiet = args.includes('--quiet') + const code = missingRef ? (quiet ? 1 : 128) : 0 + callback?.( + code ? Object.assign(new Error('missing ref'), { code }) : null, + '', + missingRef && !quiet ? 'fatal: Needed a single revision' : '' + ) + child.emit('close', code, null) + }) + return child + }) + + await expect( + getBranchConflictKind(String.raw`\\wsl.localhost\Ubuntu\repo`, 'new-feature') + ).resolves.toBeNull() + expect(execFileMock).toHaveBeenCalledTimes(2) + expect(execFileMock.mock.calls[0]?.[1]).toContain('--quiet') + }) + }) + it('keeps the fast path when direct and login Git both report an expected failure', async () => { await withPlatform('win32', async () => { seedWslGitReadEnvironmentForTests(DISTRO, LOGIN_ENVIRONMENT) diff --git a/src/main/git/worktree-add-creation-config.test.ts b/src/main/git/worktree-add-creation-config.test.ts index 44394b7728d..ff44ca7ac6d 100644 --- a/src/main/git/worktree-add-creation-config.test.ts +++ b/src/main/git/worktree-add-creation-config.test.ts @@ -199,7 +199,7 @@ describe('addWorktree', () => { }) const worktreeAddCall = gitExecFileAsyncMock.mock.calls.find( - ([argv]) => Array.isArray(argv) && argv[0] === 'worktree' && argv[1] === 'add' + ([argv]) => Array.isArray(argv) && argv.includes('worktree') && argv.includes('add') ) expect(worktreeAddCall?.[1]).toMatchObject({ timeout: WORKTREE_ADD_TIMEOUT_MS }) expect(WORKTREE_ADD_TIMEOUT_MS).toBeGreaterThan(0) @@ -214,7 +214,7 @@ describe('addWorktree', () => { }) const worktreeAddCall = gitExecFileAsyncMock.mock.calls.find( - ([argv]) => Array.isArray(argv) && argv[0] === 'worktree' && argv[1] === 'add' + ([argv]) => Array.isArray(argv) && argv.includes('worktree') && argv.includes('add') ) expect(worktreeAddCall?.[1]).toMatchObject({ timeout: 600_000 }) }) diff --git a/src/main/git/worktree-add-local-base-refresh.test.ts b/src/main/git/worktree-add-local-base-refresh.test.ts index dba4303647c..1ba2e6c16b8 100644 --- a/src/main/git/worktree-add-local-base-refresh.test.ts +++ b/src/main/git/worktree-add-local-base-refresh.test.ts @@ -1,5 +1,5 @@ // addWorktree: fast-forwarding the local base ref (reset --hard / update-ref) and its safety bailouts. -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const { gitExecFileAsyncMock, @@ -32,11 +32,14 @@ import { registerWorktreeSuiteHooks } from './worktree-test-harness' registerWorktreeSuiteHooks() describe('addWorktree', () => { + afterEach(() => vi.restoreAllMocks()) const resolveCreationBaseConfigWrite = () => { gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: '' }) // config --local --replace-all branch..base } beforeEach(() => { + // These branch-safety assertions use POSIX argv; Windows flags have separate coverage. + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') gitExecFileAsyncMock.mockReset() gitExecFileSyncMock.mockReset() translateWslOutputPathsMock.mockClear() diff --git a/src/main/git/worktree-add-local-base-suggestion.test.ts b/src/main/git/worktree-add-local-base-suggestion.test.ts index a65f77dd77d..3b449c339f6 100644 --- a/src/main/git/worktree-add-local-base-suggestion.test.ts +++ b/src/main/git/worktree-add-local-base-suggestion.test.ts @@ -1,5 +1,5 @@ // addWorktree: advisory local-base-ref update suggestions when the refresh setting is off. -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const { gitExecFileAsyncMock, @@ -32,7 +32,10 @@ import { registerWorktreeSuiteHooks } from './worktree-test-harness' registerWorktreeSuiteHooks() describe('addWorktree', () => { + afterEach(() => vi.restoreAllMocks()) beforeEach(() => { + // These branch-safety assertions use POSIX argv; Windows flags have separate coverage. + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') gitExecFileAsyncMock.mockReset() gitExecFileSyncMock.mockReset() translateWslOutputPathsMock.mockClear() diff --git a/src/main/git/worktree-add.ts b/src/main/git/worktree-add.ts index ea6ec704b46..380f3a3cc34 100644 --- a/src/main/git/worktree-add.ts +++ b/src/main/git/worktree-add.ts @@ -12,7 +12,7 @@ import { getLocalBaseRefUpdateSuggestionForWorktreeCreate, refreshLocalBaseRefForWorktreeCreate } from './worktree-base-refresh' -import { hasWorktreeBaseCommitRef } from './worktree-base-ref-probe' +import { resolveWorktreeBaseCommitOid } from './worktree-base-ref-probe' import type { AddWorktreeOptions, AddWorktreeResult, @@ -23,6 +23,7 @@ import { bumpWorktreeScanGeneration } from './worktree-scan-cache' export type WorktreeAddBaseContext = AddWorktreeResult & { effectiveBase: string + effectiveBaseOid?: string } export async function resolveWorktreeAddBaseContext( @@ -31,9 +32,11 @@ export async function resolveWorktreeAddBaseContext( refreshLocalBaseRef: boolean, options: AddWorktreeOptions ): Promise { - const effectiveBase = await resolveWorktreeAddBaseRef(baseBranch, (qualifiedRef) => - hasWorktreeBaseCommitRef(repoPath, qualifiedRef, options) - ) + let effectiveBaseOid: string | null = null + const effectiveBase = await resolveWorktreeAddBaseRef(baseBranch, async (qualifiedRef) => { + effectiveBaseOid = await resolveWorktreeBaseCommitOid(repoPath, qualifiedRef, options) + return effectiveBaseOid !== null + }) const localBaseRefRefresh = refreshLocalBaseRef ? await refreshLocalBaseRefForWorktreeCreate( repoPath, @@ -55,6 +58,10 @@ export async function resolveWorktreeAddBaseContext( : undefined return { effectiveBase, + // Refresh/suggestion work can span ref changes; only reuse the immediate resolution probe. + ...(!refreshLocalBaseRef && !options.suggestLocalBaseRefUpdate && effectiveBaseOid + ? { effectiveBaseOid } + : {}), ...(localBaseRefRefresh ? { localBaseRefRefresh } : {}), ...(localBaseRefUpdateSuggestion ? { localBaseRefUpdateSuggestion } : {}) } diff --git a/src/main/git/worktree-create-preparation-real-wsl.test.ts b/src/main/git/worktree-create-preparation-real-wsl.test.ts new file mode 100644 index 00000000000..8ccaf3c23bf --- /dev/null +++ b/src/main/git/worktree-create-preparation-real-wsl.test.ts @@ -0,0 +1,77 @@ +import { mkdir, mkdtemp, readFile, rm, writeFile } from 'node:fs/promises' +import { join } from 'node:path' +import { expect, it } from 'vitest' +import { createWorktreePreparationLockReason } from '../../shared/worktree/create-preparation' +import { gitExecFileAsync } from './runner' +import { + discardPreparedWorktree, + finalizePreparedWorktree, + prepareWorktreeCreateCheckout +} from './worktree-create-preparation' + +// Opt in on Windows with a running distro; all Git commands use the production WSL router. +const wslDistro = process.env.ORCA_TEST_WSL_DISTRO + +it.skipIf(process.platform !== 'win32' || !wslDistro)( + 'prepares, retargets, moves and cleans up a real WSL checkout from Windows', + async () => { + const fixtureParent = process.env.ORCA_TEST_WSL_ROOT ?? `\\\\wsl.localhost\\${wslDistro}\\tmp` + const root = await mkdtemp(join(fixtureParent, 'orca-create-route-')) + const repoPath = join(root, 'repo') + const preparedPath = join(root, 'prepared checkout') + const finalPath = join(root, 'final checkout') + const options = { wslDistro, timeout: 60_000 } + const git = async (cwd: string, args: string[]): Promise => + (await gitExecFileAsync(args, { cwd, ...options })).stdout.trim() + + try { + await mkdir(repoPath) + await git(repoPath, ['init', '--quiet']) + expect(await git(repoPath, ['rev-parse', '--show-toplevel'])).toMatch(/^\/(?!\/)/) + await git(repoPath, ['symbolic-ref', 'HEAD', 'refs/heads/main']) + await git(repoPath, ['config', 'user.name', 'Test User']) + await git(repoPath, ['config', 'user.email', 'test@example.com']) + await writeFile(join(repoPath, 'version.txt'), 'one\n') + await git(repoPath, ['add', 'version.txt']) + await git(repoPath, ['commit', '--quiet', '-m', 'initial']) + await prepareWorktreeCreateCheckout( + repoPath, + preparedPath, + 'main', + createWorktreePreparationLockReason('real-wsl-test'), + options + ) + expect(await git(repoPath, ['worktree', 'list', '--porcelain'])).toContain( + 'locked orca-create-preparation:v1:' + ) + + await writeFile(join(repoPath, 'version.txt'), 'two\n') + await git(repoPath, ['commit', '--quiet', '-am', 'advance base']) + const target = await git(repoPath, ['rev-parse', 'HEAD']) + await finalizePreparedWorktree( + repoPath, + preparedPath, + finalPath, + 'feature/routed', + 'main', + false, + options + ) + expect(await git(finalPath, ['rev-parse', 'HEAD'])).toBe(target) + expect(await git(finalPath, ['symbolic-ref', '--short', 'HEAD'])).toBe('feature/routed') + expect(await git(finalPath, ['status', '--porcelain'])).toBe('') + expect(await readFile(join(finalPath, 'version.txt'), 'utf8')).toBe('two\n') + expect(await git(finalPath, ['config', '--get', 'branch.feature/routed.base'])).toBe( + 'refs/heads/main' + ) + expect(await git(repoPath, ['worktree', 'list', '--porcelain'])).not.toContain('locked ') + await discardPreparedWorktree(repoPath, finalPath, options) + expect( + (await git(repoPath, ['worktree', 'list', '--porcelain'])).match(/^worktree /gm) + ).toHaveLength(1) + } finally { + await rm(root, { recursive: true, force: true }) + } + }, + 120_000 +) diff --git a/src/main/git/worktree-create-preparation.ts b/src/main/git/worktree-create-preparation.ts index b60dc01ec33..78713957690 100644 --- a/src/main/git/worktree-create-preparation.ts +++ b/src/main/git/worktree-create-preparation.ts @@ -195,25 +195,38 @@ export async function finalizePreparedWorktree( } try { return await runWithGitReadCacheInvalidation(async () => { - const baseContext = await resolveWorktreeAddBaseContext( - repoPath, - baseBranch, - refreshLocalBaseRef, - finalizeGitOptions - ) - const [targetHeadResult, preparedHeadResult] = await Promise.all([ - gitExecFileAsync( - ['rev-parse', '--verify', `${baseContext.effectiveBase}^{commit}`], - gitExecOptions(repoPath, finalizeGitOptions) - ), + const [targetResult, preparedResult] = await Promise.allSettled([ + (async () => { + const baseContext = await resolveWorktreeAddBaseContext( + repoPath, + baseBranch, + refreshLocalBaseRef, + finalizeGitOptions + ) + const targetHead = + baseContext.effectiveBaseOid ?? + ( + await gitExecFileAsync( + ['rev-parse', '--verify', `${baseContext.effectiveBase}^{commit}`], + gitExecOptions(repoPath, finalizeGitOptions) + ) + ).stdout.trim() + return { baseContext, targetHead } + })(), gitExecFileAsync( ['rev-parse', '--verify', 'HEAD'], gitExecOptions(preparedPath, finalizeGitOptions) ) ]) - const { stdout: targetHeadOutput } = targetHeadResult - const targetHead = targetHeadOutput.trim() - const { stdout: preparedHeadOutput } = preparedHeadResult + // Settle both reads before failure cleanup can remove the prepared checkout. + if (targetResult.status === 'rejected') { + throw targetResult.reason + } + if (preparedResult.status === 'rejected') { + throw preparedResult.reason + } + const { baseContext, targetHead } = targetResult.value + const preparedHeadOutput = preparedResult.value.stdout if (preparedHeadOutput.trim() !== targetHead) { await gitExecFileAsync( [...windowsLongPathGitArgs(preparedPath), 'reset', '--hard', targetHead], diff --git a/src/main/git/worktree-preparation-base-oid.test.ts b/src/main/git/worktree-preparation-base-oid.test.ts new file mode 100644 index 00000000000..5864b9d305a --- /dev/null +++ b/src/main/git/worktree-preparation-base-oid.test.ts @@ -0,0 +1,98 @@ +import { beforeEach, expect, it, vi } from 'vitest' + +const gitExec = vi.hoisted(() => vi.fn()) +vi.mock('./runner', () => ({ gitExecFileAsync: gitExec })) +vi.mock('./worktree-base-refresh', () => ({ + refreshLocalBaseRefForWorktreeCreate: vi.fn(), + getLocalBaseRefUpdateSuggestionForWorktreeCreate: vi.fn() +})) +vi.mock('./status', () => ({ runWithGitReadCacheInvalidation: (run: () => unknown) => run() })) +vi.mock('./wsl-linked-worktree-git-routing', () => ({ + invalidateWslLinkedWorktreeGitRouting: vi.fn() +})) + +import { finalizePreparedWorktree } from './worktree-create-preparation' + +const originalOid = '1'.repeat(40) +const refreshedOid = '2'.repeat(40) + +beforeEach(() => { + gitExec.mockReset().mockImplementation(async (args: string[]) => ({ + stdout: + args[0] === 'rev-parse' + ? args.includes('--quiet') || args.at(-1) === 'HEAD' + ? originalOid + : refreshedOid + : '' + })) +}) + +it('reuses the current base-resolution oid and preserves WSL routing', async () => { + await finalizePreparedWorktree('/repo', '/prepared', '/final', 'feature', 'main', false, { + wslDistro: 'Ubuntu', + timeout: 8000 + }) + const revisions = gitExec.mock.calls.filter(([args]) => args[0] === 'rev-parse') + expect(revisions.map(([args]) => args)).toEqual([ + ['rev-parse', '--verify', '--quiet', 'refs/heads/main^{commit}'], + ['rev-parse', '--verify', 'HEAD'] + ]) + expect(gitExec.mock.calls.find(([args]) => args.includes('checkout'))?.[0]).toContain(originalOid) + expect(gitExec.mock.calls.some(([args]) => args.includes('reset'))).toBe(false) + for (const [, options] of gitExec.mock.calls) { + expect(options).toMatchObject({ wslDistro: 'Ubuntu', timeout: 8000 }) + } +}) + +it.each([ + { base: 'refs/heads/main', refresh: false, options: {} }, + { base: 'main', refresh: true, options: {} }, + { base: 'main', refresh: false, options: { suggestLocalBaseRefUpdate: true } } +])('re-reads the target for $base, refresh=$refresh, options=$options', async (test) => { + await finalizePreparedWorktree( + '/repo', + '/prepared', + '/final', + 'feature', + test.base, + test.refresh, + test.options + ) + expect(gitExec).toHaveBeenCalledWith( + ['rev-parse', '--verify', 'refs/heads/main^{commit}'], + expect.objectContaining({ cwd: '/repo' }) + ) + expect(gitExec.mock.calls.find(([args]) => args.includes('reset'))?.[0]).toContain(refreshedOid) + expect(gitExec.mock.calls.find(([args]) => args.includes('checkout'))?.[0]).toContain( + refreshedOid + ) +}) + +it('starts both independent probes before either resolves and settles them before failure', async () => { + let resolveBase!: (value: { stdout: string }) => void + let rejectPrepared!: (reason: Error) => void + gitExec.mockImplementation((args: string[]) => { + if (args.includes('--quiet')) { + return new Promise((resolve) => (resolveBase = resolve)) + } + if (args.at(-1) === 'HEAD') { + return new Promise((_, reject) => (rejectPrepared = reject)) + } + return Promise.resolve({ stdout: '' }) + }) + let settled = false + const error = new Error('prepared HEAD unreadable') + const result = finalizePreparedWorktree('/repo', '/prepared', '/final', 'feature', 'main') + const checked = expect(result).rejects.toBe(error) + void result.then( + () => (settled = true), + () => (settled = true) + ) + await vi.waitFor(() => expect(gitExec).toHaveBeenCalledTimes(2)) + rejectPrepared(error) + await Promise.resolve() + expect(settled).toBe(false) + resolveBase({ stdout: originalOid }) + await checked + expect(gitExec.mock.calls.some(([args]) => args.includes('move'))).toBe(false) +}) diff --git a/src/main/ipc/worktree-logic-wsl.test.ts b/src/main/ipc/worktree-logic-wsl.test.ts index c30c387263e..c21c20e0050 100644 --- a/src/main/ipc/worktree-logic-wsl.test.ts +++ b/src/main/ipc/worktree-logic-wsl.test.ts @@ -16,6 +16,7 @@ vi.mock('../wsl', () => ({ import { computeWorktreePath, computeWorktreePathAsync, + computeWorkspaceRootAsync, getWorktreePathSettings } from './worktree-logic' import { @@ -32,6 +33,26 @@ describe('computeWorktreePath WSL layout', () => { parseWslPathMock.mockReset() }) + it('reuses an asynchronously resolved root for every name candidate without a sync probe', async () => { + parseWslPathMock.mockReturnValue({ distro: 'Ubuntu', linuxPath: '/home/jin/repo' }) + const repoPath = String.raw`\\wsl.localhost\Ubuntu\home\jin\repo` + const home = String.raw`\\wsl.localhost\Ubuntu\home\jin` + const settings = { workspaceDir: 'C:\\workspaces', nestWorkspaces: true } + let resolveHome!: (home: string) => void + getWslHomeAsyncMock.mockReturnValue(new Promise((resolve) => (resolveHome = resolve))) + const pendingRoot = computeWorkspaceRootAsync(repoPath, settings) + expect(getWslHomeMock).not.toHaveBeenCalled() + resolveHome(home) + const root = await pendingRoot + for (const name of ['feature', 'feature-2', 'feature-3']) { + expect(computeWorktreePath(name, repoPath, settings, root)).toBe( + win32.join(home, 'orca', 'workspaces', 'repo', name) + ) + } + expect(getWslHomeAsyncMock).toHaveBeenCalledExactlyOnceWith('Ubuntu') + expect(getWslHomeMock).not.toHaveBeenCalled() + }) + it('places WSL repo worktrees under the distro home workspace root', () => { parseWslPathMock.mockReturnValue({ distro: 'Ubuntu', diff --git a/src/main/ipc/worktree-logic.ts b/src/main/ipc/worktree-logic.ts index 7a8fe175c89..744572f6e37 100644 --- a/src/main/ipc/worktree-logic.ts +++ b/src/main/ipc/worktree-logic.ts @@ -103,12 +103,13 @@ export function ensurePathWithinWorkspace(targetPath: string, workspaceDir: stri export function computeWorktreePath( sanitizedName: string, repoPath: string, - settings: WorktreePathSettings + settings: WorktreePathSettings, + workspaceRoot?: string ): string { return computeWorktreePathFromWorkspaceRoot( sanitizedName, repoPath, - computeWorkspaceRoot(repoPath, settings), + workspaceRoot ?? computeWorkspaceRoot(repoPath, settings), settings.nestWorkspaces ) } @@ -130,7 +131,7 @@ function computeWorktreePathFromWorkspaceRoot( } /** Async twin of computeWorktreePath. Same result; resolves the WSL home without blocking the main - * thread, so callers off the create path never freeze the app on a stopped distro. */ + * thread, so callers never freeze the app on a stopped distro. */ export async function computeWorktreePathAsync( sanitizedName: string, repoPath: string, @@ -147,7 +148,7 @@ export async function computeWorktreePathAsync( /** Async twin of computeWorkspaceRoot. Same result; the WSL home probe spawns `wsl.exe`, so * background preparation uses this variant rather than blocking the Electron main thread for up * to the probe timeout. The sync twin below still serves callers that cannot await (allowed-roots - * resolution, the create click, CLI create, watch targets, worktree trash). */ + * resolution, CLI create, watch targets, worktree trash). */ export async function computeWorkspaceRootAsync( repoPath: string, settings: { workspaceDir: string; wslMirrorDistro?: string } diff --git a/src/main/ipc/worktree-remote.ts b/src/main/ipc/worktree-remote.ts index 27eaa9264db..ef65f2fcb53 100644 --- a/src/main/ipc/worktree-remote.ts +++ b/src/main/ipc/worktree-remote.ts @@ -87,7 +87,7 @@ import { computeValidatedBranchName, computeWorktreePath, computeRemoteWorktreePath, - computeWorkspaceRoot, + computeWorkspaceRootAsync, ensurePathWithinWorkspace, getWorktreeCreationLayout, getWorktreePathSettings, @@ -2444,7 +2444,7 @@ export async function createLocalWorktree( emitCreateWorktreeProgress(mainWindow, 'fetching', args.creationId) } } - const workspaceRoot = computeWorkspaceRoot(repo.path, worktreePathSettings) + const workspaceRoot = await computeWorkspaceRootAsync(repo.path, worktreePathSettings) // Why: this validation doesn't depend on remote refs, so it can overlap a required remote-tracking base refresh. const primarySetupScript = getEffectiveHooks(repo)?.scripts.setup @@ -2530,19 +2530,33 @@ export async function createLocalWorktree( username, localWorktreeGitOptions ) - checkoutExistingBranch = await canCheckoutExistingLocalBranch( - repo.path, - branchName, - baseBranch, - localWorktreeGitOptions - ) - if (checkoutExistingBranch && !selectedExistingLocalBranchName) { - // Why: suffix retries may need a new path, but an existing-branch checkout must keep the user-selected branch, not a sibling. - selectedExistingLocalBranchName = branchName + const tryExistingBranch = async (): Promise => { + checkoutExistingBranch = await canCheckoutExistingLocalBranch( + repo.path, + branchName, + baseBranch, + localWorktreeGitOptions + ) + return checkoutExistingBranch } + // Explicit branch selections retain the adoption-first path. + const preferExistingBranch = Boolean( + args.branchNameOverride || selectedExistingLocalBranchName + ) + checkoutExistingBranch = preferExistingBranch && (await tryExistingBranch()) lastBranchConflictKind = checkoutExistingBranch ? null - : await getBranchConflictKind(repo.path, branchName, baseBranch, localWorktreeGitOptions) + : await getBranchConflictKind( + repo.path, + branchName, + baseBranch, + localWorktreeGitOptions, + preferExistingBranch ? undefined : tryExistingBranch + ) + if (checkoutExistingBranch && !selectedExistingLocalBranchName) { + // Path retries must retain the adopted branch. + selectedExistingLocalBranchName = branchName + } const allowedPushTargetRemoteConflict = lastBranchConflictKind && isAllowedPushTargetRemoteConflict(lastBranchConflictKind, branchName, args) @@ -2604,7 +2618,7 @@ export async function createLocalWorktree( } worktreePath = ensurePathWithinWorkspace( - computeWorktreePath(effectiveSanitizedName, repo.path, worktreePathSettings), + computeWorktreePath(effectiveSanitizedName, repo.path, worktreePathSettings, workspaceRoot), workspaceRoot ) if (existsSync(worktreePath)) { @@ -3010,6 +3024,8 @@ export async function createLocalWorktree( } }) + // Startup resolves the new id before lifecycle notifications invalidate runtime caches. + runtime?.invalidateWorktreeCatalog?.(repo.id) const stagedStartup = await timing.time('spawn_startup_terminal', () => spawnLocalStartupAndSetupTerminals({ runtime, diff --git a/src/main/ipc/worktrees-local-create-flow.test.ts b/src/main/ipc/worktrees-local-create-flow.test.ts index abc711bbd9b..fc57c6969b6 100644 --- a/src/main/ipc/worktrees-local-create-flow.test.ts +++ b/src/main/ipc/worktrees-local-create-flow.test.ts @@ -2,6 +2,8 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' import { resolve } from 'node:path' import type { CreateWorktreeResult } from '../../shared/worktree/create-types' import { resolveRegisteredWorktreePath } from './registered-worktree-roots-cache' +import { computeWorkspaceRootAsync } from './worktree-logic' +import type * as WorktreeLogic from './worktree-logic' import { listWorktreesMock, describeCreatedWorktreeMock, @@ -78,11 +80,13 @@ vi.mock('../setup-hook-env-vars', async (importOriginal) => (await importOriginal()) as Record ) ) -vi.mock('./worktree-logic', async (importOriginal) => - (await import('./worktrees-test-module-mocks')).worktreeLogicModuleMock( - (await importOriginal()) as Record - ) -) +vi.mock('./worktree-logic', async (importOriginal) => { + const actual = await importOriginal() + return { + ...(await import('./worktrees-test-module-mocks')).worktreeLogicModuleMock(actual), + computeWorkspaceRootAsync: vi.fn(actual.computeWorkspaceRootAsync) + } +}) vi.mock('../terminal-history-deletion', async () => (await import('./worktrees-test-module-mocks')).terminalHistoryDeletionModuleMock() ) @@ -409,15 +413,23 @@ describe('registerWorktreeHandlers', () => { } ]) - await handlers['worktrees:create'](null, { + const root = Promise.withResolvers() + vi.mocked(computeWorkspaceRootAsync).mockReturnValueOnce(root.promise) + const create = handlers['worktrees:create'](null, { repoId: 'repo-1', name: 'feature' }) - expect(computeWorktreePathMock).toHaveBeenCalledWith('feature', '/workspace/repo', { - nestWorkspaces: false, - workspaceDir: '../worktrees' - }) + await vi.waitFor(() => expect(computeWorkspaceRootAsync).toHaveBeenCalled()) + expect(addWorktreeMock).not.toHaveBeenCalled() + root.resolve('/workspace/worktrees') + await create + expect(computeWorktreePathMock).toHaveBeenCalledWith( + 'feature', + '/workspace/repo', + { nestWorkspaces: false, workspaceDir: '../worktrees' }, + '/workspace/worktrees' + ) expect(addWorktreeMock).toHaveBeenCalledWith( '/workspace/repo', '../worktrees/feature', @@ -689,6 +701,10 @@ describe('registerWorktreeHandlers', () => { expect(setupCommand).toBe('bash /workspace/repo/.git/orca/setup-runner.sh') expect(result.setup).toBeUndefined() expect(result.startupTerminal).toEqual({ spawned: true, surface: 'visible' }) + expect(runtimeStub.invalidateWorktreeCatalog).toHaveBeenCalledWith('repo-1') + expect(runtimeStub.invalidateWorktreeCatalog.mock.invocationCallOrder[0]).toBeLessThan( + runtimeStub.createTerminal.mock.invocationCallOrder[0] + ) expect(result.timing?.phases.map((phase) => phase.phase)).toEqual( expect.arrayContaining([ 'git_worktree_add', diff --git a/src/main/ipc/worktrees-test-module-mocks.ts b/src/main/ipc/worktrees-test-module-mocks.ts index a1925d1ce72..8d2787fcf2f 100644 --- a/src/main/ipc/worktrees-test-module-mocks.ts +++ b/src/main/ipc/worktrees-test-module-mocks.ts @@ -1,4 +1,5 @@ import { type Mock, vi } from 'vitest' +import type { computeWorktreePath } from './worktree-logic' import type { HandlerMap } from './worktrees-test-ipc-surface' /** Loose signature: one mock stands in for many unrelated module exports. */ @@ -78,13 +79,7 @@ export const resolveSetupRunnerShellMock: ModuleMock = vi.fn() export const runHookMock: ModuleMock = vi.fn() export const hasHooksFileMock: ModuleMock = vi.fn() export const loadHooksMock: ModuleMock = vi.fn() -export const computeWorktreePathMock: Mock< - ( - sanitizedName: string, - repoPath: string, - settings: { nestWorkspaces: boolean; workspaceDir: string } - ) => string -> = vi.fn() +export const computeWorktreePathMock: Mock = vi.fn() export const ensurePathWithinWorkspaceMock: StringArgMock = vi.fn() export const gitExecFileAsyncMock: GitArgvMock = vi.fn() export const getSshGitProviderMock: StringArgMock = vi.fn() @@ -138,7 +133,19 @@ export const gitRepoModuleMock = () => ({ resolveDefaultBaseRefWithLocalGit: resolveDefaultBaseRefWithLocalGitMock, resolveDefaultBaseRefViaExec: resolveDefaultBaseRefViaExecMock, getDefaultRemote: getDefaultRemoteMock, - getBranchConflictKind: getBranchConflictKindMock + getBranchConflictKind: async ( + repoPath: string, + branch: string, + base?: string, + options?: { wslDistro?: string }, + allowLocalBranch?: () => Promise + ) => { + // These handler tests stub ref presence; policy tests cover the absent-ref fast path. + if (allowLocalBranch && (await allowLocalBranch())) { + return null + } + return getBranchConflictKindMock(repoPath, branch, base, options) + } }) export const githubClientModuleMock = () => ({ diff --git a/src/main/ipc/worktrees-test-runtime-stub.ts b/src/main/ipc/worktrees-test-runtime-stub.ts index 647d4801d13..bb2f33cb05e 100644 --- a/src/main/ipc/worktrees-test-runtime-stub.ts +++ b/src/main/ipc/worktrees-test-runtime-stub.ts @@ -12,6 +12,7 @@ export type WorktreeRuntimeStub = { clearOptimisticReconcileToken: ReturnType resolveManagedMrBase: ReturnType createTerminal: ReturnType + invalidateWorktreeCatalog: ReturnType splitTerminal: ReturnType notifyWorktreesChangedForRemoteClients: ReturnType closeFileWatchersForRemoval: ReturnType @@ -38,6 +39,7 @@ export function createWorktreeRuntimeStub(): WorktreeRuntimeStub { title: null, surface: 'visible' }), + invalidateWorktreeCatalog: vi.fn(), splitTerminal: vi.fn().mockResolvedValue({ handle: 'term-setup', tabId: 'tab-startup', diff --git a/src/main/providers/local-pty-provider-spawn-session.test.ts b/src/main/providers/local-pty-provider-spawn-session.test.ts index 7513d9c72cb..9dfa08d81bf 100644 --- a/src/main/providers/local-pty-provider-spawn-session.test.ts +++ b/src/main/providers/local-pty-provider-spawn-session.test.ts @@ -172,6 +172,7 @@ describe('LocalPtyProvider', () => { expect(second).toEqual({ id: 'serve-session-1', + incarnationId: first.incarnationId, pid: 12345, isReattach: true, // Why published: this attach really moved the PTY, unlike daemon/relay attach, so main @@ -220,7 +221,11 @@ describe('LocalPtyProvider', () => { attachOnly: true }) - expect(result).toMatchObject({ id: first.id, isReattach: true }) + expect(result).toMatchObject({ + id: first.id, + incarnationId: first.incarnationId, + isReattach: true + }) expect(spawnMock).not.toHaveBeenCalled() }) diff --git a/src/main/providers/local-pty-spawn-state.ts b/src/main/providers/local-pty-spawn-state.ts index d41f858cf98..2ab145f6c39 100644 --- a/src/main/providers/local-pty-spawn-state.ts +++ b/src/main/providers/local-pty-spawn-state.ts @@ -1,6 +1,7 @@ import type { PtySpawnResult } from './types' import { pendingLocalPtySpawns, + ptyIncarnations, ptyProcesses, ptyWslDistroById, type PendingLocalPtySpawn @@ -60,6 +61,7 @@ export function reattachLocalPty(id: string, cols: number, rows: number): PtySpa } return { id, + ...(ptyIncarnations.has(id) ? { incarnationId: ptyIncarnations.get(id) } : {}), pid: existing.pid, ...(ptyWslDistroById.has(id) ? { wslDistro: ptyWslDistroById.get(id) ?? null } : {}), isReattach: true, diff --git a/src/main/runtime/fetch-remote-cache.test.ts b/src/main/runtime/fetch-remote-cache.test.ts index 11cd2e0260d..5f0b8e249d6 100644 --- a/src/main/runtime/fetch-remote-cache.test.ts +++ b/src/main/runtime/fetch-remote-cache.test.ts @@ -176,8 +176,17 @@ describe('OrcaRuntimeService.fetchRemoteWithCache', () => { expect(caches.fetchLastCompletedAt.has('/repo/cache-0::origin')).toBe(false) }) - it.each(['main', 'a'.repeat(40), 'refs/remotes/main', ''])( - 'does not launch Git for a base without a remote/branch separator: %s', + it.each([ + 'main', + 'a'.repeat(40), + 'refs/remotes/main', + '', + 'origin/', + '/main', + 'refs/remotes/origin/', + 'refs/remotes//main' + ])( + 'does not launch Git for a base without both remote and branch components: %s', async (base) => { const runtime = new OrcaRuntimeService(null) await expect(runtime.resolveRemoteTrackingBase('/repo/e', base)).resolves.toBeNull() diff --git a/src/main/runtime/orca-runtime-refresh-repo-worktree-scan.ts b/src/main/runtime/orca-runtime-refresh-repo-worktree-scan.ts index 1964b5fdd2f..2df3f7b9f17 100644 --- a/src/main/runtime/orca-runtime-refresh-repo-worktree-scan.ts +++ b/src/main/runtime/orca-runtime-refresh-repo-worktree-scan.ts @@ -153,6 +153,11 @@ export class OrcaRuntimeWithRefreshRepoWorktreeScan extends OrcaRuntimeWithListK } } + invalidateWorktreeCatalog(repoId: string): void { + this.invalidateResolvedWorktreeCache() + this.invalidateWorktreeScanCacheForRepo(repoId) + } + protected invalidateSshWorktreeScanCacheInternal(targetId: string): void { const repos = this.store?.getRepos() ?? [] const affectedRepos = repos.filter((repo) => getRepoSshConnectionId(repo) === targetId) diff --git a/src/main/runtime/orca-runtime-tests/local-worktree-creation-part-02.spec.ts b/src/main/runtime/orca-runtime-tests/local-worktree-creation-part-02.spec.ts index fe6e8bfe3da..a12ffabcd61 100644 --- a/src/main/runtime/orca-runtime-tests/local-worktree-creation-part-02.spec.ts +++ b/src/main/runtime/orca-runtime-tests/local-worktree-creation-part-02.spec.ts @@ -50,7 +50,13 @@ describe('OrcaRuntimeService', () => { pushTarget: { remoteName: 'origin', branchName: 'feature/fix' } }) - expect(getBranchConflictKind).toHaveBeenCalledWith(TEST_REPO_PATH, 'feature/fix', 'abc123') + expect(getBranchConflictKind).toHaveBeenCalledWith( + TEST_REPO_PATH, + 'feature/fix', + 'abc123', + {}, + undefined + ) expect(getPRForBranchMock).toHaveBeenCalledWith(TEST_REPO_PATH, 'feature/fix') expect(addWorktree).toHaveBeenCalledWith( TEST_REPO_PATH, @@ -165,7 +171,9 @@ describe('OrcaRuntimeService', () => { expect(getBranchConflictKind).toHaveBeenCalledWith( TEST_REPO_PATH, 'feature/bitbucket', - 'abc123' + 'abc123', + {}, + undefined ) expect(getHostedReviewForBranchMock).toHaveBeenCalledWith( expect.objectContaining({ diff --git a/src/main/runtime/orca-runtime-tests/local-worktree-creation.spec.ts b/src/main/runtime/orca-runtime-tests/local-worktree-creation.spec.ts index 17fe670a09c..3dcd2d0584c 100644 --- a/src/main/runtime/orca-runtime-tests/local-worktree-creation.spec.ts +++ b/src/main/runtime/orca-runtime-tests/local-worktree-creation.spec.ts @@ -533,10 +533,14 @@ describe('OrcaRuntimeService', () => { branchNameOverride: 'feature/something' }) + // Why: an explicit branch override adopts the local branch before the conflict + // probe, so no lazy adoption callback is handed to getBranchConflictKind. expect(getBranchConflictKind).toHaveBeenCalledWith( TEST_REPO_PATH, 'feature/something', - 'origin/feature/something' + 'origin/feature/something', + {}, + undefined ) expect(addWorktree).toHaveBeenCalledWith( TEST_REPO_PATH, diff --git a/src/main/runtime/orca-runtime-tests/worktree-removal-and-reconciliation.spec.ts b/src/main/runtime/orca-runtime-tests/worktree-removal-and-reconciliation.spec.ts index 65da9caccde..630eaebf007 100644 --- a/src/main/runtime/orca-runtime-tests/worktree-removal-and-reconciliation.spec.ts +++ b/src/main/runtime/orca-runtime-tests/worktree-removal-and-reconciliation.spec.ts @@ -376,7 +376,18 @@ describe('OrcaRuntimeService', () => { TEST_REPO_PATH, 'runtime-wsl', 'origin/main', - { wslDistro: 'Ubuntu' } + { wslDistro: 'Ubuntu' }, + expect.any(Function) + ) + // Why: the lazy adoption callback is only invoked when the conflict probe + // sees a local ref, so drive it here to prove adoption also routes via WSL. + const adoptLocalBranch = vi + .mocked(getBranchConflictKind) + .mock.calls.findLast((call) => call[1] === 'runtime-wsl')?.[4] + await expect(adoptLocalBranch?.()).resolves.toBe(false) + expect(gitSpy).toHaveBeenCalledWith( + ['rev-parse', '--verify', '--quiet', 'refs/heads/runtime-wsl^{commit}'], + { cwd: TEST_REPO_PATH, wslDistro: 'Ubuntu' } ) expect(getPRForBranchMock).toHaveBeenCalledWith( TEST_REPO_PATH, diff --git a/src/main/runtime/runtime-local-worktree-create-candidate.ts b/src/main/runtime/runtime-local-worktree-create-candidate.ts index 6c454666148..9de955c5cd1 100644 --- a/src/main/runtime/runtime-local-worktree-create-candidate.ts +++ b/src/main/runtime/runtime-local-worktree-create-candidate.ts @@ -110,23 +110,31 @@ export async function resolveRuntimeLocalWorktreeCreateCandidate(args: { args.username, args.localWorktreeGitOptions ) - checkoutExistingBranch = await canCheckoutExistingLocalBranch( - args.repo.path, - branchName, - args.baseBranch, - ...args.localWorktreeGitOptionArgs - ) - if (checkoutExistingBranch && !selectedExistingLocalBranchName) { - selectedExistingLocalBranchName = branchName + const tryExistingBranch = async (): Promise => { + checkoutExistingBranch = await canCheckoutExistingLocalBranch( + args.repo.path, + branchName, + args.baseBranch, + ...args.localWorktreeGitOptionArgs + ) + return checkoutExistingBranch } + const preferExistingBranch = Boolean( + args.request.branchNameOverride || selectedExistingLocalBranchName + ) + checkoutExistingBranch = preferExistingBranch && (await tryExistingBranch()) branchConflictKind = checkoutExistingBranch ? null : await getBranchConflictKind( args.repo.path, branchName, args.baseBranch, - ...args.localWorktreeGitOptionArgs + args.localWorktreeGitOptions, + preferExistingBranch ? undefined : tryExistingBranch ) + if (checkoutExistingBranch && !selectedExistingLocalBranchName) { + selectedExistingLocalBranchName = branchName + } const allowedPushTargetRemoteConflict = branchConflictKind && isAllowedPushTargetRemoteConflict(branchConflictKind, branchName, args.request) diff --git a/src/main/runtime/runtime-remote-fetch-controller.ts b/src/main/runtime/runtime-remote-fetch-controller.ts index f40f25f6c82..dbcc240525b 100644 --- a/src/main/runtime/runtime-remote-fetch-controller.ts +++ b/src/main/runtime/runtime-remote-fetch-controller.ts @@ -231,7 +231,7 @@ export class RuntimeRemoteFetchController { ? baseBranch.slice(remoteRefPrefix.length) : baseBranch // A remote-tracking base needs both a configured remote and a branch component. - if (!shortBaseBranch.includes('/')) { + if (shortBaseBranch.indexOf('/') <= 0 || shortBaseBranch.endsWith('/')) { return null } let remotes: string[] diff --git a/src/main/runtime/worktree-scan-admin-fingerprint-gate.test.ts b/src/main/runtime/worktree-scan-admin-fingerprint-gate.test.ts index 3750884f84f..7dfe0144406 100644 --- a/src/main/runtime/worktree-scan-admin-fingerprint-gate.test.ts +++ b/src/main/runtime/worktree-scan-admin-fingerprint-gate.test.ts @@ -273,6 +273,22 @@ describe('worktree scan admin-fingerprint gate', () => { } }) + it('resolves a just-created id after invalidation even within both cache TTLs', async () => { + const { runtime, list } = makeRuntime() + listWorktreesStrictMock.mockResolvedValueOnce([ + { path: REPO_PATH, head: 'abc', branch: 'main', isBare: false, isMainWorktree: true } + ]) + await list() + await expect(runtime.showManagedWorktree(`id:${WORKTREE_ID}`)).rejects.toThrow( + 'selector_not_found' + ) + runtime.invalidateWorktreeCatalog(REPO_ID) + await expect(runtime.showManagedWorktree(`id:${WORKTREE_ID}`)).resolves.toMatchObject({ + id: WORKTREE_ID + }) + expect(scanCount()).toBe(2) + }) + it('scans when the probe cannot describe the repo', async () => { vi.useFakeTimers() try { diff --git a/src/renderer/src/components/terminal-pane/ipc-pty-connect-result.ts b/src/renderer/src/components/terminal-pane/ipc-pty-connect-result.ts index 3dda2f83fec..bcfc0d3f67c 100644 --- a/src/renderer/src/components/terminal-pane/ipc-pty-connect-result.ts +++ b/src/renderer/src/components/terminal-pane/ipc-pty-connect-result.ts @@ -16,6 +16,10 @@ export function projectIpcPtyConnectResult( snapshot: spawnResult.snapshot, snapshotCols: spawnResult.snapshotCols, snapshotRows: spawnResult.snapshotRows, + ...(spawnResult.snapshotSeq !== undefined ? { snapshotSeq: spawnResult.snapshotSeq } : {}), + ...(spawnResult.snapshotKittyKeyboardFlags !== undefined + ? { snapshotKittyKeyboardFlags: spawnResult.snapshotKittyKeyboardFlags } + : {}), ...(spawnResult.snapshotPrefixAnsi !== undefined ? { snapshotPrefixAnsi: spawnResult.snapshotPrefixAnsi } : {}), diff --git a/src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts b/src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts index fa110cdfe0e..ed38f6a31a4 100644 --- a/src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts +++ b/src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts @@ -231,6 +231,41 @@ describe('connectPanePty', () => { expect(transport.sendInput).not.toHaveBeenCalled() }) + it.each([ + { kind: 'covered backlog', snapshot: 'startup\r\n', seq: 9, count: 1 }, + { kind: 'legacy unsequenced snapshot', snapshot: 'startup\r\n', seq: undefined, count: 2 }, + { kind: 'blank snapshot', snapshot: '\x1b[2J', seq: 9, count: 1 } + ])( + 'preserves output while reconciling $kind on daemon adoption', + async ({ snapshot, seq, count }) => { + const { connectPanePty } = await import('./pty-connection') + const transport = createMockTransport('tab-pty') + transport.connect.mockImplementation( + async ({ sessionId, callbacks }: { sessionId?: string; callbacks?: ConnectCallbacks }) => { + callbacks?.onData?.('startup\r\n', { seq: 9, rawLength: 9 }) + callbacks?.onData?.('new output\r\n', { seq: 21, rawLength: 12 }) + return { id: sessionId, snapshot, snapshotSeq: seq } + } + ) + transportFactoryQueue.push(transport) + const pane = createPane(1) + const { writes, parseCallbacks } = captureCallbackTerminalWrites(pane) + const deps = createDeps({ + isVisibleRef: { current: true }, + restoredLeafId: LEAF_1, + restoredPtyIdByLeafId: { [LEAF_1]: 'tab-pty' } + }) + connectPanePty(pane as never, createManager(1) as never, deps as never) + await flushAsyncTicks(20) + for (let step = 0; step < 40; step += 1) { + parseCallbacks.shift()?.() + await flushAsyncTicks(2) + } + expect(writes.join('').match(/startup/g)).toHaveLength(count) + expect(writes.join('')).toContain('new output') + } + ) + it('drains live bytes after transport confirms an explicit reattach', async () => { const { connectPanePty } = await import('./pty-connection') const { deliverTerminalDataWithDeferredCredit } = diff --git a/src/renderer/src/components/terminal-pane/pty-connection/apply-reattach-payload.ts b/src/renderer/src/components/terminal-pane/pty-connection/apply-reattach-payload.ts index cb0806900af..ad64d458711 100644 --- a/src/renderer/src/components/terminal-pane/pty-connection/apply-reattach-payload.ts +++ b/src/renderer/src/components/terminal-pane/pty-connection/apply-reattach-payload.ts @@ -100,6 +100,13 @@ export function createReattachPayloadHandlers( // Why last: re-arm the dangling mid-escape after the reset (whose ESC would abort it) so the live continuation completes it (#7329). session.writeReplayData(ctx.connectResult.pendingEscapeTailAnsi) } + // The initial attach backlog can contain bytes already painted by this snapshot. + session.setRestoredSnapshotBaseline( + ctx.ptyId, + { seq: ctx.connectResult.snapshotSeq }, + restoredSnapshotPaintsPrintableContent({ data: daemonSnapshotReplay }) + ) + session.recordRendererOrderedSeq({ seq: ctx.connectResult.snapshotSeq }) session.sendFocusedReattachFocusInAfterReplay(ctx.ptyId, ctx.attemptGeneration) if (ctx.connectResult.coldRestore) { // Snapshot superseded the cold-restore payload; ack so the daemon doesn't redeliver it. diff --git a/src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts b/src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts index 66dca98b360..138059f372b 100644 --- a/src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts +++ b/src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts @@ -26,6 +26,24 @@ describe('createIpcPtyTransport', () => { restorePtySpecWindow(originalWindow) }) + it.each([0, 420])( + 'preserves snapshot sequence and keyboard proof %s across IPC reattach', + async (seq) => { + const { createIpcPtyTransport } = await import('./pty-transport') + vi.mocked(window.api.pty.spawn).mockResolvedValue({ + id: 'existing', + isReattach: true, + snapshot: 'ready', + snapshotSeq: seq, + snapshotKittyKeyboardFlags: 0 + }) + const transport = createIpcPtyTransport({}) + const result = await transport.connect({ url: '', sessionId: 'existing', callbacks: {} }) + expect(result).toMatchObject({ snapshotSeq: seq, snapshotKittyKeyboardFlags: 0 }) + transport.detach?.() + } + ) + it('leaves title tracking to the PTY data stream (no OpenCode IPC channel)', async () => { // Why: the OpenCode status IPC channel is gone (now the agent-hooks server), so the transport has no per-agent status callback. const { createIpcPtyTransport } = await import('./pty-transport') diff --git a/src/renderer/src/components/terminal-pane/terminal-pane-manager-options.ts b/src/renderer/src/components/terminal-pane/terminal-pane-manager-options.ts index 08eab4c0cf5..128b715a3c5 100644 --- a/src/renderer/src/components/terminal-pane/terminal-pane-manager-options.ts +++ b/src/renderer/src/components/terminal-pane/terminal-pane-manager-options.ts @@ -1,6 +1,7 @@ import type { IDisposable } from '@xterm/xterm' import type { PaneManagerOptions } from '@/lib/pane-manager/pane-manager' import { useAppStore } from '@/store' +import { resolveTerminalLigaturesEnabled } from '../../../../shared/terminal-ligatures' import { resolveTerminalFontWeights } from '../../../../shared/terminal-fonts' import { normalizeTerminalLineHeight } from '../../../../shared/terminal-line-height-settings' import { normalizeDesktopTerminalScrollbackRows } from '../../../../shared/terminal-scrollback-policy' @@ -102,6 +103,11 @@ export function createTerminalPaneManagerOptions( }, resolveExternalPaneDropTarget, onExternalPaneDrop, + terminalLigaturesEnabled: () => + resolveTerminalLigaturesEnabled( + settingsRef.current?.terminalLigatures, + settingsRef.current?.terminalFontFamily + ), terminalOptions: () => { const currentSettings = settingsRef.current const terminalFontWeights = resolveTerminalFontWeights( diff --git a/src/renderer/src/lib/pane-manager/pane-lifecycle.test.ts b/src/renderer/src/lib/pane-manager/pane-lifecycle.test.ts index 3a174253d9c..e84638c33ea 100644 --- a/src/renderer/src/lib/pane-manager/pane-lifecycle.test.ts +++ b/src/renderer/src/lib/pane-manager/pane-lifecycle.test.ts @@ -7,7 +7,7 @@ import { primeTerminalWebglAddon, resetTerminalWebglSuggestion } from './pane-webgl-renderer' -import { attachLigatures, disposePane, openTerminal } from './pane-lifecycle' +import { attachLigatures, disposePane, openTerminal, setLigaturesEnabled } from './pane-lifecycle' import { ensureArabicShapingJoinerForText } from './terminal-arabic-shaping-joiner' import { buildDefaultTerminalOptions, @@ -531,6 +531,7 @@ describe('openTerminal — addon and provider wiring', () => { }), attachCustomWheelEventHandler: vi.fn(), onWriteParsed: vi.fn(() => ({ dispose: vi.fn() })), + refresh: vi.fn(), write: vi.fn(() => { events.push('write') }), @@ -588,6 +589,31 @@ describe('openTerminal — addon and provider wiring', () => { // unicode v11 is activated (still on default v6 width tables), wide chars // lay out as single cells. The bug surfaces as the broken `?`-style glyphs // users saw on worktree switch. + it('builds one initial WebGL atlas with ligatures and still rebuilds on a live toggle', async () => { + await primeTerminalWebglAddon() + resetTerminalWebglSuggestion() + vi.mocked(WebglAddon).mockClear() + webglMock.dispose.mockClear() + vi.stubGlobal('navigator', { platform: 'MacIntel', userAgent: 'Macintosh' }) + const { pane } = createOpenTerminalHarness() + pane.terminalGpuAcceleration = 'auto' + pane.gpuRenderingEnabled = true + + openTerminal(pane, true) + expect(pane.ligaturesAddon).not.toBeNull() + expect(pane.webglAddon).not.toBeNull() + const addons = vi.mocked(pane.terminal.loadAddon).mock.calls.map(([addon]) => addon) + expect(addons.indexOf(pane.ligaturesAddon!)).toBeLessThan(addons.indexOf(pane.webglAddon!)) + setLigaturesEnabled(pane, true) + expect(WebglAddon).toHaveBeenCalledTimes(1) + expect(webglMock.dispose).not.toHaveBeenCalled() + + setLigaturesEnabled(pane, false) + expect(WebglAddon).toHaveBeenCalledTimes(2) + expect(webglMock.dispose).toHaveBeenCalledTimes(1) + expect(pane.ligaturesAddon).toBeNull() + }) + it('activates unicode 11 before any caller-driven write would be possible', () => { const { pane, events } = createOpenTerminalHarness() diff --git a/src/renderer/src/lib/pane-manager/pane-lifecycle.ts b/src/renderer/src/lib/pane-manager/pane-lifecycle.ts index 3c89a6ec723..cf4b50783d1 100644 --- a/src/renderer/src/lib/pane-manager/pane-lifecycle.ts +++ b/src/renderer/src/lib/pane-manager/pane-lifecycle.ts @@ -28,7 +28,7 @@ import { installTerminalImeCandidateAnchor } from './terminal-ime-candidate-anch export { createPaneDOM } from './pane-dom-creation' /** Open terminal into its container and load addons. Must be called after the container is in the DOM. */ -export function openTerminal(pane: ManagedPaneInternal): void { +export function openTerminal(pane: ManagedPaneInternal, ligaturesEnabled = false): void { const { terminal, container, @@ -100,6 +100,10 @@ export function openTerminal(pane: ManagedPaneInternal): void { pane.focusClassSyncCleanup = attachDomRendererFocusClassSync(terminal.element) + // Configure the first atlas with ligatures instead of immediately rebuilding it. + if (ligaturesEnabled) { + attachLigatures(pane) + } if (pane.gpuRenderingEnabled) { attachWebgl(pane) } diff --git a/src/renderer/src/lib/pane-manager/pane-manager-pane-creation.ts b/src/renderer/src/lib/pane-manager/pane-manager-pane-creation.ts index 483f49a1f55..c4c3a28c4ca 100644 --- a/src/renderer/src/lib/pane-manager/pane-manager-pane-creation.ts +++ b/src/renderer/src/lib/pane-manager/pane-manager-pane-creation.ts @@ -18,7 +18,7 @@ export function createInitialManagedPane( overflow: 'hidden' }) host.root.appendChild(pane.container) - openTerminal(pane) + openTerminal(pane, host.options.terminalLigaturesEnabled?.()) host.setActivePaneId(pane.id) applyPaneOpacity(host.panes.values(), host.getActivePaneId(), host.getStyleOptions()) diff --git a/src/renderer/src/lib/pane-manager/pane-manager-types.ts b/src/renderer/src/lib/pane-manager/pane-manager-types.ts index 7b23c177236..00637ae7f97 100644 --- a/src/renderer/src/lib/pane-manager/pane-manager-types.ts +++ b/src/renderer/src/lib/pane-manager/pane-manager-types.ts @@ -63,6 +63,7 @@ export type PaneManagerOptions = { resolveExternalPaneDropTarget?: PaneExternalDropResolver onExternalPaneDrop?: PaneExternalDropHandler terminalOptions?: (paneId: number) => Partial + terminalLigaturesEnabled?: () => boolean terminalTuiScrollSensitivity?: () => number | undefined onLinkClick?: (paneId: number, event: MouseEvent | undefined, url: string) => void /** Resolved per hover so link-routing setting changes apply without recreating panes. */ diff --git a/src/renderer/src/lib/pane-manager/pane-split-close.ts b/src/renderer/src/lib/pane-manager/pane-split-close.ts index df725156655..c5c71b03be6 100644 --- a/src/renderer/src/lib/pane-manager/pane-split-close.ts +++ b/src/renderer/src/lib/pane-manager/pane-split-close.ts @@ -141,7 +141,7 @@ function openSplitPane( newPane: ManagedPaneInternal, cwd?: string ): void { - openTerminal(newPane) + openTerminal(newPane, args.managerOptions.terminalLigaturesEnabled?.()) applyPaneOpacity(args.panes.values(), newPane.id, args.styleOptions) applyDividerStyles(args.root, args.styleOptions) newPane.terminal.focus() diff --git a/src/shared/git-binary-compatibility.test.ts b/src/shared/git-binary-compatibility.test.ts index 6387f200b4c..fb8161b9f90 100644 --- a/src/shared/git-binary-compatibility.test.ts +++ b/src/shared/git-binary-compatibility.test.ts @@ -111,6 +111,33 @@ describeBinaryCompatibility('real Git binary compatibility', () => { } }) + it('quietly distinguishes present and absent branch refs', async () => { + const head = (await runGit(['rev-parse', 'HEAD'])).stdout.trim() + await runGit(['branch', 'quiet-probe-present', head]) + await expect( + runGit(['rev-parse', '--verify', '--quiet', 'refs/heads/quiet-probe-present']) + ).resolves.toMatchObject({ stdout: `${head}\n`, stderr: '' }) + await expect( + runGit(['rev-parse', '--verify', '--quiet', 'refs/heads/quiet-probe-absent']) + ).rejects.toMatchObject({ code: 1, stdout: '', stderr: '' }) + }) + + it('distinguishes an absent branch from a ref pointing at a missing object', async () => { + const missingObject = 'a'.repeat(40) + const refPath = join(repoPath, '.git', 'refs', 'heads', 'quiet-probe-dangling') + await writeFile(refPath, `${missingObject}\n`) + try { + await expect( + runGit(['rev-parse', '--verify', '--quiet', 'refs/heads/quiet-probe-dangling']) + ).resolves.toMatchObject({ stdout: `${missingObject}\n`, stderr: '' }) + await expect( + runGit(['rev-parse', '--verify', '--quiet', 'refs/heads/quiet-probe-dangling^{commit}']) + ).rejects.toMatchObject({ code: 1, stdout: '', stderr: '' }) + } finally { + await rm(refPath) + } + }) + it('recognizes worktree-list and rev-parse compatibility boundaries', async () => { await expectPreferredOrRecognizedFallback( ['worktree', 'list', '--porcelain', '-z'], diff --git a/tests/e2e/helpers/electron-crashpad-cleanup.ts b/tests/e2e/helpers/electron-crashpad-cleanup.ts new file mode 100644 index 00000000000..8ec1a98b26b --- /dev/null +++ b/tests/e2e/helpers/electron-crashpad-cleanup.ts @@ -0,0 +1,46 @@ +import { execFileSync } from 'node:child_process' +import path from 'node:path' + +function ownsCrashpad(command: string, userDataDir: string): boolean { + return ( + command.includes('/chrome_crashpad_handler ') && + command.includes(` --database=${path.join(userDataDir, 'Crashpad')} `) + ) +} + +export function cleanupE2ECrashpad(userDataDir: string): void { + if (process.platform !== 'darwin') { + return + } + + // macOS reparents Crashpad before app exit; its inherited stderr can keep Playwright open. + try { + const table = execFileSync('ps', ['-axo', 'pid=,command='], { + encoding: 'utf8', + timeout: 5_000 + }) + for (const row of table.split('\n')) { + const match = row.match(/^\s*(\d+)\s+(.+)$/) + if (!match || !ownsCrashpad(match[2], userDataDir)) { + continue + } + const pid = Number(match[1]) + if (!Number.isSafeInteger(pid) || pid <= 1) { + continue + } + try { + const command = execFileSync('ps', ['-p', String(pid), '-o', 'command='], { + encoding: 'utf8', + timeout: 5_000 + }) + if (ownsCrashpad(command, userDataDir)) { + process.kill(pid, 'SIGTERM') + } + } catch { + // The test-owned reporter may already have exited. + } + } + } catch { + // Cleanup remains best-effort when process enumeration is unavailable. + } +} diff --git a/tests/e2e/helpers/electron-crashpad-cleanup.unit.test.ts b/tests/e2e/helpers/electron-crashpad-cleanup.unit.test.ts new file mode 100644 index 00000000000..cdf552e2548 --- /dev/null +++ b/tests/e2e/helpers/electron-crashpad-cleanup.unit.test.ts @@ -0,0 +1,45 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { execFileSync } from 'node:child_process' +import path from 'node:path' +import { cleanupE2ECrashpad } from './electron-crashpad-cleanup' + +vi.mock('node:child_process', () => ({ execFileSync: vi.fn() })) + +const profile = '/tmp/test profile' +const database = path.join(profile, 'Crashpad') +const reporter = `/Electron Framework/Helpers/chrome_crashpad_handler --database=${database} --annotation=prod=Electron` + +afterEach(() => vi.restoreAllMocks()) + +describe('test-owned macOS Crashpad cleanup', () => { + it('terminates only the reporter for the exact temporary profile after rechecking ownership', () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + const kill = vi.spyOn(process, 'kill').mockReturnValue(true) + vi.mocked(execFileSync) + .mockReturnValueOnce( + `111 ${reporter}\n222 ${reporter.replace('Crashpad ', 'Crashpad-old ')}\n333 ${reporter.replace('test profile', 'another profile')}\n444 /bin/echo --database=${database} \n` + ) + .mockReturnValueOnce(reporter) + cleanupE2ECrashpad(profile) + expect(kill).toHaveBeenCalledExactlyOnceWith(111, 'SIGTERM') + expect(execFileSync).toHaveBeenLastCalledWith('ps', ['-p', '111', '-o', 'command='], { + encoding: 'utf8', + timeout: 5_000 + }) + }) + + it('does not signal a PID whose ownership changed after enumeration', () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + const kill = vi.spyOn(process, 'kill').mockReturnValue(true) + vi.mocked(execFileSync).mockReturnValueOnce(`111 ${reporter}`).mockReturnValueOnce('/bin/sh') + cleanupE2ECrashpad(profile) + expect(kill).not.toHaveBeenCalled() + }) + + it.each(['win32', 'linux'] as const)('does not enumerate processes on %s', (platform) => { + vi.spyOn(process, 'platform', 'get').mockReturnValue(platform) + vi.mocked(execFileSync).mockClear() + cleanupE2ECrashpad(profile) + expect(execFileSync).not.toHaveBeenCalled() + }) +}) diff --git a/tests/e2e/helpers/electron-process-shutdown.ts b/tests/e2e/helpers/electron-process-shutdown.ts index 48ddb60bf43..f9b642a676e 100644 --- a/tests/e2e/helpers/electron-process-shutdown.ts +++ b/tests/e2e/helpers/electron-process-shutdown.ts @@ -2,6 +2,7 @@ import type { ChildProcess } from 'node:child_process' import { execFileSync } from 'node:child_process' import { existsSync, readFileSync, readdirSync } from 'node:fs' import path from 'node:path' +import { cleanupE2ECrashpad } from './electron-crashpad-cleanup' import type { ElectronApplication } from '@stablyai/playwright-test' const GRACEFUL_CLOSE_TIMEOUT_MS = 10_000 @@ -238,4 +239,5 @@ export async function cleanupE2EDaemons(userDataDir: string): Promise { for (const pid of readDaemonPidFiles(userDataDir)) { await forceKillPidTree(pid) } + cleanupE2ECrashpad(userDataDir) } diff --git a/tests/e2e/helpers/ssh-recovery-input-observation.ts b/tests/e2e/helpers/ssh-recovery-input-observation.ts new file mode 100644 index 00000000000..06b4bd7de3e --- /dev/null +++ b/tests/e2e/helpers/ssh-recovery-input-observation.ts @@ -0,0 +1,53 @@ +import type { Page, TestInfo } from '@playwright/test' +import type { RuntimeTerminalListResult } from '../../../src/shared/runtime-types' + +export async function attachSshRecoveryInputObservation( + page: Page, + testInfo: TestInfo, + targetId: string, + originalPtyId: string, + label: string +): Promise { + const observation = await page.evaluate( + async ({ targetId, originalPtyId }) => { + const state = window.__store?.getState() + const panes = [...(window.__paneManagers?.entries() ?? [])].flatMap(([tabId, manager]) => + manager.getPanes().map((pane) => ({ + tabId, + leafId: pane.leafId, + ptyId: pane.container.dataset.ptyId, + active: manager.getActivePane()?.id === pane.id + })) + ) + let timer: ReturnType | undefined + try { + const runtime = await Promise.race([ + window.api.runtime + .call({ method: 'terminal.list', params: { limit: 50, includeVisualLayouts: false } }) + .then((response) => + response.ok + ? { terminals: (response.result as RuntimeTerminalListResult).terminals } + : { error: response.error } + ), + new Promise<{ error: string }>((resolve) => { + timer = setTimeout(() => resolve({ error: 'Observation timed out' }), 1000) + }) + ]) + return { + originalPtyId, + authority: state?.sshConnectionStates.get(targetId), + activeWorktreeId: state?.activeWorktreeId, + panes, + runtime + } + } finally { + clearTimeout(timer) + } + }, + { targetId, originalPtyId } + ) + await testInfo.attach(`ssh-input-${label}.json`, { + body: JSON.stringify(observation, null, 2), + contentType: 'application/json' + }) +} diff --git a/tests/e2e/ssh-docker-transport-drop-recovery.spec.ts b/tests/e2e/ssh-docker-transport-drop-recovery.spec.ts index 42ac316790c..c64761ede80 100644 --- a/tests/e2e/ssh-docker-transport-drop-recovery.spec.ts +++ b/tests/e2e/ssh-docker-transport-drop-recovery.spec.ts @@ -4,6 +4,7 @@ import type { ElectronApplication } from '@playwright/test' import { test, expect } from './helpers/orca-app' import { DEFAULT_LOCAL_ORCA_PROFILE_ID } from '../../src/shared/orca-profiles' import { sshRemotePtyLeaseAllowsReattach, type SshRemotePtyLease } from '../../src/shared/ssh-types' +import { toRelaySshPtyId } from '../../src/shared/ssh-pty-id' import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' import { execInTerminal, @@ -29,6 +30,8 @@ import { withStalledDockerSshRelayTarget } from './helpers/docker-ssh-relay-faults' +import { attachSshRecoveryInputObservation } from './helpers/ssh-recovery-input-observation' + const RUN_DOCKER_SSH = process.env.ORCA_E2E_SSH_DOCKER === '1' /** @@ -338,16 +341,21 @@ test.describe('SSH transport drop recovery', () => { const generations: string[][] = [] for (let generation = 1; generation <= 5; generation++) { - const predecessor = await waitForActivePanePtyId(orcaPage, 60_000) + const previousPtyId = await waitForActivePanePtyId(orcaPage, 60_000) await recoverDockerSshRelayAfterFault(orcaPage, remote.targetId, () => { - expect(killDockerSshRelayDaemon(target!)).toBeGreaterThan(0) + expect( + killDockerSshRelayDaemon(target!), + 'no relay process was found to kill' + ).toBeGreaterThan(0) }) - await expect - .poll(() => waitForActivePanePtyId(orcaPage, 60_000), { timeout: 120_000 }) - .not.toBe(predecessor) await waitForActiveTerminalManager(orcaPage, 120_000) - // The pane must be usable again before the count is meaningful: recovery is what mints the - // successor lease that retires the generation before it. + // Transport status can still be connected while the pane retains its old binding. + await expect + .poll(() => waitForActivePanePtyId(orcaPage, 60_000).catch(() => previousPtyId), { + timeout: 120_000, + message: `pane kept its old PTY binding after relay kill ${generation}` + }) + .not.toBe(previousPtyId) const ptyId = await waitForActivePanePtyId(orcaPage, 120_000) const markerSuffix = `${generation}_${Date.now()}` const marker = `LEASE_GEN_${markerSuffix}` @@ -356,15 +364,14 @@ test.describe('SSH transport drop recovery', () => { try { await expect - .poll(() => readReattachablePtyIds(userDataDir, remote.targetId).length, { + .poll(() => readReattachablePtyIds(userDataDir, remote.targetId), { timeout: 60_000 }) - .toBe(1) + .toEqual([toRelaySshPtyId(remote.targetId, ptyId)]) } catch (error) { - // Why re-thrown with the rows: the count alone cannot say WHICH predecessor stayed - // reattachable, and the user-data dir is torn down before the report is read. + // Preserve lease ownership diagnostics before the user-data directory is removed. throw new Error( - `reattachable lease count never settled at 1 in generation ${generation}; leases: ${describeSshLeases(userDataDir, remote.targetId)}`, + `reattachable leases never settled at the active PTY ${ptyId} in generation ${generation}; leases: ${describeSshLeases(userDataDir, remote.targetId)}`, { cause: error } ) } @@ -432,6 +439,7 @@ test.describe('SSH transport drop recovery', () => { test('accepts input again after a frozen host resumes', async ({ orcaPage }, testInfo) => { test.slow() let target: DockerSshRelayTarget | null = null + let observationTarget: { targetId: string; ptyId: string } | undefined try { target = startDockerSshRelayTarget(testInfo) enableDockerSshRelayTargetShellTitle(target) @@ -444,6 +452,18 @@ test.describe('SSH transport drop recovery', () => { await waitForActiveTerminalManager(orcaPage, 60_000) const ptyId = await waitForActivePanePtyId(orcaPage, 60_000) + observationTarget = { targetId: remote.targetId, ptyId } + const beforeSuffix = Date.now() + await execInTerminal(orcaPage, ptyId, `printf 'STALL_BEFORE_%s\\n' ${beforeSuffix}`) + await waitForTerminalOutput(orcaPage, `STALL_BEFORE_${beforeSuffix}`, 60_000) + await attachSshRecoveryInputObservation( + orcaPage, + testInfo, + remote.targetId, + ptyId, + 'before-freeze' + ) + await recoverDockerSshRelayAfterFault(orcaPage, remote.targetId, async () => { await withStalledDockerSshRelayTarget(target!, async () => { await orcaPage.waitForTimeout(30_000) @@ -454,7 +474,25 @@ test.describe('SSH transport drop recovery', () => { const afterSuffix = Date.now() const afterMarker = `STALL_AFTER_${afterSuffix}` await execInTerminal(orcaPage, ptyId, `printf 'STALL_AFTER_%s\\n' ${afterSuffix}`) + await attachSshRecoveryInputObservation( + orcaPage, + testInfo, + remote.targetId, + ptyId, + 'after-write' + ) await waitForTerminalOutput(orcaPage, afterMarker, 60_000) + } catch (error) { + if (observationTarget) { + await attachSshRecoveryInputObservation( + orcaPage, + testInfo, + observationTarget.targetId, + observationTarget.ptyId, + 'failure-before-cleanup' + ).catch(() => undefined) + } + throw error } finally { if (target) { clearDockerSshRelayFaults(target) From 9faa27c5f4e3f476393aff1e17429242483e8038 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 15:12:19 -0700 Subject: [PATCH 03/23] test: align desktop platform oracles with native behavior (#18915) --- .../right-sidebar-windows-titlebar.spec.ts | 42 ++++-------- tests/e2e/settings-agent-awake.spec.ts | 68 ++++++++++++++----- 2 files changed, 64 insertions(+), 46 deletions(-) diff --git a/tests/e2e/right-sidebar-windows-titlebar.spec.ts b/tests/e2e/right-sidebar-windows-titlebar.spec.ts index 1d6d4b8981f..ce39d7de28d 100644 --- a/tests/e2e/right-sidebar-windows-titlebar.spec.ts +++ b/tests/e2e/right-sidebar-windows-titlebar.spec.ts @@ -6,41 +6,19 @@ type RightSidebarHeaderGeometry = { stripTop: number closeTop: number titlebarActivityButtonCount: number + activityButtonCount: number firstButtonCenterHitsFirst: boolean lastButtonCenterHitsLast: boolean } -test.describe('Right sidebar Windows titlebar spacing', () => { - test('top activity buttons render inside the sidebar instead of the titlebar', async ({ - orcaPage - }) => { - await orcaPage.addInitScript(() => { - const userAgent = - 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/146 Safari/537.36' - Object.defineProperty(navigator, 'userAgent', { - get: () => userAgent, - configurable: true - }) - }) - await orcaPage.reload({ waitUntil: 'domcontentloaded' }) - await orcaPage.waitForFunction(() => Boolean(window.__store), null, { timeout: 30_000 }) +test.describe('Right sidebar native titlebar spacing', () => { + test('top activity buttons follow the native desktop chrome layout', async ({ orcaPage }) => { await waitForSessionReady(orcaPage) await waitForActiveWorktree(orcaPage) await ensureTerminalVisible(orcaPage) - await expect - .poll( - async () => - orcaPage.evaluate(() => ({ - hasWindowsUserAgent: navigator.userAgent.includes('Windows'), - hasWindowsTitlebarChrome: Boolean(document.querySelector('.window-controls')) - })), - { - timeout: 5_000, - message: 'Renderer did not switch to the Windows titlebar branch' - } - ) - .toEqual({ hasWindowsUserAgent: true, hasWindowsTitlebarChrome: true }) + const hasDesktopWindowChrome = process.platform !== 'darwin' + expect(await orcaPage.evaluate(() => window.api.platform.get().platform)).toBe(process.platform) await orcaPage.evaluate(() => { const store = window.__store @@ -95,6 +73,7 @@ test.describe('Right sidebar Windows titlebar spacing', () => { stripTop: stripRect.top, closeTop: closeRect.top, titlebarActivityButtonCount, + activityButtonCount: activityButtons.length, firstButtonCenterHitsFirst: elementAtFirstCenter !== null && firstButton.contains(elementAtFirstCenter), lastButtonCenterHitsLast: @@ -117,8 +96,13 @@ test.describe('Right sidebar Windows titlebar spacing', () => { .toBe(true) expect(headerGeometry).not.toBeNull() - expect(headerGeometry!.titlebarActivityButtonCount).toBe(0) - expect(headerGeometry!.stripTop).toBeGreaterThanOrEqual(headerGeometry!.headerBottom) + if (hasDesktopWindowChrome) { + expect(headerGeometry!.titlebarActivityButtonCount).toBe(0) + expect(headerGeometry!.stripTop).toBeGreaterThanOrEqual(headerGeometry!.headerBottom) + } else { + expect(headerGeometry!.titlebarActivityButtonCount).toBe(headerGeometry!.activityButtonCount) + expect(headerGeometry!.stripTop).toBeLessThan(headerGeometry!.headerBottom) + } expect(headerGeometry!.closeTop).toBeLessThan(headerGeometry!.headerBottom) expect(headerGeometry!.firstButtonCenterHitsFirst).toBe(true) expect(headerGeometry!.lastButtonCenterHitsLast).toBe(true) diff --git a/tests/e2e/settings-agent-awake.spec.ts b/tests/e2e/settings-agent-awake.spec.ts index 8a2ad840a14..ebea82a1241 100644 --- a/tests/e2e/settings-agent-awake.spec.ts +++ b/tests/e2e/settings-agent-awake.spec.ts @@ -1,4 +1,5 @@ import { randomUUID } from 'node:crypto' +import { runProcess } from '../../src/shared/child-process/run-process' import type { ElectronApplication, Page } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' import { waitForSessionReady } from './helpers/store' @@ -104,6 +105,19 @@ async function readPowerSaveBlockerProbe( }) } +async function readMacosSleepAssertionPids(electronApp: ElectronApplication): Promise { + const result = await runProcess({ + program: '/usr/bin/pgrep', + args: ['-P', String(electronApp.process().pid), '-f', '^/usr/bin/caffeinate -i -s$'], + maxOutputBytes: 4_096 + }) + if (result.code === 1) { + return [] + } + expect(result.code, result.stderr).toBe(0) + return result.stdout.trim().split(/\s+/).filter(Boolean).map(Number) +} + async function postCodexHookEvent( electronApp: ElectronApplication, options: { @@ -176,7 +190,9 @@ test.describe('Agent awake setting', () => { electronApp, orcaPage }) => { - await installPowerSaveBlockerProbe(electronApp) + if (process.platform !== 'darwin') { + await installPowerSaveBlockerProbe(electronApp) + } await setKeepAwake(orcaPage, true) const tabId = 'e2e-awake-tab' @@ -187,24 +203,33 @@ test.describe('Agent awake setting', () => { eventName: 'UserPromptSubmit' }) - await expect - .poll(async () => await readPowerSaveBlockerProbe(electronApp), { - timeout: 5_000, - message: 'powerSaveBlocker did not start for the working agent' - }) - .toEqual( - expect.objectContaining({ - activeIds: expect.arrayContaining([expect.any(Number)]), - starts: expect.arrayContaining([ - expect.objectContaining({ type: 'prevent-display-sleep' }) - ]) + await expect( + orcaPage.getByRole('button', { name: 'Keep computer awake, Agent · Active' }) + ).toBeVisible() + let startedIds: number[] = [] + if (process.platform === 'darwin') { + // macOS uses an app-owned caffeinate assertion instead of Electron's display blocker. + await expect + .poll(() => readMacosSleepAssertionPids(electronApp), { timeout: 5_000 }) + .not.toEqual([]) + } else { + await expect + .poll(async () => await readPowerSaveBlockerProbe(electronApp), { + timeout: 5_000, + message: 'powerSaveBlocker did not start for the working agent' }) - ) + .toEqual( + expect.objectContaining({ + activeIds: expect.arrayContaining([expect.any(Number)]), + starts: expect.arrayContaining([ + expect.objectContaining({ type: 'prevent-display-sleep' }) + ]) + }) + ) - const startedIds = (await readPowerSaveBlockerProbe(electronApp)).starts.map( - (start) => start.id - ) - expect(startedIds.length).toBeGreaterThan(0) + startedIds = (await readPowerSaveBlockerProbe(electronApp)).starts.map((start) => start.id) + expect(startedIds.length).toBeGreaterThan(0) + } await postCodexHookEvent(electronApp, { paneKey, @@ -212,6 +237,15 @@ test.describe('Agent awake setting', () => { eventName: 'Stop' }) + await expect( + orcaPage.getByRole('button', { name: 'Keep computer awake, Agent · Inactive' }) + ).toBeVisible() + if (process.platform === 'darwin') { + await expect + .poll(() => readMacosSleepAssertionPids(electronApp), { timeout: 5_000 }) + .toEqual([]) + return + } await expect .poll(async () => await readPowerSaveBlockerProbe(electronApp), { timeout: 5_000, From 239e3c7e0ba5b41545f440089f76e4b696385abd Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 15:29:46 -0700 Subject: [PATCH 04/23] test: select seeded workspace and confirm sidebar reveal (#18921) --- tests/e2e/worktree-scroll-to-current.spec.ts | 24 +++++++++++++++----- 1 file changed, 18 insertions(+), 6 deletions(-) diff --git a/tests/e2e/worktree-scroll-to-current.spec.ts b/tests/e2e/worktree-scroll-to-current.spec.ts index 19c51005cfe..d61d96847a0 100644 --- a/tests/e2e/worktree-scroll-to-current.spec.ts +++ b/tests/e2e/worktree-scroll-to-current.spec.ts @@ -39,22 +39,30 @@ test.describe('Reveal active workspace button', () => { // the "outside the virtualized window" test below. test('clears sidebar filters before revealing a hidden current workspace', async ({ - orcaPage + orcaPage, + testRepoPath }) => { await prepareSidebarForScrollTest(orcaPage) - const renderedOptions = orcaPage.locator('[data-worktree-sidebar] [role="option"]') - await expect(renderedOptions).toHaveCount(2) - - const targetId = await renderedOptions.last().getAttribute('data-worktree-id') + // Other specs can add worktrees to the shared repository before this test runs. + const targetId = await orcaPage.evaluate((repoPath) => { + const state = window.__store!.getState() + const repo = state.repos.find((candidate) => candidate.path === repoPath) + return repo + ? state.worktreesByRepo[repo.id]?.find( + (worktree) => worktree.branch === 'refs/heads/e2e-secondary' + )?.id + : undefined + }, testRepoPath) if (!targetId) { - throw new Error('Bottom workspace row did not expose a data-worktree-id') + throw new Error('Seeded secondary worktree is missing') } const targetRows = orcaPage.locator( `[data-worktree-sidebar] [data-worktree-id=${JSON.stringify(targetId)}]` ) const targetRow = targetRows.first() + await expect(targetRows.and(orcaPage.getByRole('option'))).toHaveCount(1) const revealButton = orcaPage.getByRole('button', { name: 'Reveal active workspace' }) await orcaPage.evaluate((targetId) => { @@ -92,6 +100,10 @@ test.describe('Reveal active workspace button', () => { // contract under test is that reveal clears the filter (asserted below). await revealButton.click() + await orcaPage + .getByRole('dialog', { name: 'Reveal hidden workspace?' }) + .getByRole('button', { name: 'Clear filters and reveal' }) + .click() await expect(targetRow).toBeVisible() await expect(targetRow).toHaveAttribute('data-scroll-reveal-highlight', 'true') From 471a5f4aa795aaca11ea1a1a7b8dc17bd1330915 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Sat, 5 Sep 2026 15:33:04 -0700 Subject: [PATCH 05/23] feat(native-chat): model Codex MCP and web-search items instead of leaking opcodes (#18763) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(native-chat): model Codex MCP and web-search items instead of leaking opcodes Codex's app-server sends 19 thread-item types; the structured translator handled six. The rest fell through to a generic gray `codex · item:` row, even though the disposition table's own comment says it exists so a new item type cannot leak like that — the table had one entry. Give `mcpToolCall` and `webSearch` real tool-call bodies, and chrome `sleep`, which carries only a duration and renders as nothing in Codex's own TUI. `subAgentActivity` and `collabAgentToolCall` deliberately keep their generic rows. They arrive in real sessions today and are currently the only visible sign a subagent is running; hiding them before the subagent UI lands would render minutes of work as an idle turn. Tests pin that they stay visible. MCP tool names pass through verbatim when they contain `:`, `.`, `/` or `__`, so `mcp__server__tool` survives instead of being title-cased into nonsense. * fix(native-chat): keep Codex MCP tool identity and web-search results on the row Four fixes to the Codex MCP / web-search item bodies: - Drop the title-casing display name. `get_forecast` became `Get Forecast`, which no longer matches the raw snake_case identifiers that the diff renderer, question parsers, and tool-input previews dispatch on, and does not match how the Claude lane or the sibling `shell`/`apply_patch`/`web_search` bodies name a tool. The row name is now `server/tool` verbatim, the bare `tool` when no server is given, and `mcp` when the item names no tool at all. Server-qualifying also stops an MCP tool that happens to be called `apply_patch` from hijacking the diff renderer. - Pass the MCP call's own `arguments` as the tool input instead of wrapping it in `{server, tool, arguments}`. Row-label derivation only reads top-level keys, so the wrapper degraded every MCP row to a truncated raw JSON blob. A non-object `arguments` stays addressable under a key rather than being dropped; an absent one becomes null, which labels as empty rather than `{}`. - Carry a web search's `results` as the call output, bounded like every other inline payload and omitted when there are none. They were being dropped entirely, which showed less than the generic fallback row it replaced. - No streaming branches were added for these two item types: the Codex delta stream is a closed set of six methods that neither can reach, so such branches would be unreachable. * fix(native-chat): label Codex web searches and argument-less MCP calls A row label is derived from top-level `input` keys only, so a webSearch whose detail lives inside `action` — an opened page, an in-page find, or a bare `other` — fell through to the raw JSON of the whole input, as did the empty `query` Codex leaves on a completed search. Hoist the action's `url`, `pattern` and `type` beside the query, keep the full `action` object so the expanded detail loses nothing, and emit no input at all for the start frame. An MCP tool that takes no arguments sends `arguments: {}`, which passed straight through and labelled the row a literal `{}`; treat it as absent so the row reads as a bare `server/tool`. Split the durable-identity half of the item translator into `codex-thread-item-identity.ts`, re-exported so every existing import is unchanged, to keep both files under the max-lines cap. --------- Co-authored-by: Merge Sim --- src/main/codex/codex-command-action-class.ts | 71 +++++ src/main/codex/codex-item-field-readers.ts | 45 +++ .../codex-structured-item-translation.test.ts | 235 ++++++++++++++- .../codex-structured-item-translation.ts | 278 +++++++----------- src/main/codex/codex-thread-item-identity.ts | 65 ++++ .../provider-frame-disposition.test.ts | 48 +++ .../provider-frame-disposition.ts | 7 +- 7 files changed, 567 insertions(+), 182 deletions(-) create mode 100644 src/main/codex/codex-command-action-class.ts create mode 100644 src/main/codex/codex-item-field-readers.ts create mode 100644 src/main/codex/codex-thread-item-identity.ts diff --git a/src/main/codex/codex-command-action-class.ts b/src/main/codex/codex-command-action-class.ts new file mode 100644 index 00000000000..81360691ef9 --- /dev/null +++ b/src/main/codex/codex-command-action-class.ts @@ -0,0 +1,71 @@ +import { readRecord, readString } from './codex-item-field-readers' +import type { CodexThreadItem } from './codex-thread-item-identity' + +/** + * Codex's own classification of a shell call: the tool name to show, and the + * fields worth lifting into `input` for the shared label helper (a file target, + * a search term, a scanned root). A `Map`, not an object — an object index + * answers `__proto__` with a truthy non-string. Every other action type stays an + * unclassified `shell` row. + * + * Nothing is invented for a field Codex sends as null: a stand-in path is a + * claim about a target, and the label helper turns any path into a file link. + */ +type CommandActionClass = { + name: string + /** Action field to the `input` key it lifts to. A scan root and a listed + * directory lift to `directory`, never `path`: the label helper reads `path` + * as a file target, which mobile turns into a tappable open-file link. */ + keys: Readonly> +} + +const COMMAND_ACTION_CLASSES = new Map([ + ['read', { name: 'read', keys: { path: 'path' } }], + ['search', { name: 'search', keys: { query: 'query', path: 'directory' } }], + ['listFiles', { name: 'list', keys: { path: 'directory' } }] +]) + +/** The one class every classified `commandActions` entry agrees on, with the + * fields they all agree on; null leaves the row exactly as a Codex that sends no + * classification renders it. `cat a.txt && ls src` classifies as two different + * things, and naming that row after either would drop the other, so it stays a + * `shell` row that shows the whole command. */ +export function commandActionFacts( + item: CodexThreadItem +): { name: string; fields: Record } | null { + const actions = item.commandActions + if (!Array.isArray(actions)) { + return null + } + let matched: { class: CommandActionClass; fields: Record } | null = null + for (const action of actions) { + const record = readRecord(action) + const type = readString(record, 'type') + const classified = type === null ? undefined : COMMAND_ACTION_CLASSES.get(type) + if (classified === undefined) { + continue + } + if (matched === null) { + const fields: Record = {} + for (const [source, lifted] of Object.entries(classified.keys)) { + const value = readString(record, source) + if (value !== null) { + fields[lifted] = value + } + } + matched = { class: classified, fields } + continue + } + if (matched.class.name !== classified.name) { + return null + } + // The same class twice keeps the class, but only a target both entries name. + for (const [source, lifted] of Object.entries(matched.class.keys)) { + const kept = matched.fields[lifted] + if (kept !== undefined && readString(record, source) !== kept) { + delete matched.fields[lifted] + } + } + } + return matched === null ? null : { name: matched.class.name, fields: matched.fields } +} diff --git a/src/main/codex/codex-item-field-readers.ts b/src/main/codex/codex-item-field-readers.ts new file mode 100644 index 00000000000..bbe2551615a --- /dev/null +++ b/src/main/codex/codex-item-field-readers.ts @@ -0,0 +1,45 @@ +// Field readers for the loosely-typed records Codex sends on thread items. + +export function readRecord(value: unknown): Record { + return typeof value === 'object' && value !== null ? (value as Record) : {} +} + +export function readString(source: Record, key: string): string | null { + const value = source[key] + return typeof value === 'string' && value.length > 0 ? value : null +} + +export function readFirstString( + source: Record, + keys: readonly string[] +): string | null { + for (const key of keys) { + const value = readString(source, key) + if (value !== null) { + return value + } + } + return null +} + +export function readTextContent(source: Record, key: string): string | null { + const direct = readString(source, key) + if (direct) { + return direct + } + const value = source[key] + if (!Array.isArray(value)) { + return null + } + const parts = value.flatMap((part) => { + if (typeof part === 'string') { + return part.length > 0 ? [part] : [] + } + if (typeof part !== 'object' || part === null) { + return [] + } + const text = readString(part as Record, 'text') + return text ? [text] : [] + }) + return parts.length > 0 ? parts.join('\n') : null +} diff --git a/src/main/codex/codex-structured-item-translation.test.ts b/src/main/codex/codex-structured-item-translation.test.ts index 1d64158cb60..2558f4b60de 100644 --- a/src/main/codex/codex-structured-item-translation.test.ts +++ b/src/main/codex/codex-structured-item-translation.test.ts @@ -1,6 +1,10 @@ import { describe, expect, it } from 'vitest' import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' -import { createToolInputDisplay } from '../../shared/native-chat-tool-summary' +import { + briefToolArg, + createToolInputDisplay, + describeToolInput +} from '../../shared/native-chat-tool-summary' import { codexItemBody, codexItemIdentity, @@ -14,6 +18,13 @@ import { type CodexThreadItem } from './codex-structured-item-translation' +/** The tool-call input a Codex item lands on, which is what the row label and + * the collapsed run header are both derived from. */ +function toolCallInput(item: CodexThreadItem): unknown { + const body = codexItemBody(item) + return body !== null && body.kind === 'tool-call' ? body.input : null +} + const THREAD_ID = 'thread-abc' const TURN_ID = 'turn-1' @@ -572,10 +583,226 @@ describe('codex item bodies', () => { }) expect(codexItemBody({ type: 'reasoning', id: 'r' })).toBeNull() expect(codexItemBody({ type: 'agentMessage', id: 'm', text: '' })).toBeNull() - expect(codexItemBody({ type: 'webSearch', id: 'w' })).toMatchObject({ + expect(codexItemBody({ type: 'somethingCodexAddedLater', id: 'x' })).toMatchObject({ kind: 'status', - text: 'codex · item:webSearch', - providerFrame: { provider: 'codex', kind: 'item:webSearch' } + text: 'codex · item:somethingCodexAddedLater', + providerFrame: { provider: 'codex', kind: 'item:somethingCodexAddedLater' } + }) + }) + + it('gives an mcp tool call a typed body with its own arguments as input', () => { + expect( + codexItemBody({ + type: 'mcpToolCall', + id: 'mcp-1', + server: 'weather', + tool: 'get_forecast', + status: 'completed', + arguments: { city: 'Oslo' }, + result: { content: [{ type: 'text', text: '12C' }] } + }) + ).toEqual({ + kind: 'tool-call', + // Server-qualified, and the arguments stay top level so the row label can + // read `query`/`command`/`file_path` out of them. + name: 'weather/get_forecast', + input: { city: 'Oslo' }, + state: 'completed', + output: { head: '12C', byteLength: 3, truncated: false, digest: expect.any(String) } + }) + }) + + it('passes an mcp tool name through with no casing transform', () => { + // Downstream dispatch is exact-match on raw identifiers, so every shape — + // bare snake_case included — has to survive byte-identical. + for (const tool of ['get_forecast', 'mcp__server__tool', 'ns.tool', 'urn:tool', 'listTools']) { + expect( + codexItemBody({ type: 'mcpToolCall', id: 'm', tool, status: 'inProgress' }), + tool + ).toMatchObject({ kind: 'tool-call', name: tool, state: 'running' }) + expect( + codexItemBody({ type: 'mcpToolCall', id: 'm', server: 'srv', tool, status: 'inProgress' }), + tool + ).toMatchObject({ kind: 'tool-call', name: `srv/${tool}`, state: 'running' }) + } + }) + + it('falls back to the bare tool, then to `mcp`, when the item is under-specified', () => { + expect( + codexItemBody({ type: 'mcpToolCall', id: 'm', tool: 'get_forecast', status: 'inProgress' }) + ).toMatchObject({ name: 'get_forecast' }) + expect( + codexItemBody({ type: 'mcpToolCall', id: 'm', server: '', tool: 'ping', status: 'completed' }) + ).toMatchObject({ name: 'ping' }) + expect( + codexItemBody({ type: 'mcpToolCall', id: 'm', server: 'weather', status: 'completed' }) + ).toMatchObject({ name: 'mcp' }) + }) + + it('keeps non-object mcp arguments addressable and empty ones off the label', () => { + // `arguments` is arbitrary JSON upstream; a scalar or array must still reach + // the row rather than being dropped or unwrapped into a bare value. + expect( + codexItemBody({ type: 'mcpToolCall', id: 'm', tool: 't', arguments: 'raw text' }) + ).toMatchObject({ input: { arguments: 'raw text' } }) + expect( + codexItemBody({ type: 'mcpToolCall', id: 'm', tool: 't', arguments: [1, 2] }) + ).toMatchObject({ input: { arguments: [1, 2] } }) + // `arguments` is required on the wire, so `{}` — not an absent key — is what + // an argument-less MCP tool sends, and passing it through labels the row `{}`. + expect(codexItemBody({ type: 'mcpToolCall', id: 'm', tool: 't', arguments: {} })).toEqual({ + kind: 'tool-call', + name: 't', + input: null, + state: 'running' + }) + expect(codexItemBody({ type: 'mcpToolCall', id: 'm', tool: 't' })).toMatchObject({ + input: null + }) + expect( + codexItemBody({ type: 'mcpToolCall', id: 'm', tool: 't', arguments: null }) + ).toMatchObject({ input: null }) + }) + + it('renders an argument-less mcp call as a bare server/tool row', () => { + const input = toolCallInput({ + type: 'mcpToolCall', + id: 'm', + server: 'srv', + tool: 'list_tools', + arguments: {} + }) + expect(describeToolInput(input)).toBe('') + expect(briefToolArg(input)).toBe('') + }) + + it('reports an mcp error as a failed call carrying the server message', () => { + expect( + codexItemBody({ + type: 'mcpToolCall', + id: 'mcp-2', + server: 's', + tool: 'ping', + status: 'completed', + error: { message: 'server unreachable' } + }) + ).toMatchObject({ + kind: 'tool-call', + name: 's/ping', + state: 'failed', + output: { head: 'server unreachable', truncated: false } + }) + }) + + it('models a web search as a tool call that runs until codex sends the action', () => { + // The start frame Codex actually emits: empty query, no action. Nothing is + // labelable yet, so the input is absent rather than a hull of null keys. + expect(codexItemBody({ type: 'webSearch', id: 'w', query: '', action: null })).toEqual({ + kind: 'tool-call', + name: 'web_search', + input: null, + state: 'running' + }) + expect( + codexItemBody({ + type: 'webSearch', + id: 'w', + query: 'orca release notes', + action: { type: 'search', query: 'orca release notes', queries: null }, + results: null + }) + ).toEqual({ + kind: 'tool-call', + name: 'web_search', + input: { + query: 'orca release notes', + description: 'search', + action: { type: 'search', query: 'orca release notes', queries: null } + }, + state: 'completed' + }) + }) + + it('carries the web search hits as the call output', () => { + const results = [{ title: 'Orca 1.0', url: 'https://example.com/notes' }] + expect( + codexItemBody({ + type: 'webSearch', + id: 'w', + query: 'orca release notes', + action: { type: 'search', query: 'orca release notes', queries: null }, + results + }) + ).toMatchObject({ + kind: 'tool-call', + name: 'web_search', + state: 'completed', + output: { head: JSON.stringify(results), truncated: false } + }) + // Nothing to show is no output block at all, not an empty one. + for (const empty of [undefined, null, []]) { + expect( + codexItemBody({ + type: 'webSearch', + id: 'w', + query: 'q', + action: { type: 'search' }, + results: empty + }), + String(empty) + ).not.toHaveProperty('output') + } + }) + + it('labels every web search shape without falling back to raw JSON', () => { + // Both the row label and the run header read top-level input keys only, so a + // shape whose detail sits inside `action` renders as the input's raw JSON. + const url = 'https://example.com/docs/page' + const shapes: [string, unknown, string, string][] = [ + ['started', null, '', ''], + [ + 'search', + { type: 'search', query: 'a sample query', queries: null }, + 'a sample query', + 'a sample query' + ], + ['openPage', { type: 'openPage', url }, url, ''], + [ + 'findInPage', + { type: 'findInPage', url, pattern: 'a needle' }, + 'a sample query', + 'a sample query' + ], + ['other', { type: 'other' }, 'other', ''] + ] + for (const [name, action, label, brief] of shapes) { + // Codex leaves the item's own `query` empty on most completed searches. + const query = name === 'search' || name === 'findInPage' ? 'a sample query' : '' + const input = toolCallInput({ type: 'webSearch', id: 'w', query, action }) + expect(describeToolInput(input), name).toBe(label) + expect(briefToolArg(input), name).toBe(brief) + } + }) + + it('leaves subagent items on the generic row until a real renderer exists', () => { + expect( + codexJournalItem({ + type: 'subAgentActivity', + id: 'a-1', + kind: 'started', + agentThreadId: 'thread-child', + agentPath: '/root/list_directory' + }) + ).toMatchObject({ + handled: false, + body: { kind: 'status', providerFrame: { kind: 'item:subAgentActivity' } } + }) + }) + + it('drops the sleep item, which codex itself renders as nothing', () => { + expect(codexJournalItem({ type: 'sleep', id: 's-1', durationMs: 20_000 })).toEqual({ + body: null, + handled: true }) }) diff --git a/src/main/codex/codex-structured-item-translation.ts b/src/main/codex/codex-structured-item-translation.ts index b3609e076f5..ad08525a5f5 100644 --- a/src/main/codex/codex-structured-item-translation.ts +++ b/src/main/codex/codex-structured-item-translation.ts @@ -1,7 +1,4 @@ -import type { - AgentJournalItemBody, - AgentJournalItemIdentity -} from '../../shared/agent-session-journal-types' +import type { AgentJournalItemBody } from '../../shared/agent-session-journal-types' import type { NativeChatBlock } from '../../shared/native-chat-types' import { boundInlineText, @@ -9,116 +6,27 @@ import { DEFAULT_JOURNAL_PAYLOAD_LIMITS } from '../native-chat/agent-session-journal/journal-payload-bounds' import { unhandledProviderFrameJournalItem } from '../native-chat/agent-session-wire/unhandled-provider-frame' -import type { CodexTurnOrdinals } from './codex-turn-ordinals' +import { commandActionFacts } from './codex-command-action-class' +import { + readFirstString, + readRecord, + readString, + readTextContent +} from './codex-item-field-readers' +import type { CodexThreadItem } from './codex-thread-item-identity' +export { + codexItemIdentity, + isCodexMessageItemType, + readCodexThreadItem, + type CodexThreadItem +} from './codex-thread-item-identity' export { CodexTurnOrdinals, MAX_CODEX_TURN_ORDINAL_BYTES, MAX_CODEX_TURN_ORDINAL_ENTRIES } from './codex-turn-ordinals' -// Codex thread items → journal item bodies and durable identities. -// -// THE ORDINAL RULE, and why it is not "index within the turn". Codex renumbers -// item ids positionally on resume (`item-1`…`item-N` across the whole thread), -// and a resumed turn does NOT contain every item the live turn emitted — -// reasoning and command execution are dropped from persisted history. Numbering -// by live position would therefore shift every message after the first tool -// call and hand the user a duplicate of the assistant's answer after a resume. -// -// So the ordinal counts MESSAGE items only, and the same projection is applied -// to the live stream and to a resumed turn's item list. Any other item type — -// including ones this build does not model — is skipped identically on both -// sides, which is what makes the key survive a Codex release that adds one. - -/** Only these carry a durable `(threadId, turnId, ordinal)` identity. */ -const CODEX_MESSAGE_ITEM_TYPES = new Set(['userMessage', 'agentMessage']) - -export type CodexThreadItem = { - type: string - id: string - [key: string]: unknown -} - -export function isCodexMessageItemType(type: string): boolean { - return CODEX_MESSAGE_ITEM_TYPES.has(type) -} - -export function readCodexThreadItem(value: unknown): CodexThreadItem | null { - if (typeof value !== 'object' || value === null) { - return null - } - const record = value as Record - return typeof record.type === 'string' && typeof record.id === 'string' - ? (record as CodexThreadItem) - : null -} - -function readRecord(value: unknown): Record { - return typeof value === 'object' && value !== null ? (value as Record) : {} -} - -/** - * Durable identity for a Codex item, or null for one that has none. - * - * Non-message items fall back to the `orca` namespace keyed by the Codex item - * id. That id is unstable across resume, so those rows are live-session detail - * that a recovered journal simply will not contain — which is correct: Codex - * itself does not persist them either. - */ -export function codexItemIdentity(input: { - threadId: string - turnId: string | null - item: CodexThreadItem - ordinals: CodexTurnOrdinals -}): AgentJournalItemIdentity { - const { item, turnId } = input - if (turnId && isCodexMessageItemType(item.type)) { - return { - provider: 'codex', - threadId: input.threadId, - turnId, - ordinal: input.ordinals.ordinalFor(input.threadId, turnId, item.id) - } - } - return { provider: 'orca', clientMessageId: `codex-item:${input.threadId}:${item.id}` } -} - -function readString(source: Record, key: string): string | null { - const value = source[key] - return typeof value === 'string' && value.length > 0 ? value : null -} - -function readFirstString(source: Record, keys: readonly string[]): string | null { - for (const key of keys) { - const value = readString(source, key) - if (value !== null) { - return value - } - } - return null -} - -function readTextContent(source: Record, key: string): string | null { - const direct = readString(source, key) - if (direct) { - return direct - } - const value = source[key] - if (!Array.isArray(value)) { - return null - } - const parts = value.flatMap((part) => { - if (typeof part === 'string') { - return part.length > 0 ? [part] : [] - } - if (typeof part !== 'object' || part === null) { - return [] - } - const text = readString(part as Record, 'text') - return text ? [text] : [] - }) - return parts.length > 0 ? parts.join('\n') : null -} +// Codex thread items → journal item bodies. /** `userMessage` carries structured content parts; `agentMessage` a flat text. */ export function codexMessageBlocks(item: CodexThreadItem): NativeChatBlock[] { @@ -175,75 +83,6 @@ export type CodexJournalItem = { handled: boolean } -/** - * Codex's own classification of a shell call: the tool name to show, and the - * fields worth lifting into `input` for the shared label helper (a file target, - * a search term, a scanned root). A `Map`, not an object — an object index - * answers `__proto__` with a truthy non-string. Every other action type stays an - * unclassified `shell` row. - * - * Nothing is invented for a field Codex sends as null: a stand-in path is a - * claim about a target, and the label helper turns any path into a file link. - */ -type CommandActionClass = { - name: string - /** Action field to the `input` key it lifts to. A scan root and a listed - * directory lift to `directory`, never `path`: the label helper reads `path` - * as a file target, which mobile turns into a tappable open-file link. */ - keys: Readonly> -} - -const COMMAND_ACTION_CLASSES = new Map([ - ['read', { name: 'read', keys: { path: 'path' } }], - ['search', { name: 'search', keys: { query: 'query', path: 'directory' } }], - ['listFiles', { name: 'list', keys: { path: 'directory' } }] -]) - -/** The one class every classified `commandActions` entry agrees on, with the - * fields they all agree on; null leaves the row exactly as a Codex that sends no - * classification renders it. `cat a.txt && ls src` classifies as two different - * things, and naming that row after either would drop the other, so it stays a - * `shell` row that shows the whole command. */ -function commandActionFacts( - item: CodexThreadItem -): { name: string; fields: Record } | null { - const actions = item.commandActions - if (!Array.isArray(actions)) { - return null - } - let matched: { class: CommandActionClass; fields: Record } | null = null - for (const action of actions) { - const record = readRecord(action) - const type = readString(record, 'type') - const classified = type === null ? undefined : COMMAND_ACTION_CLASSES.get(type) - if (classified === undefined) { - continue - } - if (matched === null) { - const fields: Record = {} - for (const [source, lifted] of Object.entries(classified.keys)) { - const value = readString(record, source) - if (value !== null) { - fields[lifted] = value - } - } - matched = { class: classified, fields } - continue - } - if (matched.class.name !== classified.name) { - return null - } - // The same class twice keeps the class, but only a target both entries name. - for (const [source, lifted] of Object.entries(matched.class.keys)) { - const kept = matched.fields[lifted] - if (kept !== undefined && readString(record, source) !== kept) { - delete matched.fields[lifted] - } - } - } - return matched === null ? null : { name: matched.class.name, fields: matched.fields } -} - function commandItem(item: CodexThreadItem): CodexJournalItem { const output = readFirstString(item, ['aggregatedOutput', 'aggregated_output']) const bounded = output === null ? null : boundInlineText(output, DEFAULT_JOURNAL_PAYLOAD_LIMITS) @@ -296,6 +135,85 @@ function fileChangeItem(item: CodexThreadItem): CodexJournalItem { } } +/** The tool name reaches the row verbatim — downstream dispatch (diff renderer, + * question parsers, input previews) matches raw identifiers, so any casing + * transform would silently miss them. `server/` qualifies it so two servers + * exposing the same tool stay distinguishable and neither shadows a built-in. */ +function mcpToolCallName(item: CodexThreadItem): string { + const tool = readString(item, 'tool') + const server = readString(item, 'server') + return tool === null ? 'mcp' : server === null ? tool : `${server}/${tool}` +} + +/** Row-label derivation only reads top-level keys, so the call's own arguments + * have to be the input itself. `arguments` is arbitrary JSON upstream: a + * non-object stays addressable under a key rather than being dropped, while a + * no-argument call — `{}` on the wire, the shape every argument-less MCP tool + * sends — becomes null so the row reads as a bare `server/tool` instead of a + * literal `{}`. */ +function mcpToolArguments(value: unknown): unknown { + if (typeof value !== 'object' || value === null) { + return value === null || value === undefined ? null : { arguments: value } + } + return Array.isArray(value) ? { arguments: value } : Object.keys(value).length > 0 ? value : null +} + +function mcpToolCallItem(item: CodexThreadItem): CodexJournalItem { + const failure = readString(readRecord(item.error), 'message') + const text = failure ?? readTextContent(readRecord(item.result), 'content') + const bounded = text === null ? null : boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS) + return { + body: { + kind: 'tool-call', + name: mcpToolCallName(item), + input: boundToolInput(mcpToolArguments(item.arguments), DEFAULT_JOURNAL_PAYLOAD_LIMITS), + state: failure === null ? commandState(item) : 'failed', + ...(bounded === null ? {} : { output: bounded.bounded }) + }, + handled: true + } +} + +/** A row label is read off top-level keys only, so the action's own labelable + * fields are hoisted beside the query while `action` stays whole for the + * expanded detail. The action `type` lands on `description`, the lowest-ranked + * label key, so it names only an action that carries nothing better. */ +function webSearchInput(item: CodexThreadItem): Record | null { + const action = readRecord(item.action) + const fields: [string, unknown][] = [ + ['url', readString(action, 'url')], + ['pattern', readString(action, 'pattern')], + ['description', readString(action, 'type')], + ['action', item.action ?? null] + ] + const query = readString(item, 'query') ?? readString(action, 'query') + const present = fields.filter(([, value]) => value !== null) + // A blank `query` is the run header's "this call has no brief argument" + // signal; drop the key and the header stands the row's raw JSON in for one. + return query === null && present.length === 0 + ? null + : { query: query ?? '', ...Object.fromEntries(present) } +} + +/** `webSearch` carries no status: Codex starts it with an empty query and a null + * action, then sends the action, so `action` is the completion signal — a + * completed item's own `query` is routinely still empty. The hits arrive on + * `results` and are the call's output. */ +function webSearchItem(item: CodexThreadItem): CodexJournalItem { + const hits = Array.isArray(item.results) && item.results.length > 0 ? item.results : null + const bounded = hits && boundInlineText(JSON.stringify(hits), DEFAULT_JOURNAL_PAYLOAD_LIMITS) + return { + body: { + kind: 'tool-call', + name: 'web_search', + input: boundToolInput(webSearchInput(item), DEFAULT_JOURNAL_PAYLOAD_LIMITS), + state: item.action === null || item.action === undefined ? 'running' : 'completed', + ...(bounded === null ? {} : { output: bounded.bounded }) + }, + handled: true + } +} + /** * Journal body for a Codex item, or null for one with nothing to render. * @@ -319,6 +237,12 @@ export function codexJournalItem(item: CodexThreadItem): CodexJournalItem { if (item.type === 'fileChange') { return fileChangeItem(item) } + if (item.type === 'mcpToolCall') { + return mcpToolCallItem(item) + } + if (item.type === 'webSearch') { + return webSearchItem(item) + } if (item.type === 'reasoning' || item.type === 'plan') { const text = readTextContent(item, 'text') ?? diff --git a/src/main/codex/codex-thread-item-identity.ts b/src/main/codex/codex-thread-item-identity.ts new file mode 100644 index 00000000000..0488e5c00e9 --- /dev/null +++ b/src/main/codex/codex-thread-item-identity.ts @@ -0,0 +1,65 @@ +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import type { CodexTurnOrdinals } from './codex-turn-ordinals' + +// Codex thread items → durable journal identities. +// +// THE ORDINAL RULE, and why it is not "index within the turn". Codex renumbers +// item ids positionally on resume (`item-1`…`item-N` across the whole thread), +// and a resumed turn does NOT contain every item the live turn emitted — +// reasoning and command execution are dropped from persisted history. Numbering +// by live position would therefore shift every message after the first tool +// call and hand the user a duplicate of the assistant's answer after a resume. +// +// So the ordinal counts MESSAGE items only, and the same projection is applied +// to the live stream and to a resumed turn's item list. Any other item type — +// including ones this build does not model — is skipped identically on both +// sides, which is what makes the key survive a Codex release that adds one. + +/** Only these carry a durable `(threadId, turnId, ordinal)` identity. */ +const CODEX_MESSAGE_ITEM_TYPES = new Set(['userMessage', 'agentMessage']) + +export type CodexThreadItem = { + type: string + id: string + [key: string]: unknown +} + +export function isCodexMessageItemType(type: string): boolean { + return CODEX_MESSAGE_ITEM_TYPES.has(type) +} + +export function readCodexThreadItem(value: unknown): CodexThreadItem | null { + if (typeof value !== 'object' || value === null) { + return null + } + const record = value as Record + return typeof record.type === 'string' && typeof record.id === 'string' + ? (record as CodexThreadItem) + : null +} + +/** + * Durable identity for a Codex item, or null for one that has none. + * + * Non-message items fall back to the `orca` namespace keyed by the Codex item + * id. That id is unstable across resume, so those rows are live-session detail + * that a recovered journal simply will not contain — which is correct: Codex + * itself does not persist them either. + */ +export function codexItemIdentity(input: { + threadId: string + turnId: string | null + item: CodexThreadItem + ordinals: CodexTurnOrdinals +}): AgentJournalItemIdentity { + const { item, turnId } = input + if (turnId && isCodexMessageItemType(item.type)) { + return { + provider: 'codex', + threadId: input.threadId, + turnId, + ordinal: input.ordinals.ordinalFor(input.threadId, turnId, item.id) + } + } + return { provider: 'orca', clientMessageId: `codex-item:${input.threadId}:${item.id}` } +} diff --git a/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts b/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts index 22bd645d8a6..9860aaa81d8 100644 --- a/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts +++ b/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts @@ -112,4 +112,52 @@ describe('provider frame classification catalog', () => { // An item type nobody has dispositioned still falls through visibly. expect(classifyProviderFrame('codex', 'item:futureThing', {})).toBe('timeline-substantive') }) + + it('chromes the one unmodelled codex item type that carries no content', () => { + expect(classifyProviderFrame('codex', 'item:sleep', { id: 's', durationMs: 20_000 })).toBe( + 'status-chrome' + ) + // Payload inspection still outranks the item catalog, so chroming a type + // cannot swallow one that reports a failure. + expect(classifyProviderFrame('codex', 'item:sleep', { id: 's', status: 'failed' })).toBe( + 'error-surface' + ) + }) + + it('keeps subagent items visible — the only evidence a spawned agent is working', () => { + expect( + classifyProviderFrame('codex', 'item:subAgentActivity', { + id: 'a-1', + kind: 'started', + agentThreadId: 'thread-child', + agentPath: '/root/list_directory' + }) + ).toBe('timeline-substantive') + expect( + classifyProviderFrame('codex', 'item:collabAgentToolCall', { + id: 'c-1', + tool: 'spawn', + status: 'inProgress', + senderThreadId: 'thread-root', + receiverThreadIds: ['thread-child'], + agentsStates: {} + }) + ).toBe('timeline-substantive') + }) + + it('leaves content-bearing codex item types on the visible fallback', () => { + // Each carries text or a path a user would want: review output, the image + // the agent looked at or generated, injected hook prompt text. + for (const type of [ + 'imageView', + 'imageGeneration', + 'enteredReviewMode', + 'exitedReviewMode', + 'hookPrompt' + ]) { + expect(classifyProviderFrame('codex', `item:${type}`, { id: 'i' }), type).toBe( + 'timeline-substantive' + ) + } + }) }) diff --git a/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts b/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts index 474b1385a4f..f05f4cd4c6c 100644 --- a/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts +++ b/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts @@ -197,7 +197,12 @@ function hasProviderError(payload: unknown): boolean { const CODEX_ITEM_CLASSIFICATIONS: Record = { // The `thread/compacted` notification is already chrome; its item form is the // same event and must not read as a mysterious opcode row. - contextCompaction: 'status-chrome' + contextCompaction: 'status-chrome', + // `{id, durationMs}` and nothing else — Codex's own transcript renders it as + // nothing at all. Every other item type this build does not model carries text + // a user would want (review output, an image path, hook prompt text, subagent + // progress), so those keep their visible fallback row. + sleep: 'status-chrome' } function notificationKind(kind: string): string { From 2513e2139043b3091ec8d61b60dcfef502c4af27 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Sat, 5 Sep 2026 15:35:03 -0700 Subject: [PATCH 06/23] fix(native-chat): publish structured session status from the host so the sidebar never goes stale (#18776) * fix(native-chat): publish structured session status from the host The sidebar learned whether a structured chat was mid-turn by replaying the session journal in the renderer, through a reader whose lifetime was tied to the chat pane. Hiding the pane stopped the reader before the turn's settlement arrived, so the row stayed on "working" until the chat was reopened. The same coupling meant a tab never opened this session showed no status at all, and a reloaded renderer lost every settled row. The host owns the journal, so it now projects each session's status once per journal publication and fans the changes out on one stream per client (`agentSession.subscribeStatus`). The projection survives eviction of an idle session's provider child and is republished when readable sessions are restored. The renderer bridge subscribes to that feed per runtime target and never opens a transcript reader; the observation hook is gone. Additive wire surface behind the existing structured capability; old hosts reject the method and the renderer retries, showing no status. * fix(native-chat): negotiate the status feed and stop losing a change on subscribe The status stream is additive to a surface that already shipped, so a host advertising agent-session.structured.v1 can still answer subscribeStatus with method_not_found. Every renderer error path reconnected, so a remote host one release behind got a relay round-trip every 5s and no sidebar status at all. Give the method its own capability and probe it before subscribing; a failed probe still retries, an absent capability does not. Re-projecting on subscribe also wrote straight into the shared cache, so a second client could pin the first to a stale summary. Route those diffs through publish() before the arriving subscriber is registered. * fix(native-chat): bound the status prompt, merge snapshots, and prove the unread path One status frame carries every retained session and a send admits 256 KB per prompt, so ~16 large-prompt sessions could push the snapshot past the 4 MB outbound guard and into the retry loop. Bound latestPrompt to the same 200-char single-line preview every other agent-status row already carries. A snapshot also replaced the cached map wholesale, so the empty first frame from a restarting host retracted every row before restore republished them. Merge instead; the tab map, not this feed, decides which sessions are listed. Tests: the hidden-pane claim now sits at the host, where a journal with no transcript subscriber is driven from running to idle; the RPC test reads a real projection instead of its own stub. * fix(native-chat): merge the duplicated status-event type import * test(native-chat): pin the restart status publication, and log the unsupported host Startup restore indexes a readable session and publishes its status, which is what puts a never-reopened tab back in the sidebar. Only an Electron screenshot covered that wiring; a sitting status subscriber now pins it directly. The terminal "host too old" branch was silent, so a mixed-version report showed an empty sidebar with nothing in the log to explain it. --------- Co-authored-by: Merge Sim --- ...structured-agent-session-history-result.ts | 4 +- .../structured-agent-session-host-lifetime.ts | 23 ++ .../structured-agent-session-host.ts | 43 ++-- ...session-restart-status-publication.test.ts | 149 +++++++++++ ...ructured-agent-session-status-feed.test.ts | 242 ++++++++++++++++++ .../structured-agent-session-status-feed.ts | 130 ++++++++++ ...ructured-agent-session-subscribers.test.ts | 92 +++++++ .../structured-agent-session-subscribers.ts | 11 + .../structured-agent-session-status-stream.ts | 62 +++++ ...tructured-agent-session-subscription-id.ts | 29 +++ .../methods/structured-agent-session.test.ts | 79 +++++- .../rpc/methods/structured-agent-session.ts | 52 ++-- ...tructuredAgentSessionStatusBridge.test.tsx | 226 +++++++++------- .../StructuredAgentSessionStatusBridge.tsx | 73 +++--- ...use-structured-agent-session-read.test.tsx | 31 +-- .../use-structured-agent-session-read.ts | 7 - .../structured-agent-session-client.ts | 50 +++- ...ructured-agent-session-status-feed.test.ts | 140 ++++++++++ .../structured-agent-session-status-feed.ts | 211 +++++++++++++++ src/shared/agent-session-wire.ts | 30 ++- src/shared/protocol-version.ts | 5 + ...tructured-agent-session-projection.test.ts | 48 ++++ .../structured-agent-session-projection.ts | 29 +++ ...ss-version-agent-session-wire.unit.test.ts | 24 +- 24 files changed, 1556 insertions(+), 234 deletions(-) create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-restart-status-publication.test.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.test.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.ts create mode 100644 src/main/runtime/rpc/methods/structured-agent-session-status-stream.ts create mode 100644 src/main/runtime/rpc/methods/structured-agent-session-subscription-id.ts create mode 100644 src/renderer/src/runtime/structured-agent-session-status-feed.test.ts create mode 100644 src/renderer/src/runtime/structured-agent-session-status-feed.ts diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-history-result.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-history-result.ts index 70fc9a43ed2..b8e198c9b6b 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-history-result.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-history-result.ts @@ -8,7 +8,7 @@ import type { import type { AgentSessionJournal } from '../agent-session-journal/journal-store' import { readAgentSessionHistory } from './agent-session-history-page' -function providerSessionMetadata( +export function structuredAgentSessionProviderSessionMetadata( record: AgentSessionRecord | null ): AgentProviderSessionMetadata | undefined { const head = record ? agentSessionProviderHandleChainHead(record.providerHandleChain) : null @@ -27,7 +27,7 @@ export function readStructuredAgentSessionHistoryResult(input: { }): AgentSessionHistoryResult { const result = readAgentSessionHistory(input.journal, input.request) const fence = input.record?.lease.runtimeFence - const providerSession = providerSessionMetadata(input.record) + const providerSession = structuredAgentSessionProviderSessionMetadata(input.record) if (fence === undefined) { return providerSession ? { ...result, providerSession } : result } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts index 2afd94ba128..ba62db1704e 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts @@ -19,6 +19,8 @@ import type { StructuredAgentSessionHostSession } from './structured-agent-session-host-types' import { releaseStoredStructuredAgentSessionOwner } from './structured-agent-session-lease-release' +import { resumeHeldStructuredAgentSession } from './structured-agent-session-hold-resume' +import type { AgentSessionWireRefusal } from '../../../shared/agent-session-wire' export type StructuredAgentSessionLifetimeContext = { deps: StructuredAgentSessionHostDeps @@ -67,6 +69,27 @@ export async function evictHeldStructuredAgentSession( ) } +/** The first hold on a childless session: reconcile the lease, settle recovery, then attach. */ +export async function resumeStructuredAgentSessionForHold( + context: StructuredAgentSessionLifetimeContext & { + reconcileLeases: (sessionId: string) => Promise + }, + sessionId: string, + attach: Parameters[0]['attach'] +): Promise { + const unreconciled = await context.reconcileLeases(sessionId) + if (unreconciled) { + throw new Error(unreconciled.code) + } + await context.runtimeState.resolveRecovery(sessionId) + await resumeHeldStructuredAgentSession({ + sessionId, + deps: context.deps, + now: context.now, + attach + }) +} + export function createStructuredAgentSessionHolds( context: StructuredAgentSessionLifetimeContext, input: { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts index 8989f5e4d72..e4c7d191067 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts @@ -33,13 +33,13 @@ import { attachStructuredAgentSession } from './structured-agent-session-attach- import { createStructuredAgentSessionHolds, evictHeldStructuredAgentSession, + resumeStructuredAgentSessionForHold, type StructuredAgentSessionLifetimeContext } from './structured-agent-session-host-lifetime' import type { StructuredAgentSessionHolds, StructuredAgentSessionHoldOptions } from './structured-agent-session-holds' -import { resumeHeldStructuredAgentSession } from './structured-agent-session-hold-resume' import type { StructuredAgentSessionAttachContext } from './structured-agent-session-attach-context' import { listStructuredAgentSessionTabs } from './structured-agent-session-host-tabs' import { @@ -57,6 +57,7 @@ import type { StructuredAgentSessionHostDeps, StructuredAgentSessionHostSession } from './structured-agent-session-host-types' +import { StructuredAgentSessionStatusFeed } from './structured-agent-session-status-feed' import { StructuredAgentSessionEventRecovery } from './structured-agent-session-event-recovery' import { StructuredAgentSessionBackgroundTaskChannel } from './structured-agent-session-background-task-channel' import { withTimeout } from '../../../shared/promise-timeout-fallback' @@ -66,7 +67,14 @@ const HANDOFF_DRAIN_TIMEOUT_MS = 5_000 export class StructuredAgentSessionHost { private readonly sessions = new Map() - private readonly subscribers = new AgentSessionSubscribers() + private readonly statusFeed = new StructuredAgentSessionStatusFeed({ + sessions: this.sessions, + getRecord: (sessionId) => this.deps.store.getRecord(sessionId), + now: () => this.now() + }) + private readonly subscribers = new AgentSessionSubscribers({ + onJournalPublished: (sessionId, journal) => this.statusFeed.publish(sessionId, journal) + }) private readonly tasks = new StructuredAgentSessionTaskQueue() private readonly runtimeState: StructuredAgentSessionHostRuntimeState private readonly reconcileLeases: (sessionId: string) => Promise @@ -112,7 +120,12 @@ export class StructuredAgentSessionHost { now: this.now }) this.holds = createStructuredAgentSessionHolds(this.lifetimeContext(), { - resume: (sessionId) => this.resumeForHold(sessionId), + resume: (sessionId) => + resumeStructuredAgentSessionForHold( + { ...this.lifetimeContext(), reconcileLeases: this.reconcileLeases }, + sessionId, + (params) => this.attach({ callerKey: 'trusted-local:surface-hold' }, params) + ), evict: (sessionId) => this.close(sessionId) }) this.readableRestorer = new StructuredAgentSessionReadableRestorer({ @@ -125,7 +138,10 @@ export class StructuredAgentSessionHost { hasSession: (sessionId) => this.sessions.has(sessionId), // Site 10: cannot overwrite a live entry — the restorer returns early on // `hasSession` inside the same serialized step as this `set`. - onReadable: (sessionId, restored) => this.sessions.set(sessionId, restored), + onReadable: (sessionId, restored) => { + this.sessions.set(sessionId, restored) + this.statusFeed.publish(sessionId) + }, restoreHandoff: (sessionId) => this.handoffs.restore(sessionId) }) this.eventRecovery = new StructuredAgentSessionEventRecovery({ @@ -160,20 +176,6 @@ export class StructuredAgentSessionHost { /** That surface is gone. The child outlives it by the release grace, and by any running turn. */ release = (sessionId: string, holderId: string): void => this.holds.release(sessionId, holderId) - private async resumeForHold(sessionId: string): Promise { - const unreconciled = await this.reconcileLeases(sessionId) - if (unreconciled) { - throw new Error(unreconciled.code) - } - await this.runtimeState.resolveRecovery(sessionId) - await resumeHeldStructuredAgentSession({ - sessionId, - deps: this.deps, - now: () => this.now(), - attach: (params) => this.attach({ callerKey: 'trusted-local:surface-hold' }, params) - }) - } - handleAdapterEvent = (event: Parameters[0]) => this.eventRecovery.handle(event) @@ -342,6 +344,11 @@ export class StructuredAgentSessionHost { ) => this.backgroundTasks.publish(sessionId, state) unsubscribe = (sessionId: string, id: string): void => this.subscribers.close(sessionId, id) + /** Every session's projected status for session lists; unlike `subscribe`, retains nothing. */ + subscribeStatus = ( + subscriber: Parameters[0] + ): (() => void) => this.statusFeed.subscribe(subscriber) + private requireSession(sessionId: string): StructuredAgentSessionHostSession { const session = this.sessions.get(sessionId) if (!session) { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-restart-status-publication.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-restart-status-publication.test.ts new file mode 100644 index 00000000000..884fba78a4e --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-restart-status-publication.test.ts @@ -0,0 +1,149 @@ +// Startup restore has to publish status, not just index the session. +// +// A tab nobody reopens after a restart still owes the sidebar a row. The host restores such a +// session read-only, without a provider child, so the only thing that can surface its state is +// the status publication the restore wiring makes. + +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' +import type { + AgentSessionMutationEnvelope, + AgentSessionStatusEvent +} from '../../../shared/agent-session-wire' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' +import { StructuredAgentSessionHost } from './structured-agent-session-host' +import { + HOST_TEST_NOW as NOW, + HOST_TEST_SESSION as SESSION, + HOST_TEST_THREAD as THREAD, + hostTestAttachParams, + hostTestMessage, + hostTestOperationId, + resetHostTestOperationIds +} from './structured-agent-session-host-test-data' + +const CALLER = { callerKey: 'client-1' } + +const hosts: StructuredAgentSessionHost[] = [] +let root = '' + +function adapter(): StructuredAgentSessionAdapter { + return { + acquire: async ({ fence, spawnToken }) => ({ + process: { hostId: 'local', pid: 4242, processStartTimeMs: 1_700_000_000_000, spawnToken }, + link: { + linkId: `link-${fence}`, + handle: { provider: 'codex', threadId: THREAD }, + origin: 'created', + mintedAtFence: fence, + observedAt: NOW + } + }), + dispatch: async () => ({ + state: 'accepted', + providerIdentity: { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal: 1 } + }), + cancelTurn: async () => ({ cancelled: true }), + answerPrompt: async () => undefined, + setOption: async () => undefined + } +} + +function createHost(store: AgentSessionRecordStore): StructuredAgentSessionHost { + const host = new StructuredAgentSessionHost({ + store, + adapter: adapter(), + journalRoot: root, + claimKeyId: 'key-1', + mintSpawnToken: () => 'spawn-a', + probeOwner: async () => ({ + outcome: 'indeterminate', + reason: 'read does not need ownership' + }), + now: () => NOW + }) + hosts.push(host) + return host +} + +function sendEnvelope( + store: AgentSessionRecordStore, + fields: Record +): AgentSessionMutationEnvelope { + return { + sessionId: SESSION, + clientOperationId: hostTestOperationId(), + expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? 1, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.send', + sessionId: SESSION, + fields + }) + } +} + +/** Persists one turn, then hands back a restarted host over the same directories. */ +async function restartWithPersistedTurn(): Promise { + root = await mkdtemp(join(tmpdir(), 'orca-restart-status-')) + resetHostTestOperationIds() + const directory = join(root, 'store') + const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + const host = createHost(store) + expect(await host.attach(CALLER, hostTestAttachParams(null))).toMatchObject({ ok: true }) + const body = hostTestMessage('persisted conversation') + await host.send(CALLER, { envelope: sendEnvelope(store, { body }), body }) + await host.flushAllStreamedEvents() + return createHost(await AgentSessionRecordStore.open({ directory, hostId: 'local' })) +} + +afterEach(async () => { + await Promise.all(hosts.splice(0).map((host) => host.flushAllStreamedEvents())) + await rm(root, { recursive: true, force: true }) + root = '' +}) + +describe('structured session restart status publication', () => { + // Served by the subscribe-time re-projection rather than the restore's own publish, so this + // covers what a restored journal projects — not the restore wiring. The test below pins that. + it('projects the persisted turn of a session restored without a provider', async () => { + const restarted = await restartWithPersistedTurn() + + await restarted.restoreReadableSessions() + const events: AgentSessionStatusEvent[] = [] + restarted.subscribeStatus({ id: 'session-list', emit: (event) => events.push(event) }) + + expect(events).toEqual([ + { + type: 'snapshot', + sessions: [ + expect.objectContaining({ + sessionId: SESSION, + workspaceId: 'workspace-1', + agent: 'codex', + status: 'idle', + latestPrompt: 'persisted conversation' + }) + ] + } + ]) + }) + + it('publishes a restored session to a list already sitting on the stream', async () => { + const restarted = await restartWithPersistedTurn() + const events: AgentSessionStatusEvent[] = [] + restarted.subscribeStatus({ id: 'session-list', emit: (event) => events.push(event) }) + expect(events).toEqual([{ type: 'snapshot', sessions: [] }]) + + await restarted.restoreReadableSessions() + + // The restore wiring publishes; without it this list never hears about the session at all. + expect(events.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ sessionId: SESSION, status: 'idle' }) + }) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.test.ts new file mode 100644 index 00000000000..7efd147c420 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.test.ts @@ -0,0 +1,242 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import type { AgentSessionStatusEvent } from '../../../shared/agent-session-wire' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' +import { StructuredAgentSessionStatusFeed } from './structured-agent-session-status-feed' + +const SESSION = 'status-session' +const TURN_IDENTITY = { + provider: 'codex', + threadId: 'thread-1', + turnId: 'turn-1', + ordinal: 0 +} as const +const USER_IDENTITY = { + provider: 'codex', + threadId: 'thread-1', + turnId: 'turn-1', + ordinal: 1 +} as const + +let root: string +const journals = createTrackedJournalOpener() + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-agent-status-feed-')) +}) + +afterEach(async () => { + await journals.closeAll() + await rm(root, { recursive: true, force: true }) +}) + +async function openJournal(sessionId = SESSION) { + return journals.open({ + identity: { + sessionId, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } + }, + journalDir: join(root, sessionId) + }) +} + +function indexed(session: { journal: Awaited> }) { + return { + journal: session.journal, + params: { location: { workspaceId: 'workspace-1' }, provider: 'codex' as const } + } +} + +function feedFor(sessions: Map> }>) { + let now = 1_000 + const feed = new StructuredAgentSessionStatusFeed({ + sessions: { + get: (sessionId: string) => { + const session = sessions.get(sessionId) + return session ? indexed(session) : undefined + }, + [Symbol.iterator]: function* () { + for (const [sessionId, session] of sessions) { + yield [sessionId, indexed(session)] as const + } + } + } as unknown as ReadonlyMap>, + getRecord: () => null, + now: () => (now += 1) + }) + const events: AgentSessionStatusEvent[] = [] + const dispose = feed.subscribe({ id: 'list-1', emit: (event) => events.push(event) }) + return { feed, events, dispose } +} + +describe('StructuredAgentSessionStatusFeed', () => { + it('opens with every readable session and reports no status before a persisted turn', async () => { + const journal = await openJournal() + const { events } = feedFor(new Map([[SESSION, { journal }]])) + + expect(events).toEqual([ + { + type: 'snapshot', + sessions: [ + { + sessionId: SESSION, + workspaceId: 'workspace-1', + agent: 'codex', + status: null, + latestPrompt: '', + updatedAt: expect.any(Number) + } + ] + } + ]) + }) + + it('publishes working, then idle once the running marker is tombstoned, and never a repeat', async () => { + const journal = await openJournal() + const { feed, events } = feedFor(new Map([[SESSION, { journal }]])) + await journal.appendItem( + USER_IDENTITY, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'write a poem' }] }, + { fence: 1 } + ) + await journal.appendItem( + TURN_IDENTITY, + { kind: 'status', text: 'Working', turnLifecycle: { turnId: 'turn-1', state: 'running' } }, + { fence: 1 } + ) + + feed.publish(SESSION) + feed.publish(SESSION) + expect(events.slice(1)).toEqual([ + { + type: 'status', + session: expect.objectContaining({ + sessionId: SESSION, + status: 'working', + latestPrompt: 'write a poem' + }) + } + ]) + + await journal.appendTombstone(TURN_IDENTITY, { fence: 1 }) + feed.publish(SESSION) + expect(events.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ sessionId: SESSION, status: 'idle' }) + }) + expect(events).toHaveLength(3) + }) + + it('reports a pending approval as attention', async () => { + const journal = await openJournal() + const { feed, events } = feedFor(new Map([[SESSION, { journal }]])) + await journal.appendItem( + USER_IDENTITY, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'run it' }] }, + { fence: 1 } + ) + await journal.appendItem( + TURN_IDENTITY, + { + kind: 'approval', + title: 'Run command?', + detail: null, + options: [{ id: 'yes', label: 'Allow' }], + resolution: { state: 'pending', selectedOptionId: null, resolvedBy: null, resolvedAt: null } + }, + { fence: 1 } + ) + + feed.publish(SESSION) + expect(events.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ status: 'attention' }) + }) + }) + + it('keeps the last projection for an evicted session and serves it to a new subscriber', async () => { + const journal = await openJournal() + const sessions = new Map([[SESSION, { journal }]]) + const { feed, events } = feedFor(sessions) + await journal.appendItem( + USER_IDENTITY, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hello' }] }, + { fence: 1 } + ) + feed.publish(SESSION) + expect(events.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ status: 'idle', latestPrompt: 'hello' }) + }) + + // Eviction drops the host's index entry; the projection it already made stays true. + sessions.delete(SESSION) + feed.publish(SESSION) + const late: AgentSessionStatusEvent[] = [] + feed.subscribe({ id: 'list-late', emit: (event) => late.push(event) }) + + expect(events).toHaveLength(2) + expect(late).toEqual([ + { + type: 'snapshot', + sessions: [expect.objectContaining({ sessionId: SESSION, status: 'idle' })] + } + ]) + }) + + it('tells the sitting subscribers about a change a new subscriber re-projected', async () => { + const journal = await openJournal() + const { feed, events } = feedFor(new Map([[SESSION, { journal }]])) + // Journal appends and the feed's publish are separate queue submissions, so the journal + // can already hold the turn when a second client connects and re-projects it. + await journal.appendItem( + USER_IDENTITY, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hello' }] }, + { fence: 1 } + ) + + const late: AgentSessionStatusEvent[] = [] + feed.subscribe({ id: 'list-late', emit: (event) => late.push(event) }) + + expect(events.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ status: 'idle', latestPrompt: 'hello' }) + }) + // The arriving subscriber reads that same state once, from its snapshot. + expect(late).toEqual([ + { + type: 'snapshot', + sessions: [expect.objectContaining({ status: 'idle', latestPrompt: 'hello' })] + } + ]) + // The cache is not left holding a value nobody was told about. + feed.publish(SESSION) + expect(events).toHaveLength(2) + }) + + it('ends a closed subscriber and keeps publishing to the rest', async () => { + const journal = await openJournal() + const { feed, events, dispose } = feedFor(new Map([[SESSION, { journal }]])) + const others: AgentSessionStatusEvent[] = [] + feed.subscribe({ id: 'list-2', emit: (event) => others.push(event) }) + + dispose() + await journal.appendItem( + USER_IDENTITY, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hello' }] }, + { fence: 1 } + ) + feed.publish(SESSION) + + expect(events.at(-1)).toEqual({ type: 'end' }) + expect(others.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ status: 'idle' }) + }) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.ts new file mode 100644 index 00000000000..5ad494cf830 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.ts @@ -0,0 +1,130 @@ +// The host's answer to "what is every structured session doing", fanned out to session lists. +// +// A client used to learn whether a turn was running by replaying the journal through its own +// reducer, which tied the answer to whichever surface happened to hold a reader open: hide the +// chat and the sidebar froze on the last thing it had heard. The host always has the journal, so +// it projects the status once per journal publication and sends only the changes. +// +// The last projection is kept after the session's provider child is evicted: an idle session is +// still idle without a process, and a renderer that reloads must not lose every settled row until +// each chat is reopened. Restart is the one boundary that forgets, and restoring readable sessions +// republishes them. + +import { agentProviderSessionsEqual } from '../../../shared/agent-session-resume' +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { + AgentSessionStatusEvent, + AgentSessionStatusSummary +} from '../../../shared/agent-session-wire' +import { projectStructuredAgentSessionStatusSummary } from '../../../shared/structured-agent-session-projection' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { structuredAgentSessionProviderSessionMetadata } from './structured-agent-session-history-result' + +export type StructuredAgentSessionStatusSubscriber = { + id: string + emit: (event: AgentSessionStatusEvent) => void +} + +type StatusFeedSession = { + journal: AgentSessionJournal + params: { location: { workspaceId: string }; provider: AgentSessionRecord['provider'] } +} + +export type StructuredAgentSessionStatusFeedDeps = { + sessions: ReadonlyMap + getRecord: (sessionId: string) => AgentSessionRecord | null + now: () => number +} + +function summariesEqual(a: AgentSessionStatusSummary, b: AgentSessionStatusSummary): boolean { + return ( + a.workspaceId === b.workspaceId && + a.agent === b.agent && + a.status === b.status && + a.latestPrompt === b.latestPrompt && + agentProviderSessionsEqual(undefined, a.providerSession, b.providerSession) + ) +} + +export class StructuredAgentSessionStatusFeed { + private readonly subscribers = new Map() + private readonly published = new Map() + + constructor(private readonly deps: StructuredAgentSessionStatusFeedDeps) {} + + /** Opens with every session this host has projected, live ones re-read, then only changes. */ + subscribe(subscriber: StructuredAgentSessionStatusSubscriber): () => void { + // Re-project before registering: a change found here has to reach the subscribers that + // already read the old value, and the arriving one carries it in its snapshot instead. + for (const [sessionId] of this.deps.sessions) { + this.publish(sessionId) + } + this.subscribers.set(subscriber.id, subscriber) + this.emit(subscriber, { type: 'snapshot', sessions: [...this.published.values()] }) + return () => this.unsubscribe(subscriber.id) + } + + unsubscribe(id: string): void { + const subscriber = this.subscribers.get(id) + if (!subscriber) { + return + } + this.subscribers.delete(id) + try { + subscriber.emit({ type: 'end' }) + } catch { + // The transport is already gone; teardown must remain idempotent. + } + } + + /** Re-projects one session after its journal changed; equal projections are not re-sent. */ + publish(sessionId: string, journal?: AgentSessionJournal): void { + const session = this.deps.sessions.get(sessionId) + if (!session) { + return + } + const summary = this.summaryFor(sessionId, session, journal ?? session.journal) + const previous = this.published.get(sessionId) + if (previous && summariesEqual(previous, summary)) { + return + } + this.published.set(sessionId, summary) + this.broadcast({ type: 'status', session: summary }) + } + + private summaryFor( + sessionId: string, + session: StatusFeedSession, + journal: AgentSessionJournal + ): AgentSessionStatusSummary { + // An unreadable journal projects as "no turn": the chat itself shows the reset. + const items = journal.isReadOnly ? [] : journal.snapshot().items + const providerSession = structuredAgentSessionProviderSessionMetadata( + this.deps.getRecord(sessionId) + ) + return { + sessionId, + workspaceId: session.params.location.workspaceId, + agent: session.params.provider, + ...projectStructuredAgentSessionStatusSummary(items), + ...(providerSession ? { providerSession } : {}), + updatedAt: this.deps.now() + } + } + + private broadcast(event: AgentSessionStatusEvent): void { + // A Map skips entries deleted mid-iteration, so a failing subscriber can drop itself here. + for (const subscriber of this.subscribers.values()) { + this.emit(subscriber, event) + } + } + + /** A dead transport must not poison every later publication. */ + private emit(subscriber: StructuredAgentSessionStatusSubscriber, event: AgentSessionStatusEvent) { + try { + subscriber.emit(event) + } catch { + this.subscribers.delete(subscriber.id) + } + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts index 6e156d91532..81bcfa82b40 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts @@ -5,6 +5,7 @@ import { afterEach, beforeEach, describe, expect, it } from 'vitest' import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' import type { AgentSessionHandoffStatus, + AgentSessionStatusEvent, AgentSessionSubscribeEvent } from '../../../shared/agent-session-wire' import { @@ -16,6 +17,7 @@ import { journalDatabaseFile } from '../agent-session-journal/journal-paths' import { insertJournalRow } from '../agent-session-journal/journal-row-table' import type { JournalRow } from '../agent-session-journal/journal-row-schema' import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' +import { StructuredAgentSessionStatusFeed } from './structured-agent-session-status-feed' import { AgentSessionSubscribers } from './structured-agent-session-subscribers' const SESSION = 'subscriber-session' @@ -70,6 +72,96 @@ describe('AgentSessionSubscribers', () => { ]) }) + it('reports every content publication to the journal hook, subscribed or not', async () => { + const journal = await journals.open({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } + }, + journalDir: join(root, 'hook-journal') + }) + const published: string[] = [] + const subscribers = new AgentSessionSubscribers({ + onJournalPublished: (sessionId, published_journal) => { + expect(published_journal).toBe(journal) + published.push(sessionId) + } + }) + + subscribers.publish(SESSION, journal) + subscribers.reset(SESSION, journal, 'epoch_changed', 1) + subscribers.snapshot(SESSION, journal, 1) + subscribers.handoff(SESSION, 1, { + owner: 'native', + direction: null, + phase: 'idle', + stage: null, + operationId: null + }) + + expect(published).toEqual([SESSION, SESSION, SESSION]) + }) + + it('settles a session nobody is reading, from running to idle', async () => { + // The defect this whole feed exists for: status used to come from a transcript reader, so a + // session with no open pane had no reader and froze on whatever it last said. Nothing here + // ever calls `subscribers.open`. + const journal = await journals.open({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } + }, + journalDir: join(root, 'unread-journal') + }) + const statusFeed = new StructuredAgentSessionStatusFeed({ + sessions: new Map([ + [ + SESSION, + { journal, params: { location: { workspaceId: 'workspace-1' }, provider: 'codex' } } + ] + ]), + getRecord: () => null, + now: () => 1_000 + }) + const subscribers = new AgentSessionSubscribers({ + onJournalPublished: (sessionId, published) => statusFeed.publish(sessionId, published) + }) + const statuses: AgentSessionStatusEvent[] = [] + statusFeed.subscribe({ id: 'session-list', emit: (event) => statuses.push(event) }) + const turn = { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 0 } as const + + await journal.appendItem( + { ...turn, ordinal: 1 }, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'write a poem' }] }, + { fence: 1 } + ) + await journal.appendItem( + turn, + { kind: 'status', text: 'Working', turnLifecycle: { turnId: 'turn-1', state: 'running' } }, + { fence: 1 } + ) + subscribers.publish(SESSION, journal) + + expect(statuses.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ status: 'working', latestPrompt: 'write a poem' }) + }) + + await journal.appendTombstone(turn, { fence: 1 }) + subscribers.publish(SESSION, journal) + + expect(statuses.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ status: 'idle' }) + }) + }) + it('publishes handoff-only changes without serializing a transcript snapshot', async () => { const journal = await journals.open({ identity: { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts index d3131505384..29dffa6a687 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts @@ -36,9 +36,17 @@ type Subscriber = { fence: number } +export type AgentSessionSubscribersHooks = { + /** Fires after any publication that can change journal content, whether or not anyone + * is subscribed to the transcript: session lists project status from this same edge. */ + onJournalPublished?: (sessionId: string, journal: AgentSessionJournal) => void +} + export class AgentSessionSubscribers { private readonly bySession = new Map>() + constructor(private readonly hooks: AgentSessionSubscribersHooks = {}) {} + /** Opens the stream with a bounded tail page or, when the client's cursor * still resolves, with the rows it missed. Returns the disposer. */ open(input: { @@ -99,6 +107,7 @@ export class AgentSessionSubscribers { for (const subscriber of this.subscribers(sessionId)) { this.deliver(subscriber, journal) } + this.hooks.onJournalPublished?.(sessionId, journal) } /** Force every subscriber back to a bounded tail page — recovery, epoch @@ -123,6 +132,7 @@ export class AgentSessionSubscribers { subscriber.cursor = page.liveCursor ?? page.window.nextCursor subscriber.fence = fence } + this.hooks.onJournalPublished?.(sessionId, journal) } snapshot( @@ -143,6 +153,7 @@ export class AgentSessionSubscribers { subscriber.cursor = page.liveCursor ?? page.window.nextCursor subscriber.fence = fence } + this.hooks.onJournalPublished?.(sessionId, journal) } handoff(sessionId: string, fence: number, handoff: AgentSessionHandoffStatus): void { diff --git a/src/main/runtime/rpc/methods/structured-agent-session-status-stream.ts b/src/main/runtime/rpc/methods/structured-agent-session-status-stream.ts new file mode 100644 index 00000000000..8637089c254 --- /dev/null +++ b/src/main/runtime/rpc/methods/structured-agent-session-status-stream.ts @@ -0,0 +1,62 @@ +// `agentSession.subscribeStatus` — every structured session's projected status on one stream. +// +// Session lists read turn state from here instead of replaying transcripts: one stream per client +// covers every session, and unlike a transcript subscription it retains none of them. + +import { defineStreamingMethod, type RpcAnyMethod, type RpcContext } from '../core' +import { requireStructuredHost as requireHost } from './structured-agent-session-gate' +import { structuredAgentSessionStatusSubscriptionId } from './structured-agent-session-subscription-id' + +/** Ties a stream to both ends that can close it — the runtime's subscription registry and the + * transport abort — so either one runs `onClose` exactly once. */ +export function bindStructuredAgentSessionStream( + ctx: RpcContext, + subscriptionId: string, + onClose: () => void +): { isClosed: () => boolean } { + let closed = false + let releaseTransportSubscription = (): void => {} + const onTransportAbort = (): void => releaseTransportSubscription() + const cleanup = (): void => { + closed = true + ctx.signal?.removeEventListener('abort', onTransportAbort) + onClose() + } + let registration: { releaseIfCurrent: () => void } + if (typeof ctx.runtime.registerOwnedSubscriptionCleanup === 'function') { + registration = ctx.runtime.registerOwnedSubscriptionCleanup( + subscriptionId, + cleanup, + ctx.connectionId + ) + } else { + ctx.runtime.registerSubscriptionCleanup(subscriptionId, cleanup, ctx.connectionId) + registration = { releaseIfCurrent: () => ctx.runtime.cleanupSubscription(subscriptionId) } + } + releaseTransportSubscription = registration.releaseIfCurrent + ctx.signal?.addEventListener('abort', onTransportAbort, { once: true }) + if (ctx.signal?.aborted) { + onTransportAbort() + } + return { isClosed: () => closed } +} + +export const STRUCTURED_AGENT_SESSION_STATUS_METHODS: RpcAnyMethod[] = [ + defineStreamingMethod({ + name: 'agentSession.subscribeStatus', + params: null, + handler: async (_params, ctx, emit) => { + const host = requireHost(ctx) + const subscriptionId = structuredAgentSessionStatusSubscriptionId(ctx) + let dispose = (): void => {} + const stream = bindStructuredAgentSessionStream(ctx, subscriptionId, () => dispose()) + if (stream.isClosed()) { + return + } + dispose = host.subscribeStatus({ id: subscriptionId, emit }) + if (stream.isClosed()) { + dispose() + } + } + }) +] diff --git a/src/main/runtime/rpc/methods/structured-agent-session-subscription-id.ts b/src/main/runtime/rpc/methods/structured-agent-session-subscription-id.ts new file mode 100644 index 00000000000..f3d00232359 --- /dev/null +++ b/src/main/runtime/rpc/methods/structured-agent-session-subscription-id.ts @@ -0,0 +1,29 @@ +// Subscription ids for the streaming `agentSession.*` methods. +// +// Shared control multiplexes several streams over one socket, so the frame id keeps one +// subscriber from evicting another. It is appended only when present: collapsing a missing +// frame id to a constant is the collision the rule exists to prevent. + +import type { RpcContext } from '../core' + +const SUBSCRIPTION_PREFIX = 'agentSession' + +function withFrameId(ctx: RpcContext, base: string): string { + return ctx.requestId ? `${base}:${ctx.requestId}` : base +} + +/** The id a session's streams share before the frame id. `unsubscribe` addresses this + * directly and sweeps `${base}:` to reach every frame under it. */ +export function structuredAgentSessionSubscriptionBase(ctx: RpcContext, sessionId: string): string { + return `${SUBSCRIPTION_PREFIX}:${ctx.connectionId ?? 'local'}:${sessionId}` +} + +/** One session's transcript stream. */ +export function structuredAgentSessionSubscriptionId(ctx: RpcContext, sessionId: string): string { + return withFrameId(ctx, structuredAgentSessionSubscriptionBase(ctx, sessionId)) +} + +/** The status feed, which is per connection rather than per session. */ +export function structuredAgentSessionStatusSubscriptionId(ctx: RpcContext): string { + return withFrameId(ctx, `${SUBSCRIPTION_PREFIX}.status:${ctx.connectionId ?? 'local'}`) +} diff --git a/src/main/runtime/rpc/methods/structured-agent-session.test.ts b/src/main/runtime/rpc/methods/structured-agent-session.test.ts index e4888e4a9df..69b04711960 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.test.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.test.ts @@ -2,8 +2,14 @@ // accepts once they can. import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { AgentJournalRenderItem } from '../../../../shared/agent-session-journal-types' +import type { AgentSessionJournal } from '../../../native-chat/agent-session-journal/journal-store' import type { StructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-host' import { setStructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-registry' +import { + StructuredAgentSessionStatusFeed, + type StructuredAgentSessionStatusSubscriber +} from '../../../native-chat/agent-session-wire/structured-agent-session-status-feed' import { RUNTIME_CAPABILITIES, RUNTIME_PROTOCOL_VERSION, @@ -64,6 +70,44 @@ function request(method: string, params: unknown): RpcRequest { let hostCalls: Record> let runtimeCalls: Record> +const STATUS_SESSION = 'session-status' +const STATUS_ITEMS: AgentJournalRenderItem[] = [ + { + itemId: 'user-1', + sequence: 1, + revision: 1, + observedAt: 1, + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'write a poem' }] } + }, + { + itemId: 'turn-1', + sequence: 2, + revision: 1, + observedAt: 2, + body: { kind: 'status', text: 'Working', turnLifecycle: { turnId: 'turn-1', state: 'running' } } + } +] + +/** One indexed session over a journal that reads back fixed items; the projection is real. */ +function statusFeed(): StructuredAgentSessionStatusFeed { + return new StructuredAgentSessionStatusFeed({ + sessions: new Map([ + [ + STATUS_SESSION, + { + journal: { + isReadOnly: false, + snapshot: () => ({ items: STATUS_ITEMS }) + } as unknown as AgentSessionJournal, + params: { location: { workspaceId: 'workspace-1' }, provider: 'codex' as const } + } + ] + ]), + getRecord: () => null, + now: () => 1_000 + }) +} + function hostStub(): StructuredAgentSessionHost { hostCalls = { attach: vi.fn(async () => ({ @@ -122,6 +166,11 @@ function hostStub(): StructuredAgentSessionHost { })), history: vi.fn(() => ({ ok: true, page: { items: [] } })), subscribe: vi.fn(() => () => undefined), + // A real feed, so the snapshot this method hands back is a genuine projection rather + // than a shape the stub restated. + subscribeStatus: vi.fn((subscriber: StructuredAgentSessionStatusSubscriber) => + statusFeed().subscribe(subscriber) + ), unsubscribe: vi.fn() } return hostCalls as unknown as StructuredAgentSessionHost @@ -240,7 +289,7 @@ describe('capability gating', () => { } // Bump deliberately: the whole agentSession.* surface is behind the structured capability, // so an additive method is invisible to old clients and needs no protocol bump. - expect(STRUCTURED_AGENT_SESSION_METHODS).toHaveLength(17) + expect(STRUCTURED_AGENT_SESSION_METHODS).toHaveLength(18) }) it('hides the surface from a declared client that did not advertise it', async () => { @@ -583,3 +632,31 @@ describe('parameter validation', () => { expect(response).toMatchObject({ ok: true }) }) }) + +describe('agentSession.subscribeStatus', () => { + it('is invisible to a client without the structured capability', async () => { + const reply = await call('agentSession.subscribeStatus', null, { clientKind: 'runtime' }) + expect(reply.ok).toBe(false) + expect(hostCalls.subscribeStatus).not.toHaveBeenCalled() + }) + + it('opens the host status feed with a projected snapshot as its first reply', async () => { + const reply = await call('agentSession.subscribeStatus', null, STRUCTURED_CLIENT) + expect(reply).toMatchObject({ + ok: true, + result: { + type: 'snapshot', + sessions: [ + { + sessionId: STATUS_SESSION, + workspaceId: 'workspace-1', + agent: 'codex', + status: 'working', + latestPrompt: 'write a poem' + } + ] + } + }) + expect(hostCalls.subscribeStatus).toHaveBeenCalledOnce() + }) +}) diff --git a/src/main/runtime/rpc/methods/structured-agent-session.ts b/src/main/runtime/rpc/methods/structured-agent-session.ts index b69ff6fd628..61b0f4dcf4f 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.ts @@ -21,6 +21,14 @@ import { import type { AgentSessionAttachParams } from '../../../native-chat/agent-session-wire/structured-agent-session-attach' import { STRUCTURED_AGENT_SESSION_HOLD_METHODS } from './structured-agent-session-hold' import { resolveUncommittedStructuredCreate } from './structured-agent-session-precommit-refusal' +import { + bindStructuredAgentSessionStream, + STRUCTURED_AGENT_SESSION_STATUS_METHODS +} from './structured-agent-session-status-stream' +import { + structuredAgentSessionSubscriptionBase as subscriptionBaseFor, + structuredAgentSessionSubscriptionId as subscriptionIdFor +} from './structured-agent-session-subscription-id' import { AttachParams, CancelParams, @@ -37,15 +45,6 @@ import { UnsubscribeParams } from './structured-agent-session-schemas' -const SUBSCRIPTION_PREFIX = 'agentSession' - -function subscriptionIdFor(ctx: RpcContext, sessionId: string): string { - const base = `${SUBSCRIPTION_PREFIX}:${ctx.connectionId ?? 'local'}:${sessionId}` - // Shared control multiplexes several streams over one socket; the frame id - // keeps one subscriber from evicting another on the same session. - return ctx.requestId ? `${base}:${ctx.requestId}` : base -} - /** * The attach-shaped entries take the location from the client instead of resolving it from a * worktree, so they never reach the worktree-resolving create-support check. Ask the executing @@ -245,33 +244,12 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ // Retain-only: reading history must never be what starts a provider process. Current clients // explicitly hold every open surface before subscribing. const streamHolder = `subscription:${subscriptionId}` - let closed = false let dispose = (): void => {} - let releaseTransportSubscription = (): void => {} - const onTransportAbort = (): void => releaseTransportSubscription() - const cleanup = () => { - closed = true - ctx.signal?.removeEventListener('abort', onTransportAbort) + const stream = bindStructuredAgentSessionStream(ctx, subscriptionId, () => { dispose() host.release(params.sessionId, streamHolder) - } - let registration: { releaseIfCurrent: () => void } - if (typeof ctx.runtime.registerOwnedSubscriptionCleanup === 'function') { - registration = ctx.runtime.registerOwnedSubscriptionCleanup( - subscriptionId, - cleanup, - ctx.connectionId - ) - } else { - ctx.runtime.registerSubscriptionCleanup(subscriptionId, cleanup, ctx.connectionId) - registration = { releaseIfCurrent: () => ctx.runtime.cleanupSubscription(subscriptionId) } - } - releaseTransportSubscription = registration.releaseIfCurrent - ctx.signal?.addEventListener('abort', onTransportAbort, { once: true }) - if (ctx.signal?.aborted) { - onTransportAbort() - } - if (closed) { + }) + if (stream.isClosed()) { return } // The host emits the opening snapshot (or the missed batch) synchronously @@ -282,7 +260,7 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ emit, ...(params.cursor ? { cursor: params.cursor } : {}) }) - if (closed) { + if (stream.isClosed()) { dispose() } else { // Fire-and-forget, but never unhandled: a resume that refuses leaves the stream holding a @@ -300,8 +278,7 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ params: UnsubscribeParams, handler: async (params, ctx) => { requireHost(ctx) - const connection = ctx.connectionId ?? 'local' - const base = `${SUBSCRIPTION_PREFIX}:${connection}:${params.sessionId}` + const base = subscriptionBaseFor(ctx, params.sessionId) if (params.subscriptionId) { ctx.runtime.cleanupSubscription(`${base}:${params.subscriptionId}`) return { unsubscribed: true } @@ -311,5 +288,6 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ return { unsubscribed: true } } }), - ...STRUCTURED_AGENT_SESSION_HOLD_METHODS + ...STRUCTURED_AGENT_SESSION_HOLD_METHODS, + ...STRUCTURED_AGENT_SESSION_STATUS_METHODS ] diff --git a/src/renderer/src/components/native-chat/StructuredAgentSessionStatusBridge.test.tsx b/src/renderer/src/components/native-chat/StructuredAgentSessionStatusBridge.test.tsx index 8a61399d669..a8eb22ae7a5 100644 --- a/src/renderer/src/components/native-chat/StructuredAgentSessionStatusBridge.test.tsx +++ b/src/renderer/src/components/native-chat/StructuredAgentSessionStatusBridge.test.tsx @@ -2,17 +2,23 @@ import { act, cleanup, render, waitFor } from '@testing-library/react' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { + AgentSessionStatusEvent, + AgentSessionStatusSummary +} from '../../../../shared/agent-session-wire' import type { Tab } from '../../../../shared/tab-types' +import type * as RuntimeRpcClientModule from '@/runtime/runtime-rpc-client' const mocks = vi.hoisted(() => ({ - call: vi.fn(), removeAgentStatus: vi.fn(), setAgentStatus: vi.fn(), store: null as null | { getState: () => Record setState: (state: Record) => void }, - subscribe: vi.fn(), + subscribeStatus: vi.fn(), + subscribeTranscript: vi.fn(), + supportsCapability: vi.fn(), unsubscribe: vi.fn() })) @@ -73,17 +79,22 @@ vi.mock('@/lib/worktree-runtime-owner', () => ({ state.testRuntimeOwner ?? null })) +vi.mock('@/runtime/runtime-rpc-client', async (importOriginal) => ({ + ...(await importOriginal()), + runtimeEnvironmentSupportsCapability: mocks.supportsCapability +})) + vi.mock('@/runtime/structured-agent-session-client', () => ({ - callStructuredAgentSession: mocks.call, - subscribeStructuredAgentSession: mocks.subscribe + callStructuredAgentSession: vi.fn(), + subscribeStructuredAgentSession: mocks.subscribeTranscript, + subscribeStructuredAgentSessionStatus: mocks.subscribeStatus })) import { getStructuredAgentSessionTabs, StructuredAgentSessionStatusBridge } from './StructuredAgentSessionStatusBridge' -import { resetStructuredAgentSessionReadOwnersForTests } from './structured-agent-session-read-owner' -import { useStructuredAgentSessionRead } from './use-structured-agent-session-read' +import { resetStructuredAgentSessionStatusFeedsForTests } from '@/runtime/structured-agent-session-status-feed' const structuredTab = { id: 'structured-tab-1', @@ -100,60 +111,40 @@ const structuredTab = { agentSessionAgent: 'codex' } satisfies Tab -const userItem = { - itemId: 'item-1', - revision: 1, - sequence: 1, - observedAt: 1, - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hello' }] } -} as const +const providerSession = { key: 'session_id', id: '01a002e9-9a1c-7d42-a642-e481f64446f1' } as const -const historyResult = { - ok: true, - providerSession: { key: 'session_id', id: '01a002e9-9a1c-7d42-a642-e481f64446f1' }, - page: { +function summary(overrides: Partial = {}): AgentSessionStatusSummary { + return { sessionId: 'session-1', - epoch: 'epoch-1', - fence: 1, - direction: 'tail', - items: [userItem], - removedItemIds: [], - submissions: [], - window: { - oldest: { epoch: 'epoch-1', sequence: 1 }, - newest: { epoch: 'epoch-1', sequence: 1 }, - nextCursor: { epoch: 'epoch-1', sequence: 1 } - }, - liveCursor: { epoch: 'epoch-1', sequence: 1 }, - hasOlder: false, - hasNewer: false + workspaceId: 'wt-1', + agent: 'codex', + status: 'working', + latestPrompt: 'hello', + providerSession, + updatedAt: 1, + ...overrides } } -function ActiveSessionRead(): null { - useStructuredAgentSessionRead({ - sessionId: structuredTab.entityId, - target: { kind: 'local' }, - isVisible: true - }) - return null +function statuses(): Record[] { + return Object.values(mocks.store?.getState().agentStatusByPaneKey ?? {}) } -function ActiveComposition(): React.JSX.Element { - return ( - <> - - - - ) +/** The host side of the most recent status subscription. */ +function feed(index = 0): { target: unknown; emit: (event: AgentSessionStatusEvent) => void } { + const call = mocks.subscribeStatus.mock.calls[index] + if (!call) { + throw new Error('status feed not subscribed') + } + return { target: call[0], emit: call[1] as (event: AgentSessionStatusEvent) => void } } describe('StructuredAgentSessionStatusBridge', () => { beforeEach(() => { vi.clearAllMocks() - resetStructuredAgentSessionReadOwnersForTests() - mocks.call.mockResolvedValue(historyResult) - mocks.subscribe.mockResolvedValue({ unsubscribe: mocks.unsubscribe }) + resetStructuredAgentSessionStatusFeedsForTests() + mocks.subscribeStatus.mockResolvedValue({ unsubscribe: mocks.unsubscribe }) + mocks.supportsCapability.mockResolvedValue(true) mocks.store?.setState({ agentStatusByPaneKey: {}, testRuntimeOwner: null, @@ -163,7 +154,7 @@ describe('StructuredAgentSessionStatusBridge', () => { afterEach(() => { cleanup() - resetStructuredAgentSessionReadOwnersForTests() + resetStructuredAgentSessionStatusFeedsForTests() }) it('reuses the structured-tab projection for an unchanged tab map', () => { @@ -194,65 +185,111 @@ describe('StructuredAgentSessionStatusBridge', () => { ]) }) - it('keeps restored inactive tabs transport-neutral', async () => { + it('projects the host status feed without opening a transcript reader', async () => { render() - await act(() => Promise.resolve()) + await waitFor(() => expect(mocks.subscribeStatus).toHaveBeenCalledOnce()) + expect(feed().target).toEqual({ kind: 'local' }) + expect(mocks.subscribeTranscript).not.toHaveBeenCalled() - expect(mocks.call).not.toHaveBeenCalled() - expect(mocks.subscribe).not.toHaveBeenCalled() + act(() => feed().emit({ type: 'snapshot', sessions: [summary()] })) + + expect(statuses()).toEqual([ + expect.objectContaining({ + state: 'working', + prompt: 'hello', + agentType: 'codex', + sessionBoundary: false, + tabId: structuredTab.id, + worktreeId: 'wt-1', + terminalTitle: 'Codex Chat', + terminalResumeEligible: false, + providerSession + }) + ]) + }) + + // Hiddenness is the host's side of this: see structured-agent-session-subscribers.test.ts, + // which drives an unsubscribed journal through the feed. Here the transport is a mock, so + // only the summary-to-store mapping is under test. + it('maps each host status onto the sidebar agent state', async () => { + render() + await waitFor(() => expect(mocks.subscribeStatus).toHaveBeenCalledOnce()) + act(() => feed().emit({ type: 'snapshot', sessions: [summary()] })) + expect(statuses()).toEqual([expect.objectContaining({ state: 'working' })]) + + act(() => feed().emit({ type: 'status', session: summary({ status: 'idle', updatedAt: 2 }) })) + expect(statuses()).toEqual([expect.objectContaining({ state: 'done', sessionBoundary: true })]) + + act(() => + feed().emit({ type: 'status', session: summary({ status: 'attention', updatedAt: 3 }) }) + ) + expect(statuses()).toEqual([expect.objectContaining({ state: 'blocked' })]) + }) + + it('shows no status before a persisted turn', async () => { + render() + await waitFor(() => expect(mocks.subscribeStatus).toHaveBeenCalledOnce()) + + act(() => feed().emit({ type: 'snapshot', sessions: [summary({ status: null })] })) expect(mocks.setAgentStatus).not.toHaveBeenCalled() + + act(() => feed().emit({ type: 'status', session: summary({ updatedAt: 2 }) })) + expect(statuses()).toEqual([expect.objectContaining({ state: 'working' })]) }) - it('shares the visible pane subscriber with status projection', async () => { - render() - - await waitFor(() => expect(mocks.setAgentStatus).toHaveBeenCalledOnce()) - expect(mocks.call).toHaveBeenCalledOnce() - expect(mocks.subscribe).toHaveBeenCalledOnce() - expect(mocks.setAgentStatus.mock.calls[0]?.[5]).toEqual({ - providerSession: historyResult.providerSession, - terminalResumeEligible: false - }) - }) - - it('keeps the status map reference stable for coalesced assistant deltas', async () => { - render() - await waitFor(() => expect(mocks.setAgentStatus).toHaveBeenCalledOnce()) + it('keeps the status map reference stable for repeated equal summaries', async () => { + render() + await waitFor(() => expect(mocks.subscribeStatus).toHaveBeenCalledOnce()) + act(() => feed().emit({ type: 'snapshot', sessions: [summary()] })) const before = mocks.store?.getState().agentStatusByPaneKey - const onEvent = mocks.subscribe.mock.calls[0]?.[2] as (event: unknown) => void act(() => { - for (let sequence = 2; sequence <= 12; sequence += 1) { - onEvent({ - type: 'batch', - sessionId: 'session-1', - batch: { - cursor: { epoch: 'epoch-1', sequence }, - items: [ - { - itemId: 'assistant-1', - revision: sequence, - sequence, - observedAt: sequence, - body: { - kind: 'message', - role: 'assistant', - blocks: [{ type: 'text', text: `delta-${sequence}` }] - } - } - ], - removedItemIds: [], - submissions: [] - } - }) + for (let updatedAt = 2; updatedAt <= 12; updatedAt += 1) { + feed().emit({ type: 'status', session: summary({ updatedAt }) }) } }) - await act(async () => new Promise((resolve) => setTimeout(resolve, 60))) expect(mocks.setAgentStatus).toHaveBeenCalledOnce() expect(mocks.store?.getState().agentStatusByPaneKey).toBe(before) }) + it('drops the status and the feed when the last structured tab closes', async () => { + render() + await waitFor(() => expect(mocks.subscribeStatus).toHaveBeenCalledOnce()) + act(() => feed().emit({ type: 'snapshot', sessions: [summary()] })) + expect(statuses()).toHaveLength(1) + + act(() => mocks.store?.setState({ unifiedTabsByWorktree: { 'wt-1': [] } })) + + expect(statuses()).toEqual([]) + await waitFor(() => expect(mocks.unsubscribe).toHaveBeenCalledOnce()) + }) + + it('reconnects after the host ends the stream', async () => { + vi.useFakeTimers() + try { + render() + await act(() => Promise.resolve()) + expect(mocks.subscribeStatus).toHaveBeenCalledOnce() + + act(() => feed().emit({ type: 'end' })) + await act(() => vi.advanceTimersByTimeAsync(300)) + + expect(mocks.unsubscribe).toHaveBeenCalledOnce() + expect(mocks.subscribeStatus).toHaveBeenCalledTimes(2) + } finally { + vi.useRealTimers() + } + }) + + it('keys the feed by the worktree runtime environment', async () => { + mocks.store?.setState({ testRuntimeOwner: 'env-1' }) + render() + await waitFor(() => expect(mocks.subscribeStatus).toHaveBeenCalledOnce()) + + expect(feed().target).toEqual({ kind: 'environment', environmentId: 'env-1' }) + }) + it('does not project an unknown provider as Codex', async () => { mocks.store?.setState({ unifiedTabsByWorktree: { @@ -262,8 +299,7 @@ describe('StructuredAgentSessionStatusBridge', () => { render() await act(() => Promise.resolve()) - expect(mocks.call).not.toHaveBeenCalled() - expect(mocks.subscribe).not.toHaveBeenCalled() + expect(mocks.subscribeStatus).not.toHaveBeenCalled() expect(mocks.setAgentStatus).not.toHaveBeenCalled() }) }) diff --git a/src/renderer/src/components/native-chat/StructuredAgentSessionStatusBridge.tsx b/src/renderer/src/components/native-chat/StructuredAgentSessionStatusBridge.tsx index 8409858dd17..d4cb74ab93b 100644 --- a/src/renderer/src/components/native-chat/StructuredAgentSessionStatusBridge.tsx +++ b/src/renderer/src/components/native-chat/StructuredAgentSessionStatusBridge.tsx @@ -1,19 +1,14 @@ -import { useEffect, useMemo } from 'react' +import { useEffect, useMemo, useSyncExternalStore } from 'react' import { useShallow } from 'zustand/react/shallow' -import type { AgentProviderSessionMetadata } from '../../../../shared/agent-session-resume' import { agentProviderSessionsEqual } from '../../../../shared/agent-session-resume' -import { - hasPersistedStructuredAgentSessionTurn, - projectStructuredAgentSessionStatus, - structuredAgentSessionPaneKey -} from '../../../../shared/structured-agent-session-projection' -import type { StructuredAgentSessionState } from '../../../../shared/structured-agent-session-reducer' +import type { AgentSessionStatusSummary } from '../../../../shared/agent-session-wire' +import { structuredAgentSessionPaneKey } from '../../../../shared/structured-agent-session-projection' import type { Tab } from '../../../../shared/tab-types' import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner' import { useAppStore } from '@/store' -import { getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client' -import { useStructuredAgentSessionReadObservation } from './use-structured-agent-session-read' +import { getActiveRuntimeTarget, type RuntimeClientTarget } from '@/runtime/runtime-rpc-client' +import { getStructuredAgentSessionStatusFeed } from '@/runtime/structured-agent-session-status-feed' type StructuredTab = Tab & { contentType: 'agent-session' } @@ -47,35 +42,40 @@ export function getStructuredAgentSessionTabs( return tabs } -function latestPrompt(state: StructuredAgentSessionState): string { - for (let index = state.items.length - 1; index >= 0; index -= 1) { - const body = state.items[index]?.body - if (body?.kind === 'message' && body.role === 'user') { - return body.blocks.flatMap((block) => (block.type === 'text' ? [block.text] : [])).join('\n') - } - } - return '' +/** The host's projected status for one session, live while the caller is mounted. */ +function useStructuredAgentSessionStatusSummary( + sessionId: string, + target: RuntimeClientTarget +): AgentSessionStatusSummary | null { + const feed = useMemo(() => getStructuredAgentSessionStatusFeed(target), [target]) + useEffect(() => feed.activate(), [feed]) + return useSyncExternalStore( + feed.subscribe, + () => feed.getSnapshot().get(sessionId) ?? null, + () => null + ) } -function projectStatus( - tab: StructuredTab, - state: StructuredAgentSessionState, - providerSession: AgentProviderSessionMetadata | undefined -): void { +function projectStatus(tab: StructuredTab, summary: AgentSessionStatusSummary | null): void { const paneKey = structuredAgentSessionPaneKey(tab.id, tab.entityId) const store = useAppStore.getState() - if (!hasPersistedStructuredAgentSessionTurn(state.items)) { + // No persisted turn yet (or nothing known): the row shows no agent status at all. + if (!summary?.status) { if (store.agentStatusByPaneKey?.[paneKey]) { store.removeAgentStatus(paneKey) } return } - const projection = projectStructuredAgentSessionStatus(state.items) const desired = { - state: projection === 'working' ? 'working' : projection === 'attention' ? 'blocked' : 'done', - prompt: latestPrompt(state), + state: + summary.status === 'working' + ? 'working' + : summary.status === 'attention' + ? 'blocked' + : 'done', + prompt: summary.latestPrompt, agentType: tab.agentSessionAgent, - sessionBoundary: projection === 'idle' + sessionBoundary: summary.status === 'idle' } as const const current = store.agentStatusByPaneKey?.[paneKey] if ( @@ -87,7 +87,11 @@ function projectStatus( current.tabId === tab.id && current.worktreeId === tab.worktreeId && current.terminalResumeEligible === false && - agentProviderSessionsEqual(tab.agentSessionAgent, current.providerSession, providerSession) + agentProviderSessionsEqual( + tab.agentSessionAgent, + current.providerSession, + summary.providerSession + ) ) { return } @@ -98,7 +102,7 @@ function projectStatus( undefined, { tabId: tab.id, worktreeId: tab.worktreeId }, { - ...(providerSession ? { providerSession } : {}), + ...(summary.providerSession ? { providerSession: summary.providerSession } : {}), terminalResumeEligible: false } ) @@ -112,13 +116,10 @@ function StructuredAgentSessionStatusProjection({ tab }: { tab: StructuredTab }) () => getActiveRuntimeTarget({ activeRuntimeEnvironmentId: environmentId }), [environmentId] ) - const { providerSession, state } = useStructuredAgentSessionReadObservation({ - sessionId: tab.entityId, - target - }) + const summary = useStructuredAgentSessionStatusSummary(tab.entityId, target) useEffect(() => { - projectStatus(tab, state, providerSession) - }, [providerSession, state, tab]) + projectStatus(tab, summary) + }, [summary, tab]) useEffect( () => () => useAppStore.getState().removeAgentStatus(structuredAgentSessionPaneKey(tab.id, tab.entityId)), diff --git a/src/renderer/src/components/native-chat/use-structured-agent-session-read.test.tsx b/src/renderer/src/components/native-chat/use-structured-agent-session-read.test.tsx index 2d320694c9c..9e1ce8067b7 100644 --- a/src/renderer/src/components/native-chat/use-structured-agent-session-read.test.tsx +++ b/src/renderer/src/components/native-chat/use-structured-agent-session-read.test.tsx @@ -19,10 +19,7 @@ vi.mock('@/runtime/structured-agent-session-client', () => ({ subscribeStructuredAgentSession: mocks.subscribe })) -import { - useStructuredAgentSessionRead, - useStructuredAgentSessionReadObservation -} from './use-structured-agent-session-read' +import { useStructuredAgentSessionRead } from './use-structured-agent-session-read' import { resetStructuredAgentSessionReadOwnersForTests } from './structured-agent-session-read-owner' const LOCAL_TARGET = { kind: 'local' } as const @@ -357,32 +354,6 @@ describe('useStructuredAgentSessionRead history window', () => { second.unmount() }) - it('shares one subscriber when pane and projection observe the same visible session', async () => { - const unsubscribe = vi.fn() - mocks.call.mockResolvedValue({ ok: true, page: page('tail', [], false) }) - mocks.subscribe.mockResolvedValue({ unsubscribe }) - - const view = renderHook(() => { - const pane = useStructuredAgentSessionRead({ - sessionId: 'session-shared', - target: LOCAL_TARGET, - isVisible: true - }) - const projection = useStructuredAgentSessionReadObservation({ - sessionId: 'session-shared', - target: LOCAL_TARGET - }) - return { pane, projection } - }) - - await waitFor(() => expect(mocks.subscribe).toHaveBeenCalledOnce()) - expect(mocks.call).toHaveBeenCalledOnce() - expect(view.result.current.pane.state).toBe(view.result.current.projection.state) - - view.unmount() - expect(unsubscribe).toHaveBeenCalledOnce() - }) - it('preserves cached state while switching away and refreshes once on re-entry', async () => { const unsubscribe = vi.fn() mocks.call.mockImplementation((_target, _method, params) => { diff --git a/src/renderer/src/components/native-chat/use-structured-agent-session-read.ts b/src/renderer/src/components/native-chat/use-structured-agent-session-read.ts index 894730f5e65..834d11d3fa1 100644 --- a/src/renderer/src/components/native-chat/use-structured-agent-session-read.ts +++ b/src/renderer/src/components/native-chat/use-structured-agent-session-read.ts @@ -20,13 +20,6 @@ function useReadOwnerSnapshot( return { owner, snapshot } } -export function useStructuredAgentSessionReadObservation(args: { - sessionId: string - target: RuntimeClientTarget -}): StructuredAgentSessionReadSnapshot { - return useReadOwnerSnapshot(args.sessionId, args.target).snapshot -} - export function useStructuredAgentSessionRead(args: { sessionId: string target: RuntimeClientTarget diff --git a/src/renderer/src/runtime/structured-agent-session-client.ts b/src/renderer/src/runtime/structured-agent-session-client.ts index 0e8d2ce16f2..71be3f3449d 100644 --- a/src/renderer/src/runtime/structured-agent-session-client.ts +++ b/src/renderer/src/runtime/structured-agent-session-client.ts @@ -1,5 +1,8 @@ import type { RuntimeRpcResponse } from '../../../shared/runtime-rpc-envelope' -import type { AgentSessionSubscribeEvent } from '../../../shared/agent-session-wire' +import type { + AgentSessionStatusEvent, + AgentSessionSubscribeEvent +} from '../../../shared/agent-session-wire' import { getRuntimeEnvironmentRevision } from './runtime-environment-revision' import { callRuntimeRpc, type RuntimeClientTarget } from './runtime-rpc-client' @@ -11,10 +14,11 @@ export function callStructuredAgentSession( return callRuntimeRpc(target, method, params) } -export async function subscribeStructuredAgentSession( +async function subscribeStructuredAgentSessionMethod( target: RuntimeClientTarget, + method: string, params: unknown, - onEvent: (event: AgentSessionSubscribeEvent) => void, + onEvent: (event: TEvent) => void, onError: (error: unknown) => void, onClose: () => void ): Promise<{ unsubscribe: () => void }> { @@ -23,15 +27,15 @@ export async function subscribeStructuredAgentSession( onError(response.error) return } - onEvent(response.result as AgentSessionSubscribeEvent) + onEvent(response.result as TEvent) } if (target.kind === 'local') { - return window.api.runtime.subscribe({ method: 'agentSession.subscribe', params }, onResponse) + return window.api.runtime.subscribe({ method, params }, onResponse) } return window.api.runtimeEnvironments.subscribe( { selector: target.environmentId, - method: 'agentSession.subscribe', + method, params, timeoutMs: 15_000, expectedEnvironmentPairingRevision: getRuntimeEnvironmentRevision(target.environmentId) @@ -39,3 +43,37 @@ export async function subscribeStructuredAgentSession( { onResponse, onError, onClose } ) } + +export function subscribeStructuredAgentSession( + target: RuntimeClientTarget, + params: unknown, + onEvent: (event: AgentSessionSubscribeEvent) => void, + onError: (error: unknown) => void, + onClose: () => void +): Promise<{ unsubscribe: () => void }> { + return subscribeStructuredAgentSessionMethod( + target, + 'agentSession.subscribe', + params, + onEvent, + onError, + onClose + ) +} + +/** Every structured session's projected status on one runtime, as the host publishes it. */ +export function subscribeStructuredAgentSessionStatus( + target: RuntimeClientTarget, + onEvent: (event: AgentSessionStatusEvent) => void, + onError: (error: unknown) => void, + onClose: () => void +): Promise<{ unsubscribe: () => void }> { + return subscribeStructuredAgentSessionMethod( + target, + 'agentSession.subscribeStatus', + {}, + onEvent, + onError, + onClose + ) +} diff --git a/src/renderer/src/runtime/structured-agent-session-status-feed.test.ts b/src/renderer/src/runtime/structured-agent-session-status-feed.test.ts new file mode 100644 index 00000000000..d9cb4dd3d72 --- /dev/null +++ b/src/renderer/src/runtime/structured-agent-session-status-feed.test.ts @@ -0,0 +1,140 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { + AgentSessionStatusEvent, + AgentSessionStatusSummary +} from '../../../shared/agent-session-wire' +import { AGENT_SESSION_STATUS_FEED_RUNTIME_CAPABILITY } from '../../../shared/protocol-version' + +const mocks = vi.hoisted(() => ({ + subscribeStatus: vi.fn(), + supportsCapability: vi.fn(), + unsubscribe: vi.fn() +})) + +vi.mock('./structured-agent-session-client', () => ({ + subscribeStructuredAgentSessionStatus: mocks.subscribeStatus +})) + +vi.mock('./runtime-rpc-client', () => ({ + runtimeEnvironmentSupportsCapability: mocks.supportsCapability +})) + +import { + getStructuredAgentSessionStatusFeed, + resetStructuredAgentSessionStatusFeedsForTests +} from './structured-agent-session-status-feed' + +const REMOTE = { kind: 'environment', environmentId: 'env-1' } as const +const LOCAL = { kind: 'local' } as const + +function summary( + sessionId: string, + status: AgentSessionStatusSummary['status'] = 'idle' +): AgentSessionStatusSummary { + return { + sessionId, + workspaceId: 'wt-1', + agent: 'codex', + status, + latestPrompt: 'hello', + updatedAt: 1 + } +} + +/** The event callback the feed handed to the most recent subscription. */ +function hostEmit(index = 0): (event: AgentSessionStatusEvent) => void { + const call = mocks.subscribeStatus.mock.calls[index] + if (!call) { + throw new Error('status feed not subscribed') + } + return call[1] as (event: AgentSessionStatusEvent) => void +} + +describe('structured agent session status feed', () => { + beforeEach(() => { + vi.useFakeTimers() + vi.clearAllMocks() + resetStructuredAgentSessionStatusFeedsForTests() + mocks.subscribeStatus.mockResolvedValue({ unsubscribe: mocks.unsubscribe }) + mocks.supportsCapability.mockResolvedValue(true) + }) + + afterEach(() => { + resetStructuredAgentSessionStatusFeedsForTests() + vi.useRealTimers() + }) + + it('never subscribes, and never retries, against a host without the status feed', async () => { + mocks.supportsCapability.mockResolvedValue(false) + getStructuredAgentSessionStatusFeed(REMOTE).activate() + + await vi.advanceTimersByTimeAsync(0) + expect(mocks.supportsCapability).toHaveBeenCalledWith( + 'env-1', + AGENT_SESSION_STATUS_FEED_RUNTIME_CAPABILITY + ) + expect(mocks.subscribeStatus).not.toHaveBeenCalled() + expect(vi.getTimerCount()).toBe(0) + + await vi.advanceTimersByTimeAsync(60_000) + expect(mocks.subscribeStatus).not.toHaveBeenCalled() + expect(mocks.supportsCapability).toHaveBeenCalledOnce() + }) + + it('subscribes once the remote host advertises the status feed', async () => { + getStructuredAgentSessionStatusFeed(REMOTE).activate() + + await vi.advanceTimersByTimeAsync(0) + expect(mocks.subscribeStatus).toHaveBeenCalledOnce() + expect(mocks.subscribeStatus.mock.calls[0]?.[0]).toEqual(REMOTE) + }) + + it('reconnects when the capability probe fails, which is not an answer', async () => { + mocks.supportsCapability.mockRejectedValue(new Error('relay unreachable')) + getStructuredAgentSessionStatusFeed(REMOTE).activate() + + await vi.advanceTimersByTimeAsync(0) + expect(mocks.supportsCapability).toHaveBeenCalledOnce() + + await vi.advanceTimersByTimeAsync(300) + expect(mocks.supportsCapability).toHaveBeenCalledTimes(2) + expect(mocks.subscribeStatus).not.toHaveBeenCalled() + }) + + it('probes nothing for a local host, which is this build', async () => { + getStructuredAgentSessionStatusFeed(LOCAL).activate() + + await vi.advanceTimersByTimeAsync(0) + expect(mocks.supportsCapability).not.toHaveBeenCalled() + expect(mocks.subscribeStatus).toHaveBeenCalledOnce() + }) + + it('merges a snapshot over the cached rows instead of retracting them', async () => { + const feed = getStructuredAgentSessionStatusFeed(LOCAL) + feed.activate() + await vi.advanceTimersByTimeAsync(0) + hostEmit()({ type: 'snapshot', sessions: [summary('session-1'), summary('session-2')] }) + expect([...feed.getSnapshot().keys()]).toEqual(['session-1', 'session-2']) + + // A restarted host restores its readable sessions after the stream reopens. + hostEmit()({ type: 'snapshot', sessions: [] }) + expect([...feed.getSnapshot().keys()]).toEqual(['session-1', 'session-2']) + + hostEmit()({ type: 'snapshot', sessions: [summary('session-1', 'working')] }) + expect(feed.getSnapshot().get('session-1')?.status).toBe('working') + expect(feed.getSnapshot().get('session-2')?.status).toBe('idle') + }) + + it('stops a pending reconnect when the feeds are reset between tests', async () => { + getStructuredAgentSessionStatusFeed(LOCAL).activate() + await vi.advanceTimersByTimeAsync(0) + hostEmit()({ type: 'end' }) + expect(vi.getTimerCount()).toBe(1) + + resetStructuredAgentSessionStatusFeedsForTests() + + expect(vi.getTimerCount()).toBe(0) + await vi.advanceTimersByTimeAsync(10_000) + expect(mocks.subscribeStatus).toHaveBeenCalledOnce() + }) +}) diff --git a/src/renderer/src/runtime/structured-agent-session-status-feed.ts b/src/renderer/src/runtime/structured-agent-session-status-feed.ts new file mode 100644 index 00000000000..2b550679315 --- /dev/null +++ b/src/renderer/src/runtime/structured-agent-session-status-feed.ts @@ -0,0 +1,211 @@ +// One host status stream per runtime target, shared by every session-list projection. +// +// The feed is a read-only mirror: the host projects each session's status from its journal and +// this owner keeps the latest summary per session while anyone is looking. Losing the stream +// keeps the cached summaries and reconnects; a fresh snapshot merges over them. +// Which sessions are listed is the tab map's decision, so the feed never retracts a summary. + +import type { + AgentSessionStatusEvent, + AgentSessionStatusSummary +} from '../../../shared/agent-session-wire' +import { AGENT_SESSION_STATUS_FEED_RUNTIME_CAPABILITY } from '../../../shared/protocol-version' +import { + runtimeEnvironmentSupportsCapability, + type RuntimeClientTarget +} from './runtime-rpc-client' +import { subscribeStructuredAgentSessionStatus } from './structured-agent-session-client' + +export type StructuredAgentSessionStatusSnapshot = ReadonlyMap + +export type StructuredAgentSessionStatusFeedOwner = { + activate: () => () => void + getSnapshot: () => StructuredAgentSessionStatusSnapshot + subscribe: (listener: () => void) => () => void +} + +const RECONNECT_MAX_DELAY_MS = 5_000 + +/** `stop` is the map's own teardown, not part of the owner contract callers hold. */ +type OwnedStatusFeed = StructuredAgentSessionStatusFeedOwner & { stop: () => void } + +const owners = new Map() + +export function structuredAgentSessionStatusFeedKey(target: RuntimeClientTarget): string { + return target.kind === 'local' ? 'local' : `environment:${target.environmentId}` +} + +function createOwner(target: RuntimeClientTarget): OwnedStatusFeed { + let snapshot: StructuredAgentSessionStatusSnapshot = new Map() + const listeners = new Set<() => void>() + const activations = new Set() + let generation = 0 + let handle: { unsubscribe: () => void } | null = null + let reconnectTimer: ReturnType | null = null + let reconnectAttempt = 0 + + const emit = (): void => { + for (const listener of listeners) { + listener() + } + } + const setSnapshot = (next: StructuredAgentSessionStatusSnapshot): void => { + snapshot = next + emit() + } + const applyEvent = (event: AgentSessionStatusEvent): void => { + if (event.type === 'snapshot') { + reconnectAttempt = 0 + // Merged, not replaced: a restarted host restores its readable sessions asynchronously, so + // the first snapshot can be empty and dropping those rows flickers every one to no-status. + const next = new Map(snapshot) + for (const session of event.sessions) { + next.set(session.sessionId, session) + } + setSnapshot(next) + return + } + if (event.type === 'status') { + const next = new Map(snapshot) + next.set(event.session.sessionId, event.session) + setSnapshot(next) + } + } + const active = (candidate: number): boolean => activations.size > 0 && candidate === generation + const clearReconnect = (): void => { + if (reconnectTimer) { + clearTimeout(reconnectTimer) + reconnectTimer = null + } + } + const dropHandle = (): void => { + handle?.unsubscribe() + handle = null + } + let open = (): void => {} + const scheduleReconnect = (candidate: number): void => { + if (!active(candidate) || reconnectTimer) { + return + } + const delay = Math.min(250 * 2 ** reconnectAttempt, RECONNECT_MAX_DELAY_MS) + reconnectAttempt += 1 + reconnectTimer = setTimeout(() => { + reconnectTimer = null + if (active(candidate)) { + open() + } + }, delay) + } + const subscribeToHost = (candidate: number): void => { + void subscribeStructuredAgentSessionStatus( + target, + (event) => { + if (!active(candidate)) { + return + } + if (event.type === 'end') { + dropHandle() + scheduleReconnect(candidate) + return + } + applyEvent(event) + }, + () => { + if (active(candidate)) { + dropHandle() + scheduleReconnect(candidate) + } + }, + () => { + if (active(candidate)) { + dropHandle() + scheduleReconnect(candidate) + } + } + ) + .then((opened) => { + if (active(candidate)) { + handle = opened + } else { + opened.unsubscribe() + } + }) + .catch(() => scheduleReconnect(candidate)) + } + open = (): void => { + const candidate = ++generation + dropHandle() + if (target.kind !== 'environment') { + // A local host is this build; only a remote one can predate the method. + subscribeToHost(candidate) + return + } + const environmentId = target.environmentId + void runtimeEnvironmentSupportsCapability( + environmentId, + AGENT_SESSION_STATUS_FEED_RUNTIME_CAPABILITY + ) + .then((supported) => { + if (!active(candidate)) { + return + } + // A host without the method is terminal, not a fault: retrying would relay-probe + // forever. A failed probe is not an answer, so that path still reconnects. + if (supported) { + subscribeToHost(candidate) + return + } + console.warn('[structured-session-status] host too old for the status feed', environmentId) + }) + .catch(() => scheduleReconnect(candidate)) + } + const stop = (): void => { + generation += 1 + clearReconnect() + dropHandle() + reconnectAttempt = 0 + } + + return { + activate: () => { + const token = Symbol('status-feed') + activations.add(token) + if (activations.size === 1) { + open() + } + return () => { + activations.delete(token) + if (activations.size === 0) { + stop() + } + } + }, + getSnapshot: () => snapshot, + subscribe: (listener) => { + listeners.add(listener) + return () => listeners.delete(listener) + }, + stop + } +} + +export function getStructuredAgentSessionStatusFeed( + target: RuntimeClientTarget +): StructuredAgentSessionStatusFeedOwner { + const key = structuredAgentSessionStatusFeedKey(target) + let owner = owners.get(key) + if (!owner) { + owner = createOwner(target) + owners.set(key, owner) + } + return owner +} + +export function resetStructuredAgentSessionStatusFeedsForTests(): void { + // Dropping the map alone leaves a live subscription and its pending reconnect running + // into the next test, where they reopen a stream nothing is holding. + for (const owner of owners.values()) { + owner.stop() + } + owners.clear() +} diff --git a/src/shared/agent-session-wire.ts b/src/shared/agent-session-wire.ts index 40a408df899..86c0e795d84 100644 --- a/src/shared/agent-session-wire.ts +++ b/src/shared/agent-session-wire.ts @@ -12,8 +12,13 @@ import type { AgentJournalResolution, AgentJournalSubmission } from './agent-session-journal-types' -import type { AgentSessionHandoffStage, AgentSessionOwnerRuntimeKind } from './agent-session-record' +import type { + AgentSessionHandoffStage, + AgentSessionOwnerRuntimeKind, + AgentSessionRecord +} from './agent-session-record' import type { AgentProviderSessionMetadata } from './agent-session-resume' +import type { StructuredAgentSessionProjectedStatus } from './structured-agent-session-projection' export type AgentSessionHandoffDirection = 'to-tui' | 'to-native' export type AgentSessionHandoffMode = 'now' | 'after-turn' | 'stop-turn' @@ -159,6 +164,29 @@ export type AgentSessionSubscribeEvent = } | { type: 'end' } +// ─── Status feed ──────────────────────────────────────────────────────────── + +/** What a session list needs to know about one session. The host projects it + * from the journal so no client has to replay a transcript to learn whether a + * turn is running. Additive surface: an older host has no such method. */ +export type AgentSessionStatusSummary = { + sessionId: string + workspaceId: string + agent: AgentSessionRecord['provider'] + /** Null until the journal holds a persisted user or assistant message. */ + status: StructuredAgentSessionProjectedStatus | null + latestPrompt: string + providerSession?: AgentProviderSessionMetadata + updatedAt: number +} + +/** A summary outlives its provider child: an evicted idle session is still idle, so the host + * keeps the last projection and never retracts one. Tabs, not this feed, decide what is listed. */ +export type AgentSessionStatusEvent = + | { type: 'snapshot'; sessions: AgentSessionStatusSummary[] } + | { type: 'status'; session: AgentSessionStatusSummary } + | { type: 'end' } + // ─── Mutation envelope ────────────────────────────────────────────────────── /** diff --git a/src/shared/protocol-version.ts b/src/shared/protocol-version.ts index 76e1252640a..159bb84f7e3 100644 --- a/src/shared/protocol-version.ts +++ b/src/shared/protocol-version.ts @@ -133,6 +133,10 @@ export const CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY = // to stop provider children after the last surface closes without tying lifetime to a transport. export const STRUCTURED_AGENT_SESSION_HOLD_RUNTIME_CAPABILITY = 'agent-session.structured.hold.v1' as const +// Why: agentSession.subscribeStatus is additive to a surface that already shipped, so a host +// advertising agent-session.structured.v1 may still answer it with method_not_found. Clients must +// probe before subscribing or they reconnect forever and never show any status at all. +export const AGENT_SESSION_STATUS_FEED_RUNTIME_CAPABILITY = 'agent-session.status-feed.v1' as const // Why: adding kimi to RESUMABLE_TUI_AGENTS grows terminal.ensureAgentSession's enum, and an // older host answers the unknown member with invalid_argument — a code the launch fallback does // not retry on — so clients must probe before taking the host-authority path. @@ -229,6 +233,7 @@ export const RUNTIME_CAPABILITIES = [ AGENT_SESSION_OMP_RESUME_PATH_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_HOLD_RUNTIME_CAPABILITY, + AGENT_SESSION_STATUS_FEED_RUNTIME_CAPABILITY, AGENT_SESSION_KIMI_RESUME_RUNTIME_CAPABILITY, FILE_MUTATION_OWNERSHIP_RUNTIME_CAPABILITY, GITHUB_MARK_PR_READY_RUNTIME_CAPABILITY, diff --git a/src/shared/structured-agent-session-projection.test.ts b/src/shared/structured-agent-session-projection.test.ts index 048051e70a5..8bdce30577e 100644 --- a/src/shared/structured-agent-session-projection.test.ts +++ b/src/shared/structured-agent-session-projection.test.ts @@ -1,4 +1,5 @@ import { describe, expect, it } from 'vitest' +import { AGENT_STATUS_MAX_FIELD_LENGTH } from './agent-status-field-normalization' import type { AgentJournalRenderItem } from './agent-session-journal-types' import { parsePaneKey } from './stable-pane-id' import { @@ -6,6 +7,7 @@ import { hasPersistedStructuredAgentSessionTurn, projectStructuredItemToNativeChat, projectStructuredAgentSessionStatus, + projectStructuredAgentSessionStatusSummary, structuredAgentSessionPaneKey } from './structured-agent-session-projection' @@ -44,6 +46,52 @@ describe('structured agent session status projection', () => { expect(projectStructuredAgentSessionStatus([running, completed])).toBe('idle') }) + it('summarizes status with the newest user prompt, and null before any persisted turn', () => { + const running = item('running', 3, { + kind: 'status', + text: 'Working', + turnLifecycle: { turnId: 'turn-1', state: 'running' } + }) + const first = item('first', 1, { + kind: 'message', + role: 'user', + blocks: [{ type: 'text', text: 'first' }] + }) + const second = item('second', 2, { + kind: 'message', + role: 'user', + blocks: [ + { type: 'text', text: 'second' }, + { type: 'text', text: 'line' } + ] + }) + + expect(projectStructuredAgentSessionStatusSummary([running])).toEqual({ + status: null, + latestPrompt: '' + }) + expect(projectStructuredAgentSessionStatusSummary([first, second, running])).toEqual({ + status: 'working', + latestPrompt: 'second line' + }) + expect(projectStructuredAgentSessionStatusSummary([first, second])).toEqual({ + status: 'idle', + latestPrompt: 'second line' + }) + }) + + it('bounds the wire prompt at the shared agent-status preview cap', () => { + const pasted = item('pasted', 1, { + kind: 'message', + role: 'user', + blocks: [{ type: 'text', text: 'x'.repeat(AGENT_STATUS_MAX_FIELD_LENGTH * 40) }] + }) + + expect(projectStructuredAgentSessionStatusSummary([pasted]).latestPrompt).toHaveLength( + AGENT_STATUS_MAX_FIELD_LENGTH + ) + }) + it('creates a deterministic pane identity for status stores', () => { const paneKey = structuredAgentSessionPaneKey('structured-agent-session-1', 'session-1') diff --git a/src/shared/structured-agent-session-projection.ts b/src/shared/structured-agent-session-projection.ts index 94938dc955f..6a5f01ba9ea 100644 --- a/src/shared/structured-agent-session-projection.ts +++ b/src/shared/structured-agent-session-projection.ts @@ -1,3 +1,4 @@ +import { normalizePromptField } from './agent-status-field-normalization' import type { AgentJournalRenderItem } from './agent-session-journal-types' import type { NativeChatBlock, NativeChatMessage } from './native-chat-types' import { sha256 } from './sha256' @@ -162,6 +163,34 @@ export function projectStructuredAgentSessionStatus( return activeStructuredAgentSessionTurnId(items) ? 'working' : 'idle' } +/** The newest user prompt, as the sidebar quotes it. */ +export function latestStructuredAgentSessionPrompt( + items: readonly AgentJournalRenderItem[] +): string { + for (let index = items.length - 1; index >= 0; index -= 1) { + const body = items[index]?.body + if (body?.kind === 'message' && body.role === 'user') { + return body.blocks.flatMap((block) => (block.type === 'text' ? [block.text] : [])).join('\n') + } + } + return '' +} + +/** One projection shared by host and client: null status means "no turn yet", not idle. + * The prompt is bounded to the same preview every other agent-status row carries — a send + * admits 256 KB, and one status frame carries every retained session at once. */ +export function projectStructuredAgentSessionStatusSummary( + items: readonly AgentJournalRenderItem[] +): { status: StructuredAgentSessionProjectedStatus | null; latestPrompt: string } { + if (!hasPersistedStructuredAgentSessionTurn(items)) { + return { status: null, latestPrompt: '' } + } + return { + status: projectStructuredAgentSessionStatus(items), + latestPrompt: normalizePromptField(latestStructuredAgentSessionPrompt(items)) + } +} + export function structuredAgentSessionPaneKey(tabId: string, sessionId: string): string { const bytes = sha256(new TextEncoder().encode(sessionId)) const hex = Array.from(bytes.slice(0, 16), (byte) => byte.toString(16).padStart(2, '0')).join('') diff --git a/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts b/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts index a2a64897fcb..ecf3072b39a 100644 --- a/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts +++ b/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts @@ -23,7 +23,10 @@ import { setStructuredAgentSessionHost } from '../../../src/main/native-chat/age import { AgentSessionRecordStore } from '../../../src/main/runtime/agent-session-record-store' import { computeAgentSessionPayloadFingerprint } from '../../../src/shared/agent-session-mutation-envelope' import type { AgentSessionSubscribeEvent } from '../../../src/shared/agent-session-wire' -import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../src/shared/protocol-version' +import { + AGENT_SESSION_STATUS_FEED_RUNTIME_CAPABILITY, + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY +} from '../../../src/shared/protocol-version' import { resolveBaselineReleaseRef } from './release-checkout' import { loadAgentSessionWireBuild, @@ -41,6 +44,7 @@ const WORKSPACE = 'workspace-1' const THREAD = '019fd532-7c11-7a90-b6de-4e1a2c3d5f60' const NOW = 1_800_000_000_000 const CLIENT_CAPABILITY_UPDATE_METHOD = 'runtime.clientCapabilities.update' +const STATUS_FEED_METHOD = 'agentSession.subscribeStatus' /** Every method the structured surface publishes: the host method it must reach, * and the result it must hand back. A gate that hides one method and leaks @@ -107,6 +111,12 @@ const STRUCTURED_CALLS: { // A subscription that opens with nothing to say answers with no reply at all, // so reaching the host is the only signal that the gate opened. { method: 'agentSession.subscribe', hostMethod: 'subscribe' }, + // The status feed opens with a snapshot of every session, so its first reply is the contract. + { + method: STATUS_FEED_METHOD, + hostMethod: 'subscribeStatus', + result: { type: 'snapshot', sessions: [] } + }, // Teardown runs through the runtime's subscription registry rather than the // host, so its reply is the only signal that the gate opened. { method: 'agentSession.unsubscribe', hostMethod: null, result: { unsubscribed: true } } @@ -320,6 +330,10 @@ function structuredHostStub(): Record> { readOptions: vi.fn(async () => ({ models: [], current: { model: 'gpt-live' } })), history: vi.fn(() => ({ ok: true, page: { items: [] } })), subscribe: vi.fn(() => () => undefined), + subscribeStatus: vi.fn((subscriber: { emit: (event: unknown) => void }) => { + subscriber.emit({ type: 'snapshot', sessions: [] }) + return () => undefined + }), unsubscribe: vi.fn() } } @@ -443,6 +457,14 @@ describe('cross-version structured agent sessions', () => { expect(baseline.capabilities.includes(STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY)).toBe( baselineStructuredMethods().length > 0 ) + // The status feed is additive to a surface that already shipped, so it carries its own + // capability or a client cannot tell "host too old" from "the call failed" — and it + // would relay-retry a method_not_found forever instead of degrading once. + for (const build of [current, baseline]) { + expect(build.capabilities.includes(AGENT_SESSION_STATUS_FEED_RUNTIME_CAPABILITY)).toBe( + build.methodNames.includes(STATUS_FEED_METHOD) + ) + } // Additive surface: bumping the protocol number would strand every paired // device on this release rather than degrade one feature. expect(current.protocolVersion).toBe(baseline.protocolVersion) From f8780a2c869dc84c543907e48de777b7e31d8f18 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Sat, 5 Sep 2026 15:35:39 -0700 Subject: [PATCH 07/23] feat(native-chat): stop monitored tasks individually (#18807) * feat(native-chat): stop monitored tasks individually * test: expect Claude task stop capability --------- Co-authored-by: Merge Sim --- .../claude-structured-control-actions.test.ts | 38 +++++ .../claude-structured-control-actions.ts | 7 +- .../claude-structured-session-adapter.ts | 34 +++-- .../claude-structured-session-close.test.ts | 12 +- .../structured-agent-session-adapter.ts | 6 +- ...structured-agent-session-host-mutations.ts | 1 + ...structured-agent-session-mutation-plans.ts | 10 +- .../structured-agent-session-turns.test.ts | 36 +++++ .../structured-agent-session-turns.ts | 4 +- ...ude-structured-session-integration.test.ts | 40 +++++ .../structured-agent-session-schemas.ts | 6 +- .../methods/structured-agent-session.test.ts | 29 ++++ .../NativeChatBackgroundTasksStatus.tsx | 74 +++++++--- .../NativeChatStructuredSession.test.tsx | 139 +++++++++++++++++- .../NativeChatStructuredSession.tsx | 50 ++++++- .../use-structured-agent-session.test.tsx | 36 +++++ .../use-structured-agent-session.ts | 6 +- src/renderer/src/i18n/locales/en.json | 2 + src/shared/agent-session-wire.ts | 2 + .../structured-agent-session-reducer.test.ts | 33 +++++ .../structured-agent-session-reducer.ts | 7 +- 21 files changed, 519 insertions(+), 53 deletions(-) diff --git a/src/main/claude/claude-structured-control-actions.test.ts b/src/main/claude/claude-structured-control-actions.test.ts index a90cb7908ba..168ce558f53 100644 --- a/src/main/claude/claude-structured-control-actions.test.ts +++ b/src/main/claude/claude-structured-control-actions.test.ts @@ -158,4 +158,42 @@ describe('stopClaudeBackgroundTasks', () => { await stopClaudeBackgroundTasks(session, undefined, () => current) expect(stopTask).toHaveBeenCalledTimes(1) }) + + it('stops only the requested live task id', async () => { + const backgroundTasks = new ClaudeBackgroundTaskTracker() + backgroundTasks.observe({ + type: 'system', + subtype: 'background_tasks_changed', + tasks: [ + { task_id: 'task-one', task_type: 'local_agent' }, + { task_id: 'task-two', task_type: 'local_bash' } + ] + }) + const stopTask = vi.fn(async (_taskId: string) => {}) + const session = { backgroundTasks, connection: { stopTask } } as unknown as ClaudeSession + + await expect( + stopClaudeBackgroundTasks(session, 5_000, () => true, 'task-two') + ).resolves.toEqual({ cancelled: true }) + expect(stopTask).toHaveBeenCalledWith('task-two', { timeoutMs: 5_000 }) + expect(stopTask).toHaveBeenCalledTimes(1) + }) + + it('refuses a stale or unknown task id without a provider call', async () => { + const backgroundTasks = new ClaudeBackgroundTaskTracker() + backgroundTasks.observe({ + type: 'system', + subtype: 'task_started', + task_id: 'task-live', + task_type: 'local_agent', + is_backgrounded: true + }) + const stopTask = vi.fn(async (_taskId: string) => {}) + const session = { backgroundTasks, connection: { stopTask } } as unknown as ClaudeSession + + await expect( + stopClaudeBackgroundTasks(session, undefined, () => true, 'task-stale') + ).resolves.toEqual({ cancelled: false }) + expect(stopTask).not.toHaveBeenCalled() + }) }) diff --git a/src/main/claude/claude-structured-control-actions.ts b/src/main/claude/claude-structured-control-actions.ts index d216304c311..8b3bb94c7b5 100644 --- a/src/main/claude/claude-structured-control-actions.ts +++ b/src/main/claude/claude-structured-control-actions.ts @@ -46,9 +46,12 @@ export async function cancelClaudeTurn( export async function stopClaudeBackgroundTasks( session: ClaudeSession, timeoutMs: number | undefined, - isCurrent: ClaudeTurnCancellationGuard = () => true + isCurrent: ClaudeTurnCancellationGuard = () => true, + taskId?: string ): Promise<{ cancelled: boolean }> { - const taskIds = session.backgroundTasks.stoppableTaskIds + const stoppableTaskIds = session.backgroundTasks.stoppableTaskIds + const taskIds = + taskId === undefined ? stoppableTaskIds : stoppableTaskIds.includes(taskId) ? [taskId] : [] let cancelled = false for (const taskId of taskIds) { if (!isCurrent()) { diff --git a/src/main/claude/claude-structured-session-adapter.ts b/src/main/claude/claude-structured-session-adapter.ts index 9bd92e1e839..c00a588e891 100644 --- a/src/main/claude/claude-structured-session-adapter.ts +++ b/src/main/claude/claude-structured-session-adapter.ts @@ -30,6 +30,7 @@ import { settleClaudeExitedSession } from './claude-structured-session-close' import { readClaudeTranscriptLeafWithReproof } from './claude-transcript-branch-proof' +import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire' export type { ClaudeStructuredLaunch } from './claude-structured-launch-resolution' export type { @@ -40,6 +41,11 @@ export type { const DISPATCH_ACK_TIMEOUT_MS = 10_000 +function backgroundTaskState(session: ClaudeSession): AgentSessionBackgroundTaskState | null { + const state = session.backgroundTasks.state + return state ? { ...state, supportsTaskStop: true } : null +} + export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAdapter { private readonly sessions = new Map() private readonly acquisitions = new ClaudeAcquisitionRegistry() @@ -192,7 +198,10 @@ export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAda session?.translator?.handle(event) this.deps.onEvent?.(event) if (backgroundTasksChanged) { - this.deps.onBackgroundTasksChanged?.(event.sessionId, session?.backgroundTasks.state ?? null) + this.deps.onBackgroundTasksChanged?.( + event.sessionId, + session ? backgroundTaskState(session) : null + ) } } @@ -233,18 +242,25 @@ export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAda stopBackgroundTasks: StructuredAgentSessionAdapter['stopBackgroundTasks'] = (input) => { const session = this.session(input.sessionId) const acquisitionGeneration = session.acquisitionGeneration - return stopClaudeBackgroundTasks(session, this.deps.requestTimeoutMs, () => - Boolean( - this.sessions.get(input.sessionId) === session && - session.fence === input.fence && - session.acquisitionGeneration === acquisitionGeneration && - session.backgroundTasks.state - ) + return stopClaudeBackgroundTasks( + session, + this.deps.requestTimeoutMs, + () => + Boolean( + this.sessions.get(input.sessionId) === session && + session.fence === input.fence && + session.acquisitionGeneration === acquisitionGeneration && + session.backgroundTasks.state + ), + input.taskId ) } backgroundTaskState: NonNullable = ( sessionId - ) => this.sessions.get(sessionId)?.backgroundTasks.state + ) => { + const session = this.sessions.get(sessionId) + return session ? backgroundTaskState(session) : undefined + } answerPrompt: StructuredAgentSessionAdapter['answerPrompt'] = (input) => answerClaudePrompt(this.session(input.sessionId), input) setOption: StructuredAgentSessionAdapter['setOption'] = (input) => diff --git a/src/main/claude/claude-structured-session-close.test.ts b/src/main/claude/claude-structured-session-close.test.ts index 0df0e913e50..670c0daaf8c 100644 --- a/src/main/claude/claude-structured-session-close.test.ts +++ b/src/main/claude/claude-structured-session-close.test.ts @@ -53,7 +53,11 @@ describe('Claude published session close lifecycle', () => { is_backgrounded: true }) expect(backgroundStates).toEqual([ - { state: 'monitoring', tasks: [{ id: 'background-1', kind: 'agent' }] } + { + state: 'monitoring', + tasks: [{ id: 'background-1', kind: 'agent' }], + supportsTaskStop: true + } ]) const session = ( adapter as unknown as { @@ -68,7 +72,11 @@ describe('Claude published session close lifecycle', () => { expect(events.filter((event) => event.type === 'handle')).toHaveLength(0) expect(disposeTranslator).toHaveBeenCalledOnce() expect(backgroundStates).toEqual([ - { state: 'monitoring', tasks: [{ id: 'background-1', kind: 'agent' }] }, + { + state: 'monitoring', + tasks: [{ id: 'background-1', kind: 'agent' }], + supportsTaskStop: true + }, null ]) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts index e44e8c39152..e6f8e478695 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts @@ -137,7 +137,11 @@ export type StructuredAgentSessionAdapter = { turnId: string fence: number }): Promise<{ cancelled: boolean }> - stopBackgroundTasks?(input: { sessionId: string; fence: number }): Promise<{ cancelled: boolean }> + stopBackgroundTasks?(input: { + sessionId: string + fence: number + taskId?: string + }): Promise<{ cancelled: boolean }> backgroundTaskState?(sessionId: string): AgentSessionBackgroundTaskState | null | undefined /** Fires the provider callback for an approval or a question. The wire calls * this only after the durable compare-and-set won, so it runs exactly once. */ diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts index 7d2648930c4..f4a0244d0af 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts @@ -78,6 +78,7 @@ export function cancelStructuredAgentSessionTurn( envelope: AgentSessionMutationEnvelope turnId: string scope?: 'background-tasks' + taskId?: string } ): Promise> { return mutate(context, caller, params.envelope, cancelPlan(params)) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-plans.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-plans.ts index d0eb905443c..1194f0c87ff 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-plans.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-plans.ts @@ -76,15 +76,21 @@ export function cancelPlan(params: { envelope: AgentSessionMutationEnvelope turnId: string scope?: 'background-tasks' + taskId?: string }): MutationPlan { return { method: 'agentSession.cancel', - fields: { turnId: params.turnId, ...(params.scope ? { scope: params.scope } : {}) }, + fields: { + turnId: params.turnId, + ...(params.scope ? { scope: params.scope } : {}), + ...(params.taskId ? { taskId: params.taskId } : {}) + }, run: (ctx) => performCancel(ctx, { clientOperationId: params.envelope.clientOperationId, turnId: params.turnId, - ...(params.scope ? { scope: params.scope } : {}) + ...(params.scope ? { scope: params.scope } : {}), + ...(params.taskId ? { taskId: params.taskId } : {}) }), // Interrupting twice would kill a turn the client never asked to stop, so a // replay reports the turn as already handled instead. diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts index e8e6f998bdd..31df2c44551 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts @@ -104,4 +104,40 @@ describe('performCancel', () => { expect(cancelTurn).not.toHaveBeenCalled() expect(journal.snapshot().items).toEqual([]) }) + + it('routes one background task id without interrupting the foreground turn or writing a row', async () => { + root = await mkdtemp(join(tmpdir(), 'orca-background-task-targeted-cancel-')) + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) + const cancelTurn = vi.fn(async () => ({ cancelled: true })) + const stopBackgroundTasks = vi.fn(async () => ({ cancelled: true })) + const ctx: AgentSessionTurnContext = { + sessionId: 'session-1', + journal, + fence: 1, + adapter: { cancelTurn, stopBackgroundTasks } as unknown as StructuredAgentSessionAdapter, + persistOptions: async () => undefined, + resolvedBy: 'client-1', + publish: vi.fn(), + now: () => 1 + } + + const result = await performCancel(ctx, { + clientOperationId: 'cancel-background-task-2', + turnId: 'background-tasks', + scope: 'background-tasks', + taskId: 'task-2' + }) + + expect(result).toEqual({ + ok: true, + value: { turnId: 'background-tasks', cancelled: true } + }) + expect(stopBackgroundTasks).toHaveBeenCalledWith({ + sessionId: 'session-1', + fence: 1, + taskId: 'task-2' + }) + expect(cancelTurn).not.toHaveBeenCalled() + expect(journal.snapshot().items).toEqual([]) + }) }) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts index 76c4e8e89d6..3bd98a61735 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts @@ -150,6 +150,7 @@ export async function performCancel( clientOperationId: string turnId: string scope?: 'background-tasks' + taskId?: string } ): Promise> { let cancelled = false @@ -159,7 +160,8 @@ export async function performCancel( ? ( await ctx.adapter.stopBackgroundTasks?.({ sessionId: ctx.sessionId, - fence: ctx.fence + fence: ctx.fence, + ...(input.taskId ? { taskId: input.taskId } : {}) }) )?.cancelled === true : ( diff --git a/src/main/runtime/claude-structured-session-integration.test.ts b/src/main/runtime/claude-structured-session-integration.test.ts index 2e42d595b04..e9cba45ffa9 100644 --- a/src/main/runtime/claude-structured-session-integration.test.ts +++ b/src/main/runtime/claude-structured-session-integration.test.ts @@ -599,6 +599,46 @@ describe('a structured Claude session over agentSession.*', () => { `claude:${PROVIDER_SESSION}:assistant-leaf` ) + claude.live().handlers.onMessage?.({ + type: 'system', + subtype: 'background_tasks_changed', + session_id: PROVIDER_SESSION, + uuid: 'background-roster', + tasks: [ + { task_id: 'task-one', task_type: 'local_agent', description: 'First task' }, + { task_id: 'task-two', task_type: 'local_bash', description: 'Second task' } + ] + }) + const itemsBeforeTaskStop = itemsOf(stream) + const targetedStopFields = { + turnId: 'background-tasks', + scope: 'background-tasks', + taskId: 'task-two' + } + await expect( + ok('agentSession.cancel', { + envelope: envelope('agentSession.cancel', targetedStopFields, created.fence), + ...targetedStopFields + }) + ).resolves.toMatchObject({ turnId: 'background-tasks', cancelled: true }) + expect(claude.live().calls.filter((entry) => entry.subtype === 'stop_task')).toEqual([ + { subtype: 'stop_task', params: { taskId: 'task-two' } } + ]) + expect(itemsOf(stream)).toEqual(itemsBeforeTaskStop) + + const staleStopFields = { + turnId: 'background-tasks', + scope: 'background-tasks', + taskId: 'task-stale' + } + await expect( + ok('agentSession.cancel', { + envelope: envelope('agentSession.cancel', staleStopFields, created.fence), + ...staleStopFields + }) + ).resolves.toMatchObject({ turnId: 'background-tasks', cancelled: false }) + expect(claude.live().calls.filter((entry) => entry.subtype === 'stop_task')).toHaveLength(1) + const answeredPermission = Promise.resolve( claude.live().handlers.canUseTool?.('Bash', { command: 'ls' }, { requestId: 'permission-1', diff --git a/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts b/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts index da66923c731..58dece2256c 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts @@ -152,9 +152,13 @@ export const CancelParams = z .object({ envelope: MutationEnvelope, turnId: Identifier('Invalid turn id'), - scope: z.literal('background-tasks').optional() + scope: z.literal('background-tasks').optional(), + taskId: Identifier('Invalid task id').optional() }) .strict() + .refine((value) => value.taskId === undefined || value.scope === 'background-tasks', { + message: 'A task id requires background-task scope' + }) export const RespondParams = z .object({ diff --git a/src/main/runtime/rpc/methods/structured-agent-session.test.ts b/src/main/runtime/rpc/methods/structured-agent-session.test.ts index 69b04711960..5220b3a00fe 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.test.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.test.ts @@ -529,6 +529,20 @@ describe('method routing', () => { ]) }) + it('routes an optional background task id through cancellation', async () => { + const params = { + envelope: envelope(), + turnId: 'background-tasks', + scope: 'background-tasks' as const, + taskId: 'task-2' + } + + const response = await call('agentSession.cancel', params, STRUCTURED_CLIENT) + + expect(response).toMatchObject({ ok: true }) + expect(hostCalls.cancel).toHaveBeenCalledWith(expect.anything(), params) + }) + it('routes the structured handoff mutation through the host', async () => { const response = await call('agentSession.requestHandoff', { envelope: envelope(), @@ -559,6 +573,21 @@ describe('parameter validation', () => { }) }) + it('rejects invalid or unscoped background task ids', async () => { + await rejects('agentSession.cancel', { + envelope: envelope(), + turnId: 'background-tasks', + scope: 'background-tasks', + taskId: ' task-2' + }) + await rejects('agentSession.cancel', { + envelope: envelope(), + turnId: 'turn-1', + taskId: 'task-2' + }) + expect(hostCalls.cancel).not.toHaveBeenCalled() + }) + it('refuses to let a client author anything but a user turn', async () => { await rejects( 'agentSession.send', diff --git a/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.tsx b/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.tsx index 5163d5f467b..9f8f08e1efd 100644 --- a/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.tsx +++ b/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.tsx @@ -25,8 +25,10 @@ function backgroundTaskLabel(task: AgentSessionBackgroundTask): string { export function NativeChatBackgroundTasksStatus(props: { tasks: readonly AgentSessionBackgroundTask[] - stopping: boolean - onStop: () => void + supportsTaskStop: boolean + stoppingTaskIds: ReadonlySet + stoppingAll: boolean + onStop: (taskId?: string) => void }): React.JSX.Element { const [expanded, setExpanded] = useState(false) const taskListId = useId() @@ -36,7 +38,7 @@ export function NativeChatBackgroundTasksStatus(props: { className="shrink-0 bg-background px-3 pt-2 sm:px-4" >
-
+
-
{expanded ? (
- {props.tasks.map((task) => ( -
  • -
  • - ))} + {props.tasks.map((task) => { + const label = backgroundTaskLabel(task) + return ( +
  • +
  • + ) + })} ) : (

    @@ -100,6 +115,23 @@ export function NativeChatBackgroundTasksStatus(props: { )}

    )} + {!props.supportsTaskStop ? ( +
    0 ? 'mt-2 border-t border-border pt-2' : 'mt-2'}> + +
    + ) : null}
    ) : null}
    diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx index afdf5ace7c5..2b4a9686aaf 100644 --- a/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx @@ -29,8 +29,9 @@ const mocks = vi.hoisted(() => ({ pasteFromClipboard: vi.fn(), submissions: [] as unknown[], monitoringBackgroundTasks: false, + supportsBackgroundTaskStop: false, backgroundTasks: [] as AgentSessionBackgroundTask[], - stopBackgroundTasks: vi.fn() + stopBackgroundTask: vi.fn() })) vi.mock('@/runtime/structured-agent-session-client', () => ({ @@ -75,10 +76,11 @@ vi.mock('./use-structured-agent-session', async () => { retry: outbox.retry, isWorking: false, isMonitoringBackgroundTasks: mocks.monitoringBackgroundTasks, + supportsBackgroundTaskStop: mocks.supportsBackgroundTaskStop, backgroundTasks: mocks.backgroundTasks, turnId: null, cancel: vi.fn(), - stopBackgroundTasks: mocks.stopBackgroundTasks, + stopBackgroundTask: (taskId?: string) => mocks.stopBackgroundTask(props.sessionId, taskId), respond: mocks.respond, optionSnapshot: [ { @@ -166,7 +168,8 @@ describe('NativeChatStructuredSession', () => { mocks.pasteFromClipboard.mockReset() mocks.submissions = [] mocks.monitoringBackgroundTasks = false - mocks.stopBackgroundTasks.mockReset() + mocks.supportsBackgroundTaskStop = false + mocks.stopBackgroundTask.mockReset() mocks.backgroundTasks = [] }) @@ -225,11 +228,12 @@ describe('NativeChatStructuredSession', () => { it('places background monitoring above the usable composer and stops without an active turn', async () => { mocks.monitoringBackgroundTasks = true + mocks.supportsBackgroundTaskStop = true mocks.backgroundTasks = [ { id: 'task-command', kind: 'command', description: 'sleep 180' }, { id: 'task-agent', kind: 'agent' } ] - mocks.stopBackgroundTasks.mockResolvedValue({ cancelled: true }) + mocks.stopBackgroundTask.mockResolvedValue({ cancelled: true }) render( { expect(status.compareDocumentPosition(composer) & Node.DOCUMENT_POSITION_FOLLOWING).toBeTruthy() expect(mocks.composerProps?.isWorking).toBe(false) expect(screen.queryByRole('list', { name: 'Running background tasks' })).toBeNull() + expect(screen.queryByRole('button', { name: /^Stop / })).toBeNull() const disclosure = screen.getByRole('button', { name: 'Monitoring background tasks' }) expect(disclosure.getAttribute('aria-expanded')).toBe('false') @@ -260,8 +265,130 @@ describe('NativeChatStructuredSession', () => { expect(screen.getByText('sleep 180')).toBeTruthy() expect(screen.getByText('Background agent')).toBeTruthy() - fireEvent.click(screen.getByRole('button', { name: 'Stop' })) - await waitFor(() => expect(mocks.stopBackgroundTasks).toHaveBeenCalledOnce()) + fireEvent.click(screen.getByRole('button', { name: 'Stop sleep 180' })) + await waitFor(() => + expect(mocks.stopBackgroundTask).toHaveBeenCalledWith('session-background', 'task-command') + ) + }) + + it('tracks concurrent task stops independently and clears each pending result', async () => { + mocks.monitoringBackgroundTasks = true + mocks.supportsBackgroundTaskStop = true + mocks.backgroundTasks = [ + { id: 'task-one', kind: 'command', description: 'First task' }, + { id: 'task-two', kind: 'command', description: 'Second task' } + ] + let finishFirst!: (value: unknown) => void + let finishSecond!: (value: unknown) => void + mocks.stopBackgroundTask.mockImplementation( + (_sessionId: string, taskId: string) => + new Promise((resolve) => { + if (taskId === 'task-one') { + finishFirst = resolve + } else { + finishSecond = resolve + } + }) + ) + + render( + + ) + fireEvent.click(screen.getByRole('button', { name: 'Monitoring background tasks' })) + const firstStop = screen.getByRole('button', { name: 'Stop First task' }) + const secondStop = screen.getByRole('button', { name: 'Stop Second task' }) + + fireEvent.click(firstStop) + fireEvent.click(secondStop) + expect((firstStop as HTMLButtonElement).disabled).toBe(true) + expect((secondStop as HTMLButtonElement).disabled).toBe(true) + + await act(async () => finishFirst({ cancelled: true })) + await waitFor(() => expect((firstStop as HTMLButtonElement).disabled).toBe(false)) + expect((secondStop as HTMLButtonElement).disabled).toBe(true) + + await act(async () => finishSecond(null)) + await waitFor(() => expect((secondStop as HTMLButtonElement).disabled).toBe(false)) + }) + + it('keeps a stale session stop result from clearing the current session pending state', async () => { + mocks.monitoringBackgroundTasks = true + mocks.supportsBackgroundTaskStop = true + mocks.backgroundTasks = [{ id: 'task-one', kind: 'command', description: 'Shared task' }] + let finishOld!: (value: unknown) => void + let finishCurrent!: (value: unknown) => void + mocks.stopBackgroundTask.mockImplementation( + (sessionId: string) => + new Promise((resolve) => { + if (sessionId === 'session-old') { + finishOld = resolve + } else { + finishCurrent = resolve + } + }) + ) + const { rerender } = render( + + ) + fireEvent.click(screen.getByRole('button', { name: 'Monitoring background tasks' })) + fireEvent.click(screen.getByRole('button', { name: 'Stop Shared task' })) + + rerender( + + ) + const currentStop = screen.getByRole('button', { name: 'Stop Shared task' }) + expect((currentStop as HTMLButtonElement).disabled).toBe(false) + fireEvent.click(currentStop) + expect((currentStop as HTMLButtonElement).disabled).toBe(true) + + await act(async () => finishOld({ cancelled: true })) + expect((currentStop as HTMLButtonElement).disabled).toBe(true) + await act(async () => finishCurrent({ cancelled: true })) + await waitFor(() => expect((currentStop as HTMLButtonElement).disabled).toBe(false)) + }) + + it('keeps the expanded all-task stop fallback for a taskless older host', async () => { + mocks.monitoringBackgroundTasks = true + mocks.stopBackgroundTask.mockResolvedValue({ cancelled: true }) + + render( + + ) + expect(screen.queryByRole('button', { name: 'Stop background tasks' })).toBeNull() + fireEvent.click(screen.getByRole('button', { name: 'Monitoring background tasks' })) + expect(screen.getByText('Task details are unavailable for this session.')).toBeTruthy() + fireEvent.click(screen.getByRole('button', { name: 'Stop background tasks' })) + + await waitFor(() => + expect(mocks.stopBackgroundTask).toHaveBeenCalledWith( + 'session-taskless-background', + undefined + ) + ) }) it('routes a bare model command to the native option picker', async () => { diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx index 9ac354f8a73..87adb0bda73 100644 --- a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx +++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx @@ -22,6 +22,14 @@ import { useStructuredNativeChatPaneCommands } from './use-structured-native-cha import type { NativeChatStructuredViewProps } from './native-chat-view-types' import { NativeChatBackgroundTasksStatus } from './NativeChatBackgroundTasksStatus' +type StoppingBackgroundTasks = { + sessionId: string + taskIds: ReadonlySet + all: boolean +} + +const NO_STOPPING_TASKS: ReadonlySet = new Set() + function encodeQuestionAnswer(questionId: string, answer: string): string { return `${encodeURIComponent(questionId)}:${encodeURIComponent(answer)}` } @@ -31,7 +39,8 @@ export function NativeChatStructuredSession( ): React.JSX.Element { const controller = useStructuredAgentSession(props) const [composerError, setComposerError] = useState(null) - const [stoppingBackgroundTasks, setStoppingBackgroundTasks] = useState(false) + const [stoppingBackgroundTasks, setStoppingBackgroundTasks] = + useState(null) const [optionPickerRequest, setOptionPickerRequest] = useState<{ id: string sequence: number @@ -83,6 +92,8 @@ export function NativeChatStructuredSession( const fileLinkContext = useNativeChatFileLinkContext(props.tabId) const imageRuntimeContext = useNativeChatImageRuntimeContext(props.tabId) const fileLinkClick = useNativeChatFileLinkClick(fileLinkContext) + const activeStoppingBackgroundTasks = + stoppingBackgroundTasks?.sessionId === props.sessionId ? stoppingBackgroundTasks : null const prompt = controller.prompts[0] ?? null const questionBody = prompt?.body.kind === 'question' ? prompt.body : null const questions = @@ -277,10 +288,39 @@ export function NativeChatStructuredSession( {controller.isMonitoringBackgroundTasks ? ( { - setStoppingBackgroundTasks(true) - void controller.stopBackgroundTasks().finally(() => setStoppingBackgroundTasks(false)) + supportsTaskStop={controller.supportsBackgroundTaskStop} + stoppingTaskIds={activeStoppingBackgroundTasks?.taskIds ?? NO_STOPPING_TASKS} + stoppingAll={activeStoppingBackgroundTasks?.all ?? false} + onStop={(taskId) => { + const targetSessionId = props.sessionId + setStoppingBackgroundTasks((current) => { + const taskIds = new Set( + current?.sessionId === targetSessionId ? current.taskIds : NO_STOPPING_TASKS + ) + if (taskId) { + taskIds.add(taskId) + } + return { + sessionId: targetSessionId, + taskIds, + all: taskId ? current?.sessionId === targetSessionId && current.all : true + } + }) + void controller.stopBackgroundTask(taskId).finally(() => { + setStoppingBackgroundTasks((current) => { + if (current?.sessionId !== targetSessionId) { + return current + } + const taskIds = new Set(current.taskIds) + if (taskId) { + taskIds.delete(taskId) + } + const all = taskId ? current.all : false + return taskIds.size === 0 && !all + ? null + : { sessionId: targetSessionId, taskIds, all } + }) + }) }} /> ) : null} diff --git a/src/renderer/src/components/native-chat/use-structured-agent-session.test.tsx b/src/renderer/src/components/native-chat/use-structured-agent-session.test.tsx index 2e611e4c726..9a42ccb6da6 100644 --- a/src/renderer/src/components/native-chat/use-structured-agent-session.test.tsx +++ b/src/renderer/src/components/native-chat/use-structured-agent-session.test.tsx @@ -296,4 +296,40 @@ describe('useStructuredAgentSession options', () => { expect(result.current.error).toBeNull() }) + + it('includes one background task id in the cancel fingerprint and payload', async () => { + mocks.call.mockImplementation((_target, method) => + method === 'agentSession.options' + ? Promise.resolve(OPTIONS) + : Promise.resolve({ + ok: true, + value: { turnId: 'background-tasks', cancelled: true } + }) + ) + const { result } = renderHook(() => + useStructuredAgentSession({ + sessionId: 'session-1', + target: LOCAL_TARGET, + agent: 'claude', + isVisible: true + }) + ) + + await act(async () => { + await expect(result.current.stopBackgroundTask('task-2')).resolves.toMatchObject({ + cancelled: true + }) + }) + + const mutation = mocks.call.mock.calls.find(([, method]) => method === 'agentSession.cancel') + expect(mutation?.[2]).toMatchObject({ + envelope: { + sessionId: 'session-1', + expectedRuntimeFence: 3 + }, + turnId: 'background-tasks', + scope: 'background-tasks', + taskId: 'task-2' + }) + }) }) diff --git a/src/renderer/src/components/native-chat/use-structured-agent-session.ts b/src/renderer/src/components/native-chat/use-structured-agent-session.ts index 3ae45bc3f43..8af9a36e0c4 100644 --- a/src/renderer/src/components/native-chat/use-structured-agent-session.ts +++ b/src/renderer/src/components/native-chat/use-structured-agent-session.ts @@ -245,12 +245,14 @@ export function useStructuredAgentSession(args: { isWorking: turnId !== null, isMonitoringBackgroundTasks, backgroundTasks: state.backgroundTasks?.tasks ?? [], + supportsBackgroundTaskStop: state.backgroundTasks?.supportsTaskStop === true, turnId, cancel: (turnId: string) => mutate('agentSession.cancel', 'agentSession.cancel', { turnId }), - stopBackgroundTasks: () => + stopBackgroundTask: (taskId?: string) => mutate('agentSession.cancel', 'agentSession.cancel', { turnId: 'background-tasks', - scope: 'background-tasks' + scope: 'background-tasks', + ...(taskId ? { taskId } : {}) }), respond: (item: StructuredPromptItem, optionId: string) => mutate( diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 4f9919db5c4..b91c89cfa5a 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -16989,6 +16989,8 @@ "backgroundTasks": { "monitoring": "Monitoring background tasks", "stop": "Stop", + "stopTask": "Stop {{value0}}", + "stopAll": "Stop background tasks", "agent": "Background agent", "workflow": "Background workflow", "command": "Background command", diff --git a/src/shared/agent-session-wire.ts b/src/shared/agent-session-wire.ts index 86c0e795d84..a0af962583d 100644 --- a/src/shared/agent-session-wire.ts +++ b/src/shared/agent-session-wire.ts @@ -64,6 +64,8 @@ export type AgentSessionBackgroundTaskState = { state: 'monitoring' /** Optional so mixed-version clients can consume state-only hosts. */ tasks?: AgentSessionBackgroundTask[] + /** Optional so clients only send targeted stops to hosts that accept them. */ + supportsTaskStop?: boolean } /** Backward paging is the client's normal read; 40 matches the page size the diff --git a/src/shared/structured-agent-session-reducer.test.ts b/src/shared/structured-agent-session-reducer.test.ts index 99ced770641..d36b6717758 100644 --- a/src/shared/structured-agent-session-reducer.test.ts +++ b/src/shared/structured-agent-session-reducer.test.ts @@ -54,6 +54,39 @@ function hydrationPage( } describe('structured agent session reducer', () => { + it('applies an additive targeted-stop capability update without journal churn', () => { + const backgroundTasks = { + state: 'monitoring' as const, + tasks: [{ id: 'task-1', kind: 'agent' as const }] + } + const initial = reduceStructuredAgentSession(EMPTY_STRUCTURED_AGENT_SESSION, { + type: 'event', + event: { + type: 'snapshot', + sessionId: 'session-a', + fence: 1, + page: { ...hydrationPage([]), backgroundTasks } + } + }) + const updated = reduceStructuredAgentSession(initial, { + type: 'event', + event: { + type: 'batch', + sessionId: 'session-a', + batch: { + cursor: { epoch: 'epoch-a', sequence: 0 }, + items: [], + removedItemIds: [], + submissions: [] + }, + backgroundTasks: { ...backgroundTasks, supportsTaskStop: true } + } + }) + + expect(updated.backgroundTasks).toEqual({ ...backgroundTasks, supportsTaskStop: true }) + expect(updated.items).toBe(initial.items) + }) + it('uses the bounded hydration page pagination boundary', () => { const restored = reduceStructuredAgentSession(EMPTY_STRUCTURED_AGENT_SESSION, { type: 'event', diff --git a/src/shared/structured-agent-session-reducer.ts b/src/shared/structured-agent-session-reducer.ts index 0a330539548..88d41b2f8e5 100644 --- a/src/shared/structured-agent-session-reducer.ts +++ b/src/shared/structured-agent-session-reducer.ts @@ -51,7 +51,12 @@ function backgroundTaskStatesEqual( if (left === right) { return true } - if (!left || !right || left.state !== right.state) { + if ( + !left || + !right || + left.state !== right.state || + left.supportsTaskStop !== right.supportsTaskStop + ) { return false } if (left.tasks === right.tasks) { From 8ab8c950be799e1ba4ce485c47882e18aa9fba54 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Sat, 5 Sep 2026 15:36:10 -0700 Subject: [PATCH 08/23] fix(native-chat): tell a pre-SQLite chat how to carry on (#18808) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A chat whose journal is still the pre-SQLite `log.jsonl` opened empty and indistinguishable from one created seconds ago. It now carries one status row naming the transcript still on disk and saying to send a message to continue, and read restore no longer drops such sessions — an unpublished chat had its tab pruned from persisted state, leaving nowhere for the message to appear. The notice survives a crash between the epoch commit and its append (re-offered while the epoch holds nothing) and stays out of a journal the same open just repaired, where it would have retired the unreconcilable_prefix marker and permanently ended provider-history recovery. No importer: the history is explained, not replayed. Nothing reads the remnant beyond its existence, and nothing moves or deletes it. --- .../journal-file-format-remnant.test.ts | 196 ++++++++++++++++++ .../journal-file-format-remnant.ts | 60 ++++++ .../journal-store-open.ts | 50 +++++ .../journal-store-restore.ts | 1 + ...uctured-agent-session-read-restore.test.ts | 110 ++++++++++ .../structured-agent-session-read-restore.ts | 19 +- 6 files changed, 433 insertions(+), 3 deletions(-) create mode 100644 src/main/native-chat/agent-session-journal/journal-file-format-remnant.test.ts create mode 100644 src/main/native-chat/agent-session-journal/journal-file-format-remnant.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.test.ts diff --git a/src/main/native-chat/agent-session-journal/journal-file-format-remnant.test.ts b/src/main/native-chat/agent-session-journal/journal-file-format-remnant.test.ts new file mode 100644 index 00000000000..d6424ce7952 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-file-format-remnant.test.ts @@ -0,0 +1,196 @@ +// An empty chat beside a pre-SQLite journal explains itself. +// +// The SQLite move shipped no importer, so a session whose history is a +// `log.jsonl` founds a fresh empty journal beside it and looks exactly like a +// chat created seconds ago. One status row is the difference. + +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import Database from '../../sqlite/sync-database' +import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key' +import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' +import { projectStructuredItemsToNativeChat } from '../../../shared/structured-agent-session-projection' +import { openJournalDatabase } from './journal-database' +import { JOURNAL_DB_SCHEMA_VERSION } from './journal-database-schema' +import { JOURNAL_FILE_FORMAT_REMNANT_DISCLOSURE_IDENTITY } from './journal-file-format-remnant' +import { loadJournal } from './journal-open' +import { journalDatabaseFile } from './journal-paths' +import type { AgentSessionJournal } from './journal-store' +import type { openAgentSessionJournal } from './journal-store-factory' +import { createTrackedJournalOpener } from './journal-store-test-open' + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-1', + workspaceId: 'ws-1', + hostId: 'host-1', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } +} + +const DISCLOSURE_ITEM_ID = agentJournalItemKey(JOURNAL_FILE_FORMAT_REMNANT_DISCLOSURE_IDENTITY) + +let root: string +let clock = 1_000 +const journals = createTrackedJournalOpener() + +function open(overrides: Partial[0]> = {}) { + return journals.open({ + identity: IDENTITY, + journalDir: root, + now: () => (clock += 1), + mintEpoch: () => `epoch-${clock}`, + ...overrides + }) +} + +function writeRemnant(name = 'log.jsonl'): Promise { + return writeFile(join(root, name), '{"kind":"epoch","v":1,"seq":1}\n', 'utf8') +} + +function disclosure(journal: AgentSessionJournal): string | null { + const row = journal.snapshot().items.find((entry) => entry.itemId === DISCLOSURE_ITEM_ID) + return row?.body.kind === 'status' ? row.body.text : null +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-journal-remnant-')) + clock = 1_000 +}) + +afterEach(async () => { + await journals.closeAll() + await rm(root, { recursive: true, force: true }) +}) + +describe('a chat whose history is still in the pre-SQLite format', () => { + it('says how to carry on, and where the transcript is', async () => { + await writeRemnant() + + const journal = await open() + + expect(disclosure(journal)).toContain('send a message to pick up where you left off') + expect(disclosure(journal)).toContain(join(root, 'log.jsonl')) + expect(disclosure(journal)).toContain('Codex') + }) + + // Both files is the normal shape of a pre-SQLite directory: every epoch roll + // staged a snapshot whether or not anything compacted into it, so preferring + // the snapshot would name an empty file for ~every affected chat. + it('names the log, not the snapshot staged beside it', async () => { + await writeRemnant('log.jsonl') + await writeRemnant('snapshot.json') + + const journal = await open() + + expect(disclosure(journal)).toContain(join(root, 'log.jsonl')) + expect(disclosure(journal)).not.toContain('snapshot.json') + }) + + it('falls back to the snapshot when a chat has no log beside it', async () => { + await writeRemnant('snapshot.json') + + const journal = await open() + + expect(disclosure(journal)).toContain(join(root, 'snapshot.json')) + }) + + it('says nothing to a chat that is genuinely new', async () => { + const journal = await open() + + expect(journal.snapshot().items).toEqual([]) + }) + + // Counting rows proves nothing here — the append upserts by identity, so a + // second append would still leave exactly one. The revision is what moves. + it('does not re-append the row on a later open', async () => { + await writeRemnant() + const first = await open() + const firstRevision = first + .snapshot() + .items.find((e) => e.itemId === DISCLOSURE_ITEM_ID)?.revision + await first.close() + + const reopened = await open() + + const row = reopened.snapshot().items.find((e) => e.itemId === DISCLOSURE_ITEM_ID) + expect(firstRevision).toBe(1) + expect(row?.revision).toBe(1) + expect(reopened.cursor().sequence).toBe(2) + }) + + // The epoch commit and this append are separate transactions; if the append is + // lost the epoch exists but holds nothing, and every later open takes the + // adopt branch. The offer has to survive that. + it('offers the message again when a committed epoch holds nothing', async () => { + const founded = await open() + await founded.close() + await writeRemnant() + + const reopened = await open() + + expect(disclosure(reopened)).toContain(join(root, 'log.jsonl')) + }) + + // A repair's epoch is the marker that history was deleted and never rebuilt, + // and any row that is not the repair's own disclosure retires it. Appending + // here would silently stop the session ever asking the provider for that + // history — with the journal still holding none. + it('stays out of a journal this open just repaired', async () => { + const journal = await open() + await journal.appendItem( + { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 0 }, + { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'history' }] }, + { fence: 1 } + ) + await journal.close() + // Deleting the anchor leaves every row unanchored: replay keeps nothing, so + // the repair publishes an empty `unreconcilable_prefix` epoch and — costing + // no malformed row — appends no disclosure of its own. That is the one state + // where this branch and a repair meet. + const opened = openJournalDatabase(journalDatabaseFile(root)) + try { + opened.db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(1) + } finally { + opened.db.close() + } + await writeRemnant() + + const repaired = await open() + + expect(disclosure(repaired)).toBeNull() + // Still asking the provider for the history the repair dropped. + expect(loadJournal(root, IDENTITY.sessionId)).toMatchObject({ corrupt: true }) + }) + + // A latched journal loads empty, so it reaches the same branch — and an append + // into one throws, which would make the session unopenable rather than read-only. + it('writes nothing into a journal latched by a newer schema', async () => { + const founded = await open() + await founded.close() + const db = new Database(journalDatabaseFile(root)) + try { + db.pragma(`user_version = ${JOURNAL_DB_SCHEMA_VERSION + 1}`) + } finally { + db.close() + } + await writeRemnant() + + const latched = await open() + + expect(latched.isReadOnly).toBe(true) + expect(disclosure(latched)).toBeNull() + }) + + // A row nothing projects is a row nobody reads. + it('renders in the transcript as a system line', async () => { + await writeRemnant() + + const journal = await open() + + const messages = projectStructuredItemsToNativeChat(journal.snapshot().items) + expect(messages).toHaveLength(1) + expect(messages[0]?.role).toBe('system') + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-file-format-remnant.ts b/src/main/native-chat/agent-session-journal/journal-file-format-remnant.ts new file mode 100644 index 00000000000..69dcc04bf64 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-file-format-remnant.ts @@ -0,0 +1,60 @@ +// A journal directory left behind by the pre-SQLite file format. +// +// Not `journal-legacy-import.ts`, which reads the PROVIDER's own transcript. +// This is Orca's own `log.jsonl`, which no build after the SQLite move reads. +// Nothing imports it, so the session it belonged to opens empty and is +// indistinguishable from a chat created seconds ago — same `session_created` +// epoch, same empty timeline. The remnant is the one durable fact that tells +// them apart, so the empty session says where its history went and how to carry +// on instead of silently claiming it never had any. + +import { existsSync } from 'node:fs' +import { join } from 'node:path' +import type { AgentJournalItemIdentity } from '../../../shared/agent-session-journal-types' +import { boundJournalStatusText } from './journal-prompt-body-bounds' +import { formatAgentTypeLabel } from '../../../shared/agent-type-label' +import type { AgentType } from '../../../shared/agent-status-types' + +/** The remnant's transcript, or null when the directory never held one. + * + * `log.jsonl` first, and the order matters: every epoch roll staged a + * `snapshot.json` whether or not anything was ever compacted into it, so the + * file's existence says nothing about where the history lives. Measured across + * a real profile, `compactedThrough` was 0 in all 80 — the log holds the + * transcript and the snapshot is the fallback for a session that has no log. */ +export function findJournalFileFormatRemnant(journalDir: string): string | null { + for (const name of ['log.jsonl', 'snapshot.json']) { + const path = join(journalDir, name) + if (existsSync(path)) { + return path + } + } + return null +} + +/** One stable identity, so a reopen upserts the same row instead of adding one. */ +export const JOURNAL_FILE_FORMAT_REMNANT_DISCLOSURE_IDENTITY: AgentJournalItemIdentity = { + provider: 'orca', + clientMessageId: 'journal-file-format-remnant' +} + +/** How to carry on. The session attaches on the record's own provider handle, so it + * still names the conversation the transcript no longer shows — whether the provider + * itself still holds that thread is its own business, hence "points at". */ +export function journalFileFormatRemnantDisclosure(input: { + transcriptPath: string + agent: AgentType +}): { identity: AgentJournalItemIdentity; body: { kind: 'status'; text: string } } { + return { + identity: JOURNAL_FILE_FORMAT_REMNANT_DISCLOSURE_IDENTITY, + body: { + kind: 'status', + text: boundJournalStatusText( + `This chat's history was saved in an older format Orca no longer reads, so it starts ` + + `empty. The session still points at the same ${formatAgentTypeLabel(input.agent)} ` + + `conversation — send a message to pick up where you left off. The original ` + + `transcript is on the session's host at \`${input.transcriptPath}\`` + ) + } + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-store-open.ts b/src/main/native-chat/agent-session-journal/journal-store-open.ts index e002048a094..721e5f4ba7f 100644 --- a/src/main/native-chat/agent-session-journal/journal-store-open.ts +++ b/src/main/native-chat/agent-session-journal/journal-store-open.ts @@ -1,7 +1,16 @@ import { mkdir } from 'node:fs/promises' +import type { AgentType } from '../../../shared/agent-status-types' +import { + findJournalFileFormatRemnant, + journalFileFormatRemnantDisclosure +} from './journal-file-format-remnant' import type { JournalLoad } from './journal-open' import { journalRepairDisclosure, type JournalRepairDisclosure } from './journal-repair-disclosure' +/** What any of this file's disclosures hands the store — a repair's, or the + * pre-SQLite notice's. Same shape, and neither is only a repair. */ +type JournalDisclosure = JournalRepairDisclosure + export async function ensureJournalDir(journalDir: string): Promise { await mkdir(journalDir, { recursive: true }) } @@ -32,6 +41,7 @@ export async function openJournalStoreState(input: { body: JournalRepairDisclosure['body'], fence: number ) => Promise + agent: AgentType highestFence: () => number malformedRows: () => number setMalformedRows: (count: number) => void @@ -40,6 +50,7 @@ export async function openJournalStoreState(input: { const loaded = input.loaded !== undefined ? input.loaded : input.replay() if (!loaded) { input.start() + await discloseFileFormatRemnant(input) return } input.adopt(loaded) @@ -59,4 +70,43 @@ export async function openJournalStoreState(input: { const disclosure = journalRepairDisclosure({ malformedRows: input.malformedRows() }) await input.appendDisclosure(disclosure.identity, disclosure.body, input.highestFence()) } + // Founding the epoch and appending the row are two transactions, and a + // committed epoch sends every later open down this branch instead. Anything + // that interrupts between them — a quit during startup restore, a failed + // append — would otherwise lose the message for good. An epoch holding nothing + // is exactly the state that append was owed, so offer it again. + // + // Never onto a repair, though: `loaded.state` is the PRE-repair load, so a + // journal this open just emptied looks identical. The repair's epoch is the + // marker that its history was deleted and never rebuilt, and any row that is + // not the repair's own disclosure retires it — this row would silently stop + // the session ever asking the provider for that history again. + if (!loaded.corrupt && loaded.state.items.size === 0 && loaded.state.submissions.size === 0) { + await discloseFileFormatRemnant(input) + } +} + +/** Says what happened to a chat whose history is in the abandoned file format. + * Upserts by a constant identity, so the offer above is exactly-once in effect: + * once the row exists the epoch is no longer empty. */ +async function discloseFileFormatRemnant(input: { + journalDir: string + agent: AgentType + appendDisclosure: ( + identity: JournalDisclosure['identity'], + body: JournalDisclosure['body'], + fence: number + ) => Promise + highestFence: () => number + readOnly: () => boolean +}): Promise { + if (input.readOnly()) { + return + } + const transcriptPath = findJournalFileFormatRemnant(input.journalDir) + if (!transcriptPath) { + return + } + const disclosure = journalFileFormatRemnantDisclosure({ transcriptPath, agent: input.agent }) + await input.appendDisclosure(disclosure.identity, disclosure.body, input.highestFence()) } diff --git a/src/main/native-chat/agent-session-journal/journal-store-restore.ts b/src/main/native-chat/agent-session-journal/journal-store-restore.ts index 53683a797d1..3a5d3c7ac6e 100644 --- a/src/main/native-chat/agent-session-journal/journal-store-restore.ts +++ b/src/main/native-chat/agent-session-journal/journal-store-restore.ts @@ -41,6 +41,7 @@ export function restoreJournalStore( adopt: host.adopt, appendDisclosure: (identity, body, fence) => host.journal().appendItem(identity, body, { fence }), + agent: host.identity.agent, highestFence: () => host.state().highestFence, malformedRows: host.malformedRows, setMalformedRows: host.setMalformedRows, diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.test.ts new file mode 100644 index 00000000000..34e3fb8a4cf --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.test.ts @@ -0,0 +1,110 @@ +// Read restore decides whether a session comes back at all. +// +// A chat still in the pre-SQLite format has no `journal.db`, so the probe that +// loads one reports nothing. Reading that as "no session" is what removed these +// chats: an unpublished session is also what prunes its tab out of the saved +// workspace, so the tab is gone before anything can explain itself. + +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { journalDirectoryFor } from '../agent-session-journal/journal-paths' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { restoreStructuredAgentSessionRead } from './structured-agent-session-read-restore' + +const SESSION_ID = 'codex_read_restore_fixture' +const WORKSPACE_ID = 'repo-1::/tmp/workspace' + +const RECORD = { + schemaVersion: 2, + sessionId: SESSION_ID, + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: WORKSPACE_ID, + workspaceKind: 'git-worktree' + }, + provider: 'codex', + providerHandleChain: [ + { + linkId: 'codex-1-thread-1', + handle: { provider: 'codex', threadId: 'thread-1' }, + origin: 'created', + mintedAtFence: 1, + observedAt: 1 + } + ], + accountHome: { variable: 'CODEX_HOME', path: '/tmp/codex-home' }, + createdAt: 1, + updatedAt: 2, + lease: { sessionId: SESSION_ID, runtimeKind: 'native', runtimeFence: 1 } +} as unknown as AgentSessionRecord + +const store = { + getRecord: (sessionId: string) => (sessionId === SESSION_ID ? RECORD : null) +} as unknown as AgentSessionRecordStore + +let journalRoot: string +const opened: AgentSessionJournal[] = [] + +async function writeRemnant(name: string): Promise { + const dir = journalDirectoryFor(journalRoot, { + workspaceId: WORKSPACE_ID, + sessionId: SESSION_ID + }) + await mkdir(dir, { recursive: true }) + await writeFile(join(dir, name), '{"kind":"epoch","v":1,"seq":1}\n', 'utf8') + return join(dir, name) +} + +beforeEach(async () => { + journalRoot = await mkdtemp(join(tmpdir(), 'orca-read-restore-')) +}) + +afterEach(async () => { + await Promise.allSettled(opened.splice(0).map((journal) => journal.close())) + await rm(journalRoot, { recursive: true, force: true }) +}) + +describe('a session whose journal is still the pre-SQLite format', () => { + it('is published, carrying the message that explains it', async () => { + const transcript = await writeRemnant('log.jsonl') + + const restored = await restoreStructuredAgentSessionRead(store, journalRoot, SESSION_ID) + + expect(restored).not.toBeNull() + opened.push(restored!.journal) + const disclosed = restored!.journal + .snapshot() + .items.map((entry) => (entry.body.kind === 'status' ? entry.body.text : '')) + expect(disclosed.join('')).toContain(transcript) + // Publishing it costs no agent process; acquisition still waits for the user. + expect(restored!.hasProviderChild).toBe(false) + }) + + it('is published for a remnant whose log is gone', async () => { + await writeRemnant('snapshot.json') + + const restored = await restoreStructuredAgentSessionRead(store, journalRoot, SESSION_ID) + + expect(restored).not.toBeNull() + opened.push(restored!.journal) + }) + + it('still drops a session with neither a journal nor a remnant', async () => { + const restored = await restoreStructuredAgentSessionRead(store, journalRoot, SESSION_ID) + + expect(restored).toBeNull() + }) + + it('still drops a session with no record', async () => { + await writeRemnant('log.jsonl') + + const restored = await restoreStructuredAgentSessionRead(store, journalRoot, 'unknown-session') + + expect(restored).toBeNull() + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.ts index 5bed8f2920e..34fd6452b08 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.ts @@ -3,6 +3,7 @@ import type { AgentSessionRecord } from '../../../shared/agent-session-record' import type { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { findJournalFileFormatRemnant } from '../agent-session-journal/journal-file-format-remnant' import { loadJournal } from '../agent-session-journal/journal-open' import { journalDirectoryFor } from '../agent-session-journal/journal-paths' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' @@ -40,15 +41,27 @@ export async function restoreStructuredAgentSessionRead( sessionId }) const loaded = loadJournal(journalDir, sessionId) - if (!loaded || loaded.corrupt) { + if (loaded?.corrupt) { + return null + } + // A session still in the pre-SQLite format has no `journal.db` to load. Dropping + // it here leaves it unpublished, which is also what prunes its tab out of the + // saved workspace — so the chat disappears with nowhere to explain itself. + if (!loaded && !findJournalFileFormatRemnant(journalDir)) { return null } const journal = await openAgentSessionJournal({ identity: journalIdentityFor(record, params), journalDir, - loaded + // Omitted, not `null`: the store reads `null` as "replay already ran and + // found nothing" and founds a fresh epoch. In process the probe above is the + // previous statement, so the window is zero-width; this holds the line for a + // database another process creates in between. + ...(loaded ? { loaded } : {}) }) - // Read restore opens the journal and nothing else: no adapter call, so no provider child. + // Read restore opens the journal and nothing else: no adapter call, so no + // provider child. Opening it can still write — a session whose history is in + // the old format founds its epoch and commits the row explaining that here. return { journal, params, From c58d7a0ecdc732dce615fbb548147e0e1a78a1ac Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Sat, 5 Sep 2026 15:37:52 -0700 Subject: [PATCH 09/23] fix(agent-session): never open a sibling terminal on an unproven create (#18735) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An `agentSession.create` the host could not confirm — it committed the session but could not publish its tab, and answered `agent_session_operation_unknown` — was rejected with a bare `Error` carrying a `code`. Nothing in the type said "unknown", so the verdict lived only in the code string, and the shared transport matcher was still free to re-read that error's *message*: an unknown refusal whose text ends in a definitive token (`Owner check failed: method_not_found`) classified as definitive, which is exactly the answer that permits a legacy sibling terminal. Make the class the verdict. `StructuredAgentSessionCreateUnknownOutcomeError` is a sibling of `StructuredAgentSessionCreateRefusalError`, not a subclass, so the nine existing `instanceof` consumers keep reading "refusal" as "you may fall back" with zero edits, and an unknown outcome flows down the lost-reply path instead — replaying the same envelope, re-publishing the tab the host failed to publish, and parking as visibility-unknown rather than creating anything. Classification now short-circuits on our own classes, so a message we wrote can never invert the verdict we already reached. Adds an end-to-end guard that drives the real classifier through `startStructuredAgentLaunch`: an unknown outcome opens zero legacy terminals, a definitive refusal opens exactly one. Ablating the branch turns that green suite red with `['legacy-terminal']` — the duplicate session the guard exists to prevent. Co-authored-by: Merge Sim --- .../launch-structured-agent-session.test.ts | 23 +- .../lib/launch-structured-agent-session.ts | 50 ++++- ...nt-session-launch-refusal-fallback.test.ts | 208 ++++++++++++++++++ 3 files changed, 268 insertions(+), 13 deletions(-) create mode 100644 src/renderer/src/lib/structured-agent-session-launch-refusal-fallback.test.ts diff --git a/src/renderer/src/lib/launch-structured-agent-session.test.ts b/src/renderer/src/lib/launch-structured-agent-session.test.ts index e9f65f3477b..d9a75ee2827 100644 --- a/src/renderer/src/lib/launch-structured-agent-session.test.ts +++ b/src/renderer/src/lib/launch-structured-agent-session.test.ts @@ -5,7 +5,8 @@ import { createStructuredAgentSessionLaunchIntent, isDefinitiveStructuredAgentSessionCreateError, launchStructuredAgentSession, - StructuredAgentSessionCreateRefusalError + StructuredAgentSessionCreateRefusalError, + StructuredAgentSessionCreateUnknownOutcomeError } from './launch-structured-agent-session' vi.mock('@/runtime/structured-agent-session-client', () => ({ @@ -256,11 +257,31 @@ describe('structured agent session launch', () => { createStructuredAgentSessionLaunchIntent('workspace-unknown', 'codex') ).catch((caught: unknown) => caught) + expect(error).toBeInstanceOf(StructuredAgentSessionCreateUnknownOutcomeError) expect(error).not.toBeInstanceOf(StructuredAgentSessionCreateRefusalError) expect(error).toMatchObject({ code: 'agent_session_operation_unknown' }) expect(isDefinitiveStructuredAgentSessionCreateError(error)).toBe(false) }) + /** The class is the verdict, so a refusal message that happens to end in a definitive token + * must not be re-read into one by the transport-error matcher. */ + it('keeps an unknown outcome unknown even when its message ends in a definitive token', async () => { + vi.mocked(callStructuredAgentSession).mockResolvedValue({ + ok: false, + refusal: { + code: 'agent_session_ownership_unknown', + message: 'Owner check failed: method_not_found' + } + }) + + const error = await launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-unknown-token', 'codex') + ).catch((caught: unknown) => caught) + + expect(error).toBeInstanceOf(StructuredAgentSessionCreateUnknownOutcomeError) + expect(isDefinitiveStructuredAgentSessionCreateError(error)).toBe(false) + }) + it('preserves a definitive refusal code for the fallback path', async () => { vi.mocked(callStructuredAgentSession).mockResolvedValue({ ok: false, diff --git a/src/renderer/src/lib/launch-structured-agent-session.ts b/src/renderer/src/lib/launch-structured-agent-session.ts index 6c2d1694437..3d7f94a1135 100644 --- a/src/renderer/src/lib/launch-structured-agent-session.ts +++ b/src/renderer/src/lib/launch-structured-agent-session.ts @@ -33,24 +33,52 @@ export type StructuredAgentSessionLaunchIntent = { params: StructuredAgentSessionCreateParams } -export class StructuredAgentSessionCreateRefusalError extends Error { +class StructuredAgentSessionCreateError extends Error { constructor( message: string, - readonly code: string = 'structured_agent_session_unsupported' + /** The wire refusal code, or the RPC error code when the create never reached a handler. */ + readonly code: string ) { super(message) + } +} + +/** + * The host proved it created nothing, so a caller may open a legacy terminal instead. The class + * itself is the verdict: `launchStructuredAgentSession` is the only place that decides it, against + * the shared allowlist, so no consumer has to remember to re-check a code. + */ +export class StructuredAgentSessionCreateRefusalError extends StructuredAgentSessionCreateError { + constructor(message: string, code: string = 'structured_agent_session_unsupported') { + super(message, code) this.name = 'StructuredAgentSessionCreateRefusalError' } } +/** + * Refused with a code that does not prove the session is absent. A sibling opened here would sit + * beside a session the host may already hold, so this deliberately is NOT a refusal error: it flows + * down the same path as a lost reply, which replays the intent and reconciles. + */ +export class StructuredAgentSessionCreateUnknownOutcomeError extends StructuredAgentSessionCreateError { + constructor(message: string, code: string) { + super(message, code) + this.name = 'StructuredAgentSessionCreateUnknownOutcomeError' + } +} + const DEFINITIVE_CREATE_FAILURE_CODES = [ 'structured_agent_session_unsupported', 'method_not_found' ] as const function definitiveStructuredAgentSessionCreateErrorCode(error: unknown): string | null { - if (error instanceof StructuredAgentSessionCreateRefusalError) { - return isDefinitiveAgentSessionCreateRefusal(error.code) ? error.code : null + if (error instanceof StructuredAgentSessionCreateError) { + // Our own classes already carry the verdict; message sniffing below could only invert it. + return error instanceof StructuredAgentSessionCreateRefusalError && + isDefinitiveAgentSessionCreateRefusal(error.code) + ? error.code + : null } for (const code of DEFINITIVE_CREATE_FAILURE_CODES) { if (hasRuntimeRpcErrorCode(error, code)) { @@ -200,15 +228,13 @@ export async function launchStructuredAgentSession( throw error } if (!result.ok) { - const error = new StructuredAgentSessionCreateRefusalError( - result.refusal.message, - result.refusal.code - ) - if (isDefinitiveStructuredAgentSessionCreateError(error)) { - abandonStructuredAgentSessionLaunchIntent(intent) - throw error + const { code, message } = result.refusal + if (!isDefinitiveAgentSessionCreateRefusal(code)) { + // Keep the focus intent: the session may exist, and recovery still has to adopt it. + throw new StructuredAgentSessionCreateUnknownOutcomeError(message, code) } - throw Object.assign(new Error(error.message), { code: error.code }) + abandonStructuredAgentSessionLaunchIntent(intent) + throw new StructuredAgentSessionCreateRefusalError(message, code) } return { sessionId: result.value.sessionId, fence: result.value.fence } } diff --git a/src/renderer/src/lib/structured-agent-session-launch-refusal-fallback.test.ts b/src/renderer/src/lib/structured-agent-session-launch-refusal-fallback.test.ts new file mode 100644 index 00000000000..45c41bd111e --- /dev/null +++ b/src/renderer/src/lib/structured-agent-session-launch-refusal-fallback.test.ts @@ -0,0 +1,208 @@ +// @vitest-environment happy-dom + +// The duplicate-session guard: which create refusals may open a legacy terminal beside the chat. +// Deliberately exercises the real `launch-structured-agent-session`, because the classification +// under test lives there — mocking it out would assert nothing. + +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { toast } from 'sonner' +import type { RuntimeMobileSessionTabsResult } from '../../../shared/runtime-session-contracts' +import { RuntimeRpcCallError } from '@/runtime/runtime-rpc-client' + +const mocks = vi.hoisted(() => ({ + call: vi.fn(), + refresh: vi.fn() +})) + +vi.mock('sonner', () => ({ + toast: { error: vi.fn(), message: vi.fn() } +})) + +vi.mock('@/i18n/i18n', () => ({ + translate: (_key: string, fallback: string, options?: { value0?: string }) => + fallback.replace('{{value0}}', options?.value0 ?? '') +})) + +vi.mock('@/lib/agent-catalog', () => ({ + getAgentCatalog: () => [{ id: 'codex', label: 'Codex' }] +})) + +vi.mock('@/runtime/structured-agent-session-client', () => ({ + callStructuredAgentSession: mocks.call +})) + +vi.mock('@/runtime/local-structured-session-tabs-sync', () => ({ + LOCAL_STRUCTURED_SESSION_OWNER: 'local', + refreshLocalStructuredSessionTabs: mocks.refresh +})) + +vi.mock('@/store', () => ({ + useAppStore: { + getState: () => ({ unifiedTabsByWorktree: {} }), + subscribe: () => () => {} + } +})) + +import { + StructuredAgentSessionCreateRefusalError, + StructuredAgentSessionCreateUnknownOutcomeError +} from '@/lib/launch-structured-agent-session' +import { + getStructuredAgentLaunchStatus, + startStructuredAgentLaunch +} from './structured-agent-session-launch' + +type CreateReply = { ok: boolean; refusal?: { code: string; message: string } } + +/** Replies to every `agentSession.create` in turn, repeating the last reply thereafter. */ +function replyToCreates(...replies: CreateReply[]): void { + let index = 0 + mocks.call.mockImplementation(async (_target: unknown, method: string, params: unknown) => { + if (method !== 'agentSession.create') { + return { ok: true, page: { fence: 1 } } + } + const reply = replies[Math.min(index, replies.length - 1)] + index += 1 + if (!reply.ok) { + return reply + } + const sessionId = (params as { envelope: { sessionId: string } }).envelope.sessionId + return { ok: true, replayed: index > 1, fence: 1, value: { sessionId, fence: 1 } } + }) +} + +function refused(code: string): CreateReply { + return { ok: false, refusal: { code, message: `create refused: ${code}` } } +} + +function publishedSnapshot(worktreeId: string, sessionId: string): RuntimeMobileSessionTabsResult { + return { + worktree: worktreeId, + publicationEpoch: 'epoch-1', + snapshotVersion: 1, + activeGroupId: null, + activeTabId: null, + activeTabType: null, + tabs: [ + { + type: 'agent-session', + id: 'tab-1', + title: 'Codex', + sessionId, + agent: 'codex', + isActive: true + } + ] + } +} + +async function flushLaunchSettlement(): Promise { + for (let i = 0; i < 20; i += 1) { + await Promise.resolve() + } +} + +describe('legacy terminal fallback after a refused structured create', () => { + beforeEach(() => { + vi.clearAllMocks() + localStorage.clear() + mocks.refresh.mockResolvedValue([]) + }) + + it.each(['agent_session_operation_unknown', 'agent_session_ownership_unknown'])( + 'opens no sibling terminal when the host answers %s', + async (code) => { + const worktreeId = `wt-${code}` + const legacyTerminals: string[] = [] + replyToCreates(refused(code)) + + const launch = startStructuredAgentLaunch(worktreeId, 'codex') + void launch.claimDefinitiveRefusalFallback(() => { + legacyTerminals.push('legacy-terminal') + }) + + await expect(launch.launchResult).rejects.toBeInstanceOf( + StructuredAgentSessionCreateUnknownOutcomeError + ) + await flushLaunchSettlement() + + // The host may already hold the session, so the user keeps exactly one thing: no chat it + // could confirm, and no terminal beside a session it could not rule out. + expect(legacyTerminals).toEqual([]) + expect(launch.isVisibilityUnknown()).toBe(true) + expect(toast.error).toHaveBeenCalledOnce() + } + ) + + it('adopts the session an unknown outcome had already created, without a sibling', async () => { + const worktreeId = 'wt-unknown-then-published' + const legacyTerminals: string[] = [] + replyToCreates(refused('agent_session_operation_unknown'), { ok: true }) + + const launch = startStructuredAgentLaunch(worktreeId, 'codex') + const fallbackRan = launch.claimDefinitiveRefusalFallback(() => { + legacyTerminals.push('legacy-terminal') + }) + mocks.refresh + .mockResolvedValueOnce([]) + .mockResolvedValue([publishedSnapshot(worktreeId, launch.sessionId)]) + + await expect(launch.launchResult).resolves.toEqual({ + sessionId: launch.sessionId, + fence: 1 + }) + await expect(fallbackRan).resolves.toBe(false) + await flushLaunchSettlement() + + expect(legacyTerminals).toEqual([]) + expect(toast.error).not.toHaveBeenCalled() + }) + + it('opens exactly one legacy terminal when the refusal is on the definitive allowlist', async () => { + const worktreeId = 'wt-unsupported' + const legacyTerminals: string[] = [] + replyToCreates(refused('structured_agent_session_unsupported')) + + const launch = startStructuredAgentLaunch(worktreeId, 'codex') + const fallbackRan = launch.claimDefinitiveRefusalFallback(() => { + legacyTerminals.push('legacy-terminal') + }) + + await expect(launch.launchResult).rejects.toBeInstanceOf( + StructuredAgentSessionCreateRefusalError + ) + await expect(fallbackRan).resolves.toBe(true) + await flushLaunchSettlement() + + expect(legacyTerminals).toEqual(['legacy-terminal']) + // A proven "nothing was created" needs no replay, so the terminal is the only surface open. + expect( + mocks.call.mock.calls.filter(([, method]) => method === 'agentSession.create') + ).toHaveLength(1) + expect(launch.isVisibilityUnknown()).toBe(false) + }) + + it('opens exactly one legacy terminal when an older runtime has no create method', async () => { + const legacyTerminals: string[] = [] + mocks.call.mockRejectedValue( + new RuntimeRpcCallError({ + id: 'rpc-old-runtime', + ok: false, + error: { code: 'method_not_found', message: 'Unknown method: agentSession.create' } + }) + ) + + const launch = startStructuredAgentLaunch('wt-old-runtime', 'codex') + const fallbackRan = launch.claimDefinitiveRefusalFallback(() => { + legacyTerminals.push('legacy-terminal') + }) + + await expect(launch.launchResult).rejects.toBeInstanceOf( + StructuredAgentSessionCreateRefusalError + ) + await expect(fallbackRan).resolves.toBe(true) + expect(legacyTerminals).toEqual(['legacy-terminal']) + expect(mocks.call).toHaveBeenCalledOnce() + expect(getStructuredAgentLaunchStatus('wt-old-runtime', 'codex')).toBe('idle') + }) +}) From 6a5c1f9535ee8d6b433eca58e9268ebc41d33e1a Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:02:14 -0700 Subject: [PATCH 10/23] refactor(agent-session): consolidate wire type imports below lint limit (#18930) --- .../structured-agent-session-host.ts | 31 +++++++------------ 1 file changed, 12 insertions(+), 19 deletions(-) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts index e4c7d191067..aef76c16cdb 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts @@ -2,17 +2,7 @@ // Mutations share one durable admission path and serialize per session. import type { AgentSessionExecutionLocation } from '../../../shared/agent-session-record' -import type { - AgentSessionAttachResult, - AgentSessionHistoryRequest, - AgentSessionHistoryResult, - AgentSessionHandoffRequest, - AgentSessionHandoffResult, - AgentSessionHandoffStatus, - AgentSessionMutationResult, - AgentSessionOptionsResult, - AgentSessionWireRefusal -} from '../../../shared/agent-session-wire' +import type * as SessionWire from '../../../shared/agent-session-wire' import type { AgentSessionAttachParams } from './structured-agent-session-attach' import { AGENT_SESSION_NOT_ATTACHED } from './structured-agent-session-mutation-admission' import { createRestartReconciler } from './structured-agent-session-restart-reconcile' @@ -77,7 +67,9 @@ export class StructuredAgentSessionHost { }) private readonly tasks = new StructuredAgentSessionTaskQueue() private readonly runtimeState: StructuredAgentSessionHostRuntimeState - private readonly reconcileLeases: (sessionId: string) => Promise + private readonly reconcileLeases: ( + sessionId: string + ) => Promise private readonly handoffs: StructuredAgentSessionHostHandoff private readonly readableRestorer: StructuredAgentSessionReadableRestorer private readonly restartRestore = new StructuredAgentSessionRestartRestoreGate() @@ -252,7 +244,7 @@ export class StructuredAgentSessionHost { attach( caller: StructuredAgentSessionCaller, params: AgentSessionAttachParams - ): Promise> { + ): Promise> { return attachStructuredAgentSession(this.attachContext(), caller.callerKey, params) } @@ -318,22 +310,23 @@ export class StructuredAgentSessionHost { requestHandoff = ( caller: StructuredAgentSessionCaller, - params: AgentSessionHandoffRequest - ): Promise> => + params: SessionWire.AgentSessionHandoffRequest + ): Promise> => this.handoffs.request(caller.callerKey, params) - readOptions = (sessionId: string): Promise => + readOptions = (sessionId: string): Promise => readStructuredAgentSessionOptions(this.mutationContext(), sessionId) - async handoffStatus(sessionId: string): Promise { + async handoffStatus(sessionId: string): Promise { this.requireSession(sessionId) return this.serialize(sessionId, () => refreshRecoverableStructuredHandoffStatus(this.handoffs, this.deps.store, sessionId) ) } - history = (request: AgentSessionHistoryRequest): AgentSessionHistoryResult => - this.backgroundTasks.history(request) + history = ( + request: SessionWire.AgentSessionHistoryRequest + ): SessionWire.AgentSessionHistoryResult => this.backgroundTasks.history(request) subscribe = (input: AgentSessionSubscribeInput): (() => void) => this.backgroundTasks.subscribe(input) From 08c3e854403e184f1b5badac67389dacee09bfe4 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:11:04 -0700 Subject: [PATCH 11/23] test(e2e): stabilize terminal launch and rename menu fixtures (#18928) --- ...ackground-terminal-mount-authority.spec.ts | 47 ++++++++++++------- tests/e2e/tab-rename.spec.ts | 2 +- 2 files changed, 31 insertions(+), 18 deletions(-) diff --git a/tests/e2e/live-background-terminal-mount-authority.spec.ts b/tests/e2e/live-background-terminal-mount-authority.spec.ts index ea6ffd1871a..a785454f695 100644 --- a/tests/e2e/live-background-terminal-mount-authority.spec.ts +++ b/tests/e2e/live-background-terminal-mount-authority.spec.ts @@ -23,6 +23,10 @@ import type { } from '../../src/shared/runtime-types' import { PROTOCOL_VERSION } from '../../src/main/daemon/types' import { makePaneKey } from '../../src/shared/stable-pane-id' +import { + buildFakeAgentCommandOverride, + FAKE_AGENT_WINDOWS_SHELL +} from './helpers/fake-agent-command-override' type SpawnEvent = { args: string[]; pid: number } type TerminalIdentity = Pick< @@ -70,6 +74,10 @@ if (process.platform === 'win32') { chmodSync(executable, 0o755) } +const fakeCodexCommand = buildFakeAgentCommandOverride( + path.join(fakeCliDir, process.platform === 'win32' ? 'codex.cmd' : 'codex') +) + const test = base.extend({ launchEnv: [ { @@ -535,23 +543,28 @@ test('adopts runtime-owned agent and Setup PTYs on first mount', async ({ const repoId = added.result.repo.id await expect .poll(() => - orcaPage.evaluate(async (repoId) => { - const state = window.__store?.getState() - await state?.fetchRepos() - const repo = window.__store?.getState().repos.find((candidate) => candidate.id === repoId) - if (!repo) { - return false - } - await window.__store?.getState().updateRepo(repoId, { - hookSettings: { ...repo.hookSettings, setupAgentStartupPolicy: 'start-immediately' } - }) - await window.__store?.getState().updateSettings({ - disabledTuiAgents: [], - setupScriptLaunchMode: 'new-tab', - terminalHiddenViewParking: false - }) - return true - }, repoId) + orcaPage.evaluate( + async ({ repoId, command, windowsShell }) => { + const state = window.__store?.getState() + await state?.fetchRepos() + const repo = window.__store?.getState().repos.find((candidate) => candidate.id === repoId) + if (!repo) { + return false + } + await window.__store?.getState().updateRepo(repoId, { + hookSettings: { ...repo.hookSettings, setupAgentStartupPolicy: 'start-immediately' } + }) + await window.__store?.getState().updateSettings({ + agentCmdOverrides: { codex: command }, + terminalWindowsShell: windowsShell, + disabledTuiAgents: [], + setupScriptLaunchMode: 'new-tab', + terminalHiddenViewParking: false + }) + return true + }, + { repoId, command: fakeCodexCommand, windowsShell: FAKE_AGENT_WINDOWS_SHELL } + ) ) .toBe(true) diff --git a/tests/e2e/tab-rename.spec.ts b/tests/e2e/tab-rename.spec.ts index 6e7f0a7fdc1..30cdb9175a9 100644 --- a/tests/e2e/tab-rename.spec.ts +++ b/tests/e2e/tab-rename.spec.ts @@ -126,7 +126,7 @@ test.describe('Tab Rename (Inline)', () => { expect(originalTitle.length).toBeGreaterThan(0) await tabLocatorByTitle(orcaPage, originalTitle).click({ button: 'right' }) - await orcaPage.getByRole('menuitem', { name: 'Change Title', exact: true }).click() + await orcaPage.getByRole('menuitem', { name: /^Change Title(?:\s|$)/ }).click() const renameInput = orcaPage.getByRole('textbox', { name: `Rename tab ${originalTitle}`, From a730becd7a6274b61b141b205b0d94f27f3a5e0b Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:12:52 -0700 Subject: [PATCH 12/23] fix(automation): keep explicit background launches off screen (#18898) --- config/scripts/run-electron-vite-dev.mjs | 2 +- .../createMainWindow-startup-reveal.test.ts | 28 ++++++++++ src/main/window/focus-existing-window.test.ts | 27 +++++++++- src/main/window/focus-existing-window.ts | 8 ++- .../foreground-activation-policy.test.ts | 51 +++++++++++++------ .../window/foreground-activation-policy.ts | 22 ++++---- tests/AGENTS.md | 12 +++-- 7 files changed, 116 insertions(+), 34 deletions(-) diff --git a/config/scripts/run-electron-vite-dev.mjs b/config/scripts/run-electron-vite-dev.mjs index dfb0a0aceb7..c520083cb6a 100644 --- a/config/scripts/run-electron-vite-dev.mjs +++ b/config/scripts/run-electron-vite-dev.mjs @@ -616,7 +616,7 @@ if (!isHelpOrVersion && process.env.ORCA_DEV_INSTANCE_LABEL) { // Why: automation launches this app while someone is working; announce that the // window will come up without taking the foreground so the mode is visible in logs. if (!isHelpOrVersion && process.env.ORCA_BACKGROUND_LAUNCH === '1') { - console.error('[orca-dev] Background launch: window shows without stealing focus') + console.error('[orca-dev] Background launch: window stays off screen; automate through CDP') } let forwardedExtras = [] if (!userPassedPort && !isHelpOrVersion) { diff --git a/src/main/window/createMainWindow-startup-reveal.test.ts b/src/main/window/createMainWindow-startup-reveal.test.ts index 1200132d319..f103881ea83 100644 --- a/src/main/window/createMainWindow-startup-reveal.test.ts +++ b/src/main/window/createMainWindow-startup-reveal.test.ts @@ -78,6 +78,34 @@ describe('createMainWindow', () => { } } + it.each(['darwin', 'linux', 'win32'] as const)( + 'keeps explicit background startup hidden through ready/load/fallback on %s', + (platform) => { + vi.useFakeTimers() + vi.stubEnv('ORCA_BACKGROUND_LAUNCH', '1') + const { browserWindowInstance, windowHandlers } = createStartupRevealWindowFixture() + const showInactive = vi.fn() + Object.assign(browserWindowInstance, { showInactive }) + try { + withPlatform(platform, () => { + createMainWindow(createStartupRevealStore(true) as never, { revealOnDidFinishLoad: true }) + const revealAfterLoad = browserWindowInstance.webContents.on.mock.calls.find( + ([event]) => event === 'did-finish-load' + )?.[1] + expect(revealAfterLoad).toBeTypeOf('function') + revealAfterLoad?.() + windowHandlers['ready-to-show']() + vi.advanceTimersByTime(10_000) + expect(browserWindowInstance.show).not.toHaveBeenCalled() + expect(showInactive).not.toHaveBeenCalled() + expect(browserWindowInstance.maximize).not.toHaveBeenCalled() + }) + } finally { + vi.unstubAllEnvs() + } + } + ) + it('ignores duplicate ready-to-show events after startup maximize has already run', () => { const { browserWindowInstance, windowHandlers } = createStartupRevealWindowFixture() diff --git a/src/main/window/focus-existing-window.test.ts b/src/main/window/focus-existing-window.test.ts index babb9a15490..9f5dc522150 100644 --- a/src/main/window/focus-existing-window.test.ts +++ b/src/main/window/focus-existing-window.test.ts @@ -1,5 +1,5 @@ import type { App, BrowserWindow } from 'electron' -import { describe, expect, it, vi } from 'vitest' +import { afterEach, describe, expect, it, vi } from 'vitest' import { focusExistingMainWindow } from './focus-existing-window' type FakeWindowOptions = { @@ -78,7 +78,32 @@ function makeTimer(): { } } +afterEach(() => vi.unstubAllEnvs()) + describe('focusExistingMainWindow', () => { + it.each(['darwin', 'linux', 'win32'] as const)( + 'never restores or activates a background window on %s', + (platform) => { + vi.stubEnv('ORCA_BACKGROUND_LAUNCH', '1') + vi.stubEnv('ORCA_E2E_FOREGROUND', '1') + const app = makeFakeApp() + const window = makeFakeWindow({ minimized: true }) + const timer = makeTimer() + focusExistingMainWindow({ + app, + getWindow: () => window, + openWindow: vi.fn(), + platform, + setTimeout: timer.setTimeout + }) + expect(app.focus).not.toHaveBeenCalled() + for (const call of Object.values(window.calls)) { + expect(call).not.toHaveBeenCalled() + } + expect(timer.scheduledMs()).toEqual([]) + } + ) + it('aggressively foregrounds an existing Windows window on second launch', () => { const app = makeFakeApp() const window = makeFakeWindow() diff --git a/src/main/window/focus-existing-window.ts b/src/main/window/focus-existing-window.ts index 4cadb49a743..903e6a8b321 100644 --- a/src/main/window/focus-existing-window.ts +++ b/src/main/window/focus-existing-window.ts @@ -1,5 +1,9 @@ import type { App, BrowserWindow } from 'electron' -import { isBackgroundLaunch, showWindowWithoutStealingFocus } from './foreground-activation-policy' +import { + isBackgroundLaunch, + isWindowlessLaunch, + showWindowWithoutStealingFocus +} from './foreground-activation-policy' type FocusTimer = (callback: () => void, ms: number) => unknown @@ -34,7 +38,7 @@ function safelyFocusApp(app: Pick): void { } export function safelyRevealWindow(window: BrowserWindow): void { - if (window.isDestroyed()) { + if (window.isDestroyed() || isWindowlessLaunch()) { return } if (window.isMinimized()) { diff --git a/src/main/window/foreground-activation-policy.test.ts b/src/main/window/foreground-activation-policy.test.ts index 0a45f00387e..3b33c9be881 100644 --- a/src/main/window/foreground-activation-policy.test.ts +++ b/src/main/window/foreground-activation-policy.test.ts @@ -32,6 +32,10 @@ describe('isBackgroundLaunch', () => { expect(isBackgroundLaunch({})).toBe(false) }) + it('keeps an explicit background request despite inherited foreground flags', () => { + expect(isBackgroundLaunch({ ORCA_BACKGROUND_LAUNCH: '1', ORCA_E2E_FOREGROUND: '1' })).toBe(true) + }) + it('lets native-focus specs opt back into the foreground', () => { expect(isBackgroundLaunch({ ORCA_E2E_HEADFUL: '1', ORCA_E2E_FOREGROUND: '1' })).toBe(false) expect(isWindowlessLaunch({ ORCA_E2E_HEADLESS: '1', ORCA_E2E_FOREGROUND: '1' })).toBe(false) @@ -39,10 +43,17 @@ describe('isBackgroundLaunch', () => { }) describe('isWindowlessLaunch', () => { - it('is headless-only; a headful run still paints', () => { + it('keeps explicit background launches hidden while headful E2E can paint', () => { expect(isWindowlessLaunch({ ORCA_E2E_HEADLESS: '1' })).toBe(true) expect(isWindowlessLaunch({ ORCA_E2E_HEADLESS: '1', ORCA_E2E_HEADFUL: '1' })).toBe(false) - expect(isWindowlessLaunch({ ORCA_BACKGROUND_LAUNCH: '1' })).toBe(false) + expect(isWindowlessLaunch({ ORCA_BACKGROUND_LAUNCH: '1' })).toBe(true) + expect( + isWindowlessLaunch({ + ORCA_BACKGROUND_LAUNCH: '1', + ORCA_E2E_HEADFUL: '1', + ORCA_E2E_FOREGROUND: '1' + }) + ).toBe(true) }) }) @@ -54,9 +65,16 @@ describe('showWindowWithoutStealingFocus', () => { expect(window.showInactive).not.toHaveBeenCalled() }) - it('shows a background window without activating it', () => { + it('never reveals an explicitly background window', () => { const window = makeWindow() showWindowWithoutStealingFocus(window, { ORCA_BACKGROUND_LAUNCH: '1' }) + expect(window.showInactive).not.toHaveBeenCalled() + expect(window.show).not.toHaveBeenCalled() + }) + + it('still reveals explicitly headful E2E without activation', () => { + const window = makeWindow() + showWindowWithoutStealingFocus(window, { ORCA_E2E_HEADFUL: '1' }) expect(window.showInactive).toHaveBeenCalledOnce() expect(window.show).not.toHaveBeenCalled() }) @@ -83,18 +101,21 @@ describe('applyBackgroundActivationPolicy', () => { } } - it('drops the macOS Dock tile and menu bar for headless runs', () => { - const app = makeApp() - expect( - applyBackgroundActivationPolicy({ - app, - env: { ORCA_E2E_HEADLESS: '1' }, - platform: 'darwin' - }) - ).toBe(true) - expect(app.dock.hide).toHaveBeenCalledOnce() - expect(app.setActivationPolicy).toHaveBeenCalledWith('accessory') - }) + it.each(['ORCA_E2E_HEADLESS', 'ORCA_BACKGROUND_LAUNCH'])( + 'drops the macOS Dock tile and menu bar for %s', + (flag) => { + const app = makeApp() + expect( + applyBackgroundActivationPolicy({ + app, + env: { [flag]: '1' }, + platform: 'darwin' + }) + ).toBe(true) + expect(app.dock.hide).toHaveBeenCalledOnce() + expect(app.setActivationPolicy).toHaveBeenCalledWith('accessory') + } + ) it('leaves a headful or user launch with its normal Dock presence', () => { const headful = makeApp() diff --git a/src/main/window/foreground-activation-policy.ts b/src/main/window/foreground-activation-policy.ts index c2ee6b19e73..5d51f50e487 100644 --- a/src/main/window/foreground-activation-policy.ts +++ b/src/main/window/foreground-activation-policy.ts @@ -5,8 +5,8 @@ import { app as electronApp, type BrowserWindow } from 'electron' * validation). These runs may use the machine, but must never take the OS * foreground away from whatever the developer is doing. * - * ORCA_BACKGROUND_LAUNCH=1 opts a normal launch in; ORCA_E2E_FOREGROUND=1 opts - * back out for the few specs whose subject *is* native focus (IME, key events). + * ORCA_BACKGROUND_LAUNCH=1 keeps automation off screen. Native-focus specs + * can use ORCA_E2E_FOREGROUND=1 only without an explicit background request. */ type ActivationPolicyApp = { @@ -19,19 +19,21 @@ type PolicyEnv = Readonly> /** True when this process must not steal focus, raise windows, or activate the app. */ export function isBackgroundLaunch(env: PolicyEnv = process.env): boolean { + if (env.ORCA_BACKGROUND_LAUNCH === '1') { + return true + } if (env.ORCA_E2E_FOREGROUND === '1') { return false } - return ( - env.ORCA_BACKGROUND_LAUNCH === '1' || - env.ORCA_E2E_HEADLESS === '1' || - env.ORCA_E2E_HEADFUL === '1' - ) + return env.ORCA_E2E_HEADLESS === '1' || env.ORCA_E2E_HEADFUL === '1' } -/** True when no window should reach the screen at all (headless E2E; Playwright drives via CDP). */ +/** True when no window should reach the screen at all (background or headless E2E; Playwright drives via CDP). */ export function isWindowlessLaunch(env: PolicyEnv = process.env): boolean { - return isBackgroundLaunch(env) && env.ORCA_E2E_HEADLESS === '1' && env.ORCA_E2E_HEADFUL !== '1' + return ( + env.ORCA_BACKGROUND_LAUNCH === '1' || + (isBackgroundLaunch(env) && env.ORCA_E2E_HEADLESS === '1' && env.ORCA_E2E_HEADFUL !== '1') + ) } /** @@ -63,7 +65,7 @@ export function applyBackgroundActivationPolicy( /** * Reveal a window without taking the foreground: hidden entirely when windowless, - * `showInactive()` (visible, not raised over the active app) in background launches. + * `showInactive()` for explicitly headful E2E runs. */ export function showWindowWithoutStealingFocus( window: BrowserWindow, diff --git a/tests/AGENTS.md b/tests/AGENTS.md index f445415e7df..26987a87c25 100644 --- a/tests/AGENTS.md +++ b/tests/AGENTS.md @@ -6,16 +6,18 @@ take the foreground — no window raised over the editor, no focus stolen, no Do `src/main/window/foreground-activation-policy.ts` enforces this in the main process. It is on whenever `ORCA_E2E_HEADLESS=1`, `ORCA_E2E_HEADFUL=1`, or `ORCA_BACKGROUND_LAUNCH=1`: -- headless → the window never reaches the screen (Playwright drives it via CDP) -- headful / background → `showInactive()`, no `app.focus({ steal: true })`, no +- headless / explicit background → the window never reaches the screen (Playwright drives it via CDP) +- headful without explicit background → `showInactive()`, no `app.focus({ steal: true })`, no `moveTop()`/always-on-top reinforcement -- macOS headless → `accessory` activation policy, so no Dock tile and no menu-bar takeover +- macOS headless / explicit background → `accessory` activation policy, so no Dock tile and no menu-bar takeover Rules when adding tests or scripts: - Launch through `tests/e2e/helpers/orca-app.ts` (or `orca-restart.ts`) — they already set the env. - A raw `electron.launch()` outside those helpers must pass `ORCA_BACKGROUND_LAUNCH: '1'`. -- Call `showInactive()`, never `show()`, when an `app.evaluate()` block reveals a window. +- Do not reveal windows in explicit background or headless runs. Only an explicitly headful run + may call `showInactive()`; never call `show()` or `bringToFront()` in automated background checks. - Tag a spec `@headful` only when it needs real pixels; it still runs in the background. - `ORCA_E2E_FOREGROUND=1` is the only opt-out, for runs whose subject _is_ native focus (IME and - other OS-level key injection). Add a comment saying why. + other OS-level key injection). Clear `ORCA_BACKGROUND_LAUNCH` for that isolated run and add a + comment saying why; an explicit background request takes precedence. From abdee9ebd370d3e2a7eb968b642df6e625846b33 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:16:08 -0700 Subject: [PATCH 13/23] feat(automations): restore column sorting on the list (#18885) The flat-table redesign in #16532 dropped the sort UI, orphaning AutomationListSortHeader, nextAutomationListSort and the whole AutomationListViewItem layer. Wire them back to the rendered list. Name and Last run become interactive header cells again; the other six columns stay plain text. Sorting now spans local and external rows as one list, so the panel renders per-row components from a single sorted collection instead of two independent sections. Two model fixes fall out of that: - View items key on the host-qualified row key, not the bare automation ID. The old builder predated automation-list-row-identity, so under All hosts two authorities returning the same ID collapsed in the sort tie-break. - sortAutomationListViewItems takes the locale as a parameter instead of reading getIntlLocale(). A hidden global read is invisible to a dependency array, and the list result is memoized. Keyboard traversal and focus recovery now read the sorted order, so arrow navigation matches what is on screen. The dead unified filter is removed in favor of the live row/entry filters the page already used. --- .../automations/AutomationListExternalRow.tsx | 260 +++++++++++ .../AutomationListExternalRows.tsx | 286 +----------- .../automations/AutomationListLocalRow.tsx | 391 +++++++++++++++++ .../automations/AutomationListLocalRows.tsx | 406 +----------------- .../automations/AutomationListSortHeader.tsx | 51 +++ .../AutomationListTableHeader.test.tsx | 45 +- .../automations/AutomationListTableHeader.tsx | 101 +++-- .../automations/AutomationsListPanel.test.tsx | 43 +- .../automations/AutomationsListPanel.tsx | 87 ++-- ...utomationsPage.create-destination.test.tsx | 2 +- ...tionsPage.cross-authority-actions.test.tsx | 9 +- .../AutomationsPage.external-scope.test.tsx | 7 +- .../AutomationsPage.notice-recovery.test.tsx | 2 +- ...AutomationsPage.refresh-selection.test.tsx | 7 +- .../AutomationsPage.run-visibility.test.tsx | 4 +- .../automations/AutomationsPage.test.tsx | 8 +- .../automations/AutomationsPageListPanel.tsx | 8 +- .../automation-list-view-sort.test.ts | 83 ++-- .../automations/automation-list-view.test.ts | 207 ++++----- .../automations/automation-list-view.ts | 90 ++-- .../automations-page-listed-items.ts | 32 ++ .../automations-page-test-harness.tsx | 65 ++- .../use-automations-page-list-state.ts | 22 +- .../use-automations-page-local-state.ts | 9 +- .../pane-agent-identity-inventory.test.ts | 2 +- 25 files changed, 1241 insertions(+), 986 deletions(-) create mode 100644 src/renderer/src/components/automations/AutomationListExternalRow.tsx create mode 100644 src/renderer/src/components/automations/AutomationListLocalRow.tsx create mode 100644 src/renderer/src/components/automations/AutomationListSortHeader.tsx create mode 100644 src/renderer/src/components/automations/automations-page-listed-items.ts diff --git a/src/renderer/src/components/automations/AutomationListExternalRow.tsx b/src/renderer/src/components/automations/AutomationListExternalRow.tsx new file mode 100644 index 00000000000..b26173467ab --- /dev/null +++ b/src/renderer/src/components/automations/AutomationListExternalRow.tsx @@ -0,0 +1,260 @@ +import React from 'react' +import { MoreHorizontal, Pause, Pencil, Play, Trash2 } from 'lucide-react' +import { + ContextMenu, + ContextMenuContent, + ContextMenuItem, + ContextMenuSeparator, + ContextMenuTrigger +} from '@/components/ui/context-menu' +import { + DropdownMenu, + DropdownMenuContent, + DropdownMenuItem, + DropdownMenuSeparator, + DropdownMenuTrigger +} from '@/components/ui/dropdown-menu' +import { Button } from '@/components/ui/button' +import { cn } from '@/lib/utils' +import type { + ExternalAutomationAction, + ExternalAutomationJob, + ExternalAutomationManager +} from '../../../../shared/automations-types' +import type { SshConnectionState } from '../../../../shared/ssh-types' +import type { ExternalAutomationListEntry } from './external-automation-list-entries' +import type { ExternalAutomationScope } from './external-automation-scope-client' +import { + formatExternalDate, + getExternalProviderLabel, + getExternalTargetKindLabel +} from './external-automation-display' +import { getExternalAutomationScheduleDisplay } from './external-automation-schedule-display' +import { getExternalAutomationActionDisabledMessage } from './external-automation-source-availability' +import { AUTOMATIONS_TABLE_GRID_CLASS } from './automations-table-layout' +import { + LIST_TABLE_ROW_CLASS, + LIST_TABLE_ROW_SELECTED_CLASS, + LIST_TABLE_STICKY_ROW_CELL_CLASS +} from '@/lib/list-table-layout' +import { isPortaledRowMenuClick, isRowActivationKey } from '@/lib/list-row-interaction' +import { getExternalAutomationLastRunSnapshot } from './automation-list-last-run' +import { AutomationListLastRunCell } from './AutomationListLastRunCell' +import { AutomationListStatusCell } from './AutomationListStatusCell' +import { translate } from '@/i18n/i18n' + +export type AutomationListExternalRowProps = { + entry: ExternalAutomationListEntry + selectedExternalKey: string | null | undefined + relativeNow: number + sshConnectionStates: ReadonlyMap> + externalActionKey: string | null + onSelect: (entryKey: string) => void + onRequestAction: ( + manager: ExternalAutomationManager, + job: ExternalAutomationJob, + action: ExternalAutomationAction, + scope: ExternalAutomationScope + ) => void + onEdit: ( + manager: ExternalAutomationManager, + job: ExternalAutomationJob, + scope: ExternalAutomationScope + ) => void +} + +export function AutomationListExternalRow({ + entry, + selectedExternalKey, + relativeNow, + sshConnectionStates, + externalActionKey, + onSelect, + onRequestAction, + onEdit +}: AutomationListExternalRowProps): React.JSX.Element { + const providerLabel = getExternalProviderLabel(entry.manager) + const targetKindLabel = getExternalTargetKindLabel(entry.manager) + const isSelected = selectedExternalKey === entry.key + const sshStatus = + entry.manager.target.type === 'ssh' + ? sshConnectionStates.get(entry.manager.target.connectionId)?.status + : undefined + const disabledMessage = getExternalAutomationActionDisabledMessage({ + manager: entry.manager, + providerLabel, + targetKindLabel, + sshStatus, + actionInProgress: externalActionKey !== null + }) + const actionDisabled = disabledMessage !== null + const scheduleLabel = getExternalAutomationScheduleDisplay(entry.manager, entry.job).label + const hostLabel = entry.manager.targetLabel || entry.manager.label || 'Local' + const projectLabel = entry.job.workdir ?? providerLabel + const nextRunLabel = entry.job.enabled + ? formatExternalDate(entry.job.nextRunAt, relativeNow) + : translate('auto.components.automations.AutomationsPage.paused', 'Paused') + const lastRunSnapshot = getExternalAutomationLastRunSnapshot(entry.job) + + return ( + + +
    { + // Why: Radix portals menus out of the row DOM, but React still + // bubbles those clicks here — ignore so menu actions don't open detail. + if (isPortaledRowMenuClick(event)) { + return + } + onSelect(entry.key) + }} + onKeyDown={(event) => { + if (!isRowActivationKey(event)) { + return + } + event.preventDefault() + onSelect(entry.key) + }} + className={cn( + AUTOMATIONS_TABLE_GRID_CLASS, + LIST_TABLE_ROW_CLASS, + isSelected && LIST_TABLE_ROW_SELECTED_CLASS + )} + > + + {entry.job.name} + + + {scheduleLabel} + + + {projectLabel} + + + {hostLabel} + + + {nextRunLabel} + + + + + {providerLabel} + + + + + + + onRequestAction(entry.manager, entry.job, 'run', entry.scope)} + > + + + {disabledMessage ?? + translate('auto.components.automations.AutomationsPage.2faecab10b', 'Run Now')} + + + {entry.manager.provider === 'hermes' ? ( + onEdit(entry.manager, entry.job, entry.scope)} + > + + {translate('auto.components.automations.AutomationsPage.f4612e3f78', 'Edit')} + + ) : null} + + onRequestAction( + entry.manager, + entry.job, + entry.job.enabled ? 'pause' : 'resume', + entry.scope + ) + } + > + {entry.job.enabled ? : } + {entry.job.enabled + ? translate('auto.components.automations.AutomationsPage.b457436d6a', 'Pause') + : translate('auto.components.automations.AutomationsPage.376631ef2b', 'Resume')} + + + onRequestAction(entry.manager, entry.job, 'delete', entry.scope)} + > + + {translate('auto.components.automations.AutomationsPage.15e0bfb13b', 'Delete')} + + + +
    +
    + + onRequestAction(entry.manager, entry.job, 'run', entry.scope)} + > + + + {disabledMessage ?? + translate('auto.components.automations.AutomationsPage.2faecab10b', 'Run Now')} + + + {entry.manager.provider === 'hermes' ? ( + onEdit(entry.manager, entry.job, entry.scope)} + > + + {translate('auto.components.automations.AutomationsPage.f4612e3f78', 'Edit')} + + ) : null} + + onRequestAction( + entry.manager, + entry.job, + entry.job.enabled ? 'pause' : 'resume', + entry.scope + ) + } + > + {entry.job.enabled ? : } + {entry.job.enabled + ? translate('auto.components.automations.AutomationsPage.b457436d6a', 'Pause') + : translate('auto.components.automations.AutomationsPage.376631ef2b', 'Resume')} + + + onRequestAction(entry.manager, entry.job, 'delete', entry.scope)} + > + + {translate('auto.components.automations.AutomationsPage.15e0bfb13b', 'Delete')} + + +
    + ) +} diff --git a/src/renderer/src/components/automations/AutomationListExternalRows.tsx b/src/renderer/src/components/automations/AutomationListExternalRows.tsx index 976a93b2433..3ed78fc2a69 100644 --- a/src/renderer/src/components/automations/AutomationListExternalRows.tsx +++ b/src/renderer/src/components/automations/AutomationListExternalRows.tsx @@ -1,285 +1,23 @@ import React from 'react' -import { MoreHorizontal, Pause, Pencil, Play, Trash2 } from 'lucide-react' -import { - ContextMenu, - ContextMenuContent, - ContextMenuItem, - ContextMenuSeparator, - ContextMenuTrigger -} from '@/components/ui/context-menu' -import { - DropdownMenu, - DropdownMenuContent, - DropdownMenuItem, - DropdownMenuSeparator, - DropdownMenuTrigger -} from '@/components/ui/dropdown-menu' -import { Button } from '@/components/ui/button' -import { cn } from '@/lib/utils' -import type { - ExternalAutomationAction, - ExternalAutomationJob, - ExternalAutomationManager -} from '../../../../shared/automations-types' -import type { SshConnectionState } from '../../../../shared/ssh-types' import type { ExternalAutomationListEntry } from './external-automation-list-entries' -import type { ExternalAutomationScope } from './external-automation-scope-client' import { - formatExternalDate, - getExternalProviderLabel, - getExternalTargetKindLabel -} from './external-automation-display' -import { getExternalAutomationScheduleDisplay } from './external-automation-schedule-display' -import { getExternalAutomationActionDisabledMessage } from './external-automation-source-availability' -import { AUTOMATIONS_TABLE_GRID_CLASS } from './automations-table-layout' -import { - LIST_TABLE_ROW_CLASS, - LIST_TABLE_ROW_SELECTED_CLASS, - LIST_TABLE_STICKY_ROW_CELL_CLASS -} from '@/lib/list-table-layout' -import { isPortaledRowMenuClick, isRowActivationKey } from '@/lib/list-row-interaction' -import { getExternalAutomationLastRunSnapshot } from './automation-list-last-run' -import { AutomationListLastRunCell } from './AutomationListLastRunCell' -import { AutomationListStatusCell } from './AutomationListStatusCell' -import { translate } from '@/i18n/i18n' + AutomationListExternalRow, + type AutomationListExternalRowProps +} from './AutomationListExternalRow' + +export type AutomationListExternalRowsProps = Omit & { + entries: readonly ExternalAutomationListEntry[] +} export function AutomationListExternalRows({ entries, - selectedExternalKey, - relativeNow, - sshConnectionStates, - externalActionKey, - onSelect, - onRequestAction, - onEdit -}: { - entries: readonly ExternalAutomationListEntry[] - selectedExternalKey: string | null | undefined - relativeNow: number - sshConnectionStates: ReadonlyMap> - externalActionKey: string | null - onSelect: (entryKey: string) => void - onRequestAction: ( - manager: ExternalAutomationManager, - job: ExternalAutomationJob, - action: ExternalAutomationAction, - scope: ExternalAutomationScope - ) => void - onEdit: ( - manager: ExternalAutomationManager, - job: ExternalAutomationJob, - scope: ExternalAutomationScope - ) => void -}): React.JSX.Element { + ...rowProps +}: AutomationListExternalRowsProps): React.JSX.Element { return ( <> - {entries.map((entry) => { - const providerLabel = getExternalProviderLabel(entry.manager) - const targetKindLabel = getExternalTargetKindLabel(entry.manager) - const isSelected = selectedExternalKey === entry.key - const sshStatus = - entry.manager.target.type === 'ssh' - ? sshConnectionStates.get(entry.manager.target.connectionId)?.status - : undefined - const disabledMessage = getExternalAutomationActionDisabledMessage({ - manager: entry.manager, - providerLabel, - targetKindLabel, - sshStatus, - actionInProgress: externalActionKey !== null - }) - const actionDisabled = disabledMessage !== null - const scheduleLabel = getExternalAutomationScheduleDisplay(entry.manager, entry.job).label - const hostLabel = entry.manager.targetLabel || entry.manager.label || 'Local' - const projectLabel = entry.job.workdir ?? providerLabel - const nextRunLabel = entry.job.enabled - ? formatExternalDate(entry.job.nextRunAt, relativeNow) - : translate('auto.components.automations.AutomationsPage.paused', 'Paused') - const lastRunSnapshot = getExternalAutomationLastRunSnapshot(entry.job) - - return ( - - -
    { - // Why: Radix portals menus out of the row DOM, but React still - // bubbles those clicks here — ignore so menu actions don't open detail. - if (isPortaledRowMenuClick(event)) { - return - } - onSelect(entry.key) - }} - onKeyDown={(event) => { - if (!isRowActivationKey(event)) { - return - } - event.preventDefault() - onSelect(entry.key) - }} - className={cn( - AUTOMATIONS_TABLE_GRID_CLASS, - LIST_TABLE_ROW_CLASS, - isSelected && LIST_TABLE_ROW_SELECTED_CLASS - )} - > - - {entry.job.name} - - - {scheduleLabel} - - - {projectLabel} - - - {hostLabel} - - - {nextRunLabel} - - - - - {providerLabel} - - - - - - - onRequestAction(entry.manager, entry.job, 'run', entry.scope)} - > - - - {disabledMessage ?? - translate( - 'auto.components.automations.AutomationsPage.2faecab10b', - 'Run Now' - )} - - - {entry.manager.provider === 'hermes' ? ( - onEdit(entry.manager, entry.job, entry.scope)} - > - - {translate( - 'auto.components.automations.AutomationsPage.f4612e3f78', - 'Edit' - )} - - ) : null} - - onRequestAction( - entry.manager, - entry.job, - entry.job.enabled ? 'pause' : 'resume', - entry.scope - ) - } - > - {entry.job.enabled ? ( - - ) : ( - - )} - {entry.job.enabled - ? translate( - 'auto.components.automations.AutomationsPage.b457436d6a', - 'Pause' - ) - : translate( - 'auto.components.automations.AutomationsPage.376631ef2b', - 'Resume' - )} - - - - onRequestAction(entry.manager, entry.job, 'delete', entry.scope) - } - > - - {translate( - 'auto.components.automations.AutomationsPage.15e0bfb13b', - 'Delete' - )} - - - -
    -
    - - onRequestAction(entry.manager, entry.job, 'run', entry.scope)} - > - - - {disabledMessage ?? - translate('auto.components.automations.AutomationsPage.2faecab10b', 'Run Now')} - - - {entry.manager.provider === 'hermes' ? ( - onEdit(entry.manager, entry.job, entry.scope)} - > - - {translate('auto.components.automations.AutomationsPage.f4612e3f78', 'Edit')} - - ) : null} - - onRequestAction( - entry.manager, - entry.job, - entry.job.enabled ? 'pause' : 'resume', - entry.scope - ) - } - > - {entry.job.enabled ? : } - {entry.job.enabled - ? translate('auto.components.automations.AutomationsPage.b457436d6a', 'Pause') - : translate('auto.components.automations.AutomationsPage.376631ef2b', 'Resume')} - - - onRequestAction(entry.manager, entry.job, 'delete', entry.scope)} - > - - {translate('auto.components.automations.AutomationsPage.15e0bfb13b', 'Delete')} - - -
    - ) - })} + {entries.map((entry) => ( + + ))} ) } diff --git a/src/renderer/src/components/automations/AutomationListLocalRow.tsx b/src/renderer/src/components/automations/AutomationListLocalRow.tsx new file mode 100644 index 00000000000..a9c1a5bc8b6 --- /dev/null +++ b/src/renderer/src/components/automations/AutomationListLocalRow.tsx @@ -0,0 +1,391 @@ +import React from 'react' +import { MoreHorizontal, Pause, Pencil, Play, Trash2 } from 'lucide-react' +import { + ContextMenu, + ContextMenuContent, + ContextMenuItem, + ContextMenuSeparator, + ContextMenuTrigger +} from '@/components/ui/context-menu' +import { + DropdownMenu, + DropdownMenuContent, + DropdownMenuItem, + DropdownMenuSeparator, + DropdownMenuTrigger +} from '@/components/ui/dropdown-menu' +import { Button } from '@/components/ui/button' +import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' +import { AgentIcon } from '@/lib/agent-catalog' +import { cn } from '@/lib/utils' +import type { AutomationRun } from '../../../../shared/automations-types' +import { getAutomationRunRepoId } from '../../../../shared/automation-run-identity' +import { formatUiAutomationSchedule } from './automation-schedule-label' +import { + getExecutionHostLabel, + getLocalExecutionHostLabel, + getRepoExecutionHostId +} from '../../../../shared/execution-host' +import type { SshConnectionState } from '../../../../shared/ssh-types' +import type { ProjectHostSetup } from '../../../../shared/project-types' +import type { Repo } from '../../../../shared/repo-types' +import type { Worktree } from '../../../../shared/worktree/types' +import type { RuntimeStatus } from '../../../../shared/runtime-types' +import type { TaskSourceHostAvailability } from '../task-source-context-summary' +import type { AutomationRowAction } from './automation-captured-owner' +import type { AutomationHostTarget } from './automation-host-client' +import { + getAutomationRowLastRunSnapshot, + getLocalAutomationLastRunSnapshot +} from './automation-list-last-run' +import { AutomationListLastRunCell } from './AutomationListLastRunCell' +import { formatAutomationDateTimeWithRelative } from './automation-page-parts' +import { getAutomationTargetAvailability } from './automation-target-availability' +import { getAgentLabel } from './automation-draft-model' +import type { AutomationListRow } from './automation-list-row-identity' +import { + formatAutomationCost, + formatAutomationTokens, + type AutomationUsageSummary +} from './automation-usage-model' +import { AUTOMATIONS_TABLE_GRID_CLASS } from './automations-table-layout' +import { + LIST_TABLE_ROW_CLASS, + LIST_TABLE_ROW_SELECTED_CLASS, + LIST_TABLE_STICKY_ROW_CELL_CLASS +} from '@/lib/list-table-layout' +import { isPortaledRowMenuClick, isRowActivationKey } from '@/lib/list-row-interaction' +import { AutomationListStatusCell } from './AutomationListStatusCell' +import { translate } from '@/i18n/i18n' + +export type AutomationListLocalRowProps = { + row: AutomationListRow + selectedRowKey: string | null | undefined + isSelectedLocal: boolean + lastRunByAutomationId: ReadonlyMap + relativeNow: number + repoMap: ReadonlyMap + worktreeMap: ReadonlyMap + repoForRow?: (row: AutomationListRow) => Repo | undefined + worktreeForRow?: (row: AutomationListRow, repo: Repo | undefined) => Worktree | undefined + projectHostSetups: readonly ProjectHostSetup[] + sshConnectionStates: ReadonlyMap> + runtimeStatusByEnvironmentId: ReadonlyMap< + string, + { status: RuntimeStatus | null; checkedAt: number } + > + hostTargetFor: (row: AutomationListRow) => AutomationHostTarget | null + automationSourceHostAvailabilityByRowKey: ReadonlyMap + hostLabelById?: ReadonlyMap + isActionEnabled?: (row: AutomationListRow, action: AutomationRowAction) => boolean + onSelect: (rowKey: string) => void + onRunNow: (row: AutomationListRow) => void + onEdit: (row: AutomationListRow) => void + onToggle: (row: AutomationListRow) => void + onDelete: (row: AutomationListRow) => void +} + +const EMPTY_HOST_LABELS: ReadonlyMap = new Map() + +function automationUsageText(summary: AutomationUsageSummary | undefined): string { + if (!summary || summary.unavailableRuns > 0) { + return summary?.knownRuns + ? usageAmountText(summary) + : translate( + 'auto.components.automations.AutomationsPage.usageUnavailable', + 'Usage unavailable' + ) + } + return summary.knownRuns > 0 + ? usageAmountText(summary) + : translate('auto.components.automations.AutomationsPage.noRunUsageYet', 'No run usage yet') +} + +function usageAmountText(summary: AutomationUsageSummary): string { + return translate( + 'auto.components.automations.AutomationsPage.runUsageSummary', + '{{cost}} est. · {{tokens}} tokens', + { + cost: formatAutomationCost(summary.estimatedCostUsd), + tokens: formatAutomationTokens(summary.totalTokens) + } + ) +} + +export function AutomationListLocalRow({ + row, + selectedRowKey, + isSelectedLocal, + lastRunByAutomationId, + relativeNow, + repoMap, + worktreeMap, + repoForRow, + worktreeForRow, + projectHostSetups, + sshConnectionStates, + runtimeStatusByEnvironmentId, + hostTargetFor, + automationSourceHostAvailabilityByRowKey, + hostLabelById = EMPTY_HOST_LABELS, + isActionEnabled, + onSelect, + onRunNow, + onEdit, + onToggle, + onDelete +}: AutomationListLocalRowProps): React.JSX.Element { + const allows = (row: AutomationListRow, action: AutomationRowAction): boolean => + isActionEnabled?.(row, action) ?? true + const { automation } = row + const automationRepo = repoForRow?.(row) ?? repoMap.get(getAutomationRunRepoId(automation)) + const automationWorktree = automation.workspaceId + ? (worktreeForRow?.(row, automationRepo) ?? worktreeMap.get(automation.workspaceId)) + : null + const automationRunAvailability = getAutomationTargetAvailability({ + automation, + repo: automationRepo, + workspace: automationWorktree, + projectHostSetups, + sshConnectionStates, + runtimeStatusByEnvironmentId, + automationHostTarget: hostTargetFor(row), + sourceHostAvailability: automationSourceHostAvailabilityByRowKey.get(row.key) + }) + const projectLabel = + automationRepo?.displayName ?? + translate('auto.components.automations.AutomationsPage.13118faadf', 'Unknown project') + const scheduleLabel = formatUiAutomationSchedule(automation.rrule) + const nextRunLabel = automation.enabled + ? formatAutomationDateTimeWithRelative(automation.nextRunAt, relativeNow) + : translate('auto.components.automations.enablement.paused', 'Paused') + const isSelected = isSelectedLocal && selectedRowKey === row.key + const agentLabel = getAgentLabel(automation.agentId) + const hostId = + automation.runContext?.hostId ?? + (automationRepo ? getRepoExecutionHostId(automationRepo) : null) + const hostLabel = + row.hostLabel || + (hostId + ? (hostLabelById.get(hostId) ?? getExecutionHostLabel(hostId)) + : getLocalExecutionHostLabel()) + const agentTooltipLabel = `${agentLabel} · ${hostLabel} · ${automationUsageText(row.usageSummary ?? undefined)}` + const canRunNow = automationRunAvailability.canRunNow && allows(row, 'run') + const lastRun = lastRunByAutomationId.get(automation.id) + // Without a fetched run, the row's projected summary carries the newest + // retained run's status — the list never downloads run history for this. + const lastRunSnapshot = lastRun + ? getLocalAutomationLastRunSnapshot(automation, lastRun) + : getAutomationRowLastRunSnapshot(row) + + const actionItems = ( + <> + onRunNow(row)} + /> + } + label={translate('auto.components.automations.AutomationsPage.f4612e3f78', 'Edit')} + onSelect={() => onEdit(row)} + /> + : } + label={ + automation.enabled + ? translate('auto.components.automations.AutomationsPage.b457436d6a', 'Pause') + : translate('auto.components.automations.AutomationsPage.376631ef2b', 'Resume') + } + onSelect={() => onToggle(row)} + /> + + } + label={translate('auto.components.automations.AutomationsPage.15e0bfb13b', 'Delete')} + variant="destructive" + onSelect={() => onDelete(row)} + /> + + ) + + return ( + + +
    { + // Why: Radix portals menus out of the row DOM, but React still + // bubbles those clicks here — ignore so menu actions don't open detail. + if (isPortaledRowMenuClick(event)) { + return + } + onSelect(row.key) + }} + onKeyDown={(event) => { + if (!isRowActivationKey(event)) { + return + } + event.preventDefault() + onSelect(row.key) + }} + className={cn( + AUTOMATIONS_TABLE_GRID_CLASS, + LIST_TABLE_ROW_CLASS, + isSelected && LIST_TABLE_ROW_SELECTED_CLASS + )} + > + + {automation.name} + + + {scheduleLabel} + + + {projectLabel} + + + {hostLabel} + + + {nextRunLabel} + + + + + + + + + + + {agentTooltipLabel} + + + + + + + + { + if (canRunNow) { + onRunNow(row) + } + }} + > + + + {automationRunAvailability.canRunNow + ? translate('auto.components.automations.AutomationsPage.2faecab10b', 'Run Now') + : automationRunAvailability.message} + + + onEdit(row)}> + + {translate('auto.components.automations.AutomationsPage.f4612e3f78', 'Edit')} + + onToggle(row)}> + {automation.enabled ? ( + + ) : ( + + )} + {automation.enabled + ? translate('auto.components.automations.AutomationsPage.b457436d6a', 'Pause') + : translate('auto.components.automations.AutomationsPage.376631ef2b', 'Resume')} + + + onDelete(row)} + > + + {translate('auto.components.automations.AutomationsPage.15e0bfb13b', 'Delete')} + + + +
    +
    + {actionItems} +
    + ) +} + +function MenuRunItem({ + disabled, + label, + onSelect +}: { + disabled: boolean + label: string + onSelect: () => void +}): React.JSX.Element { + return ( + { + if (disabled) { + event.preventDefault() + return + } + onSelect() + }} + > + + {label} + + ) +} + +function MenuItem({ + disabled, + icon, + label, + onSelect, + variant +}: { + disabled?: boolean + icon: React.ReactNode + label: string + onSelect: () => void + variant?: 'destructive' +}): React.JSX.Element { + return ( + + {icon} + {label} + + ) +} + +function MenuSeparator(): React.JSX.Element { + return +} diff --git a/src/renderer/src/components/automations/AutomationListLocalRows.tsx b/src/renderer/src/components/automations/AutomationListLocalRows.tsx index 292eb545b4d..3fa02cc1884 100644 --- a/src/renderer/src/components/automations/AutomationListLocalRows.tsx +++ b/src/renderer/src/components/automations/AutomationListLocalRows.tsx @@ -1,414 +1,20 @@ import React from 'react' -import { MoreHorizontal, Pause, Pencil, Play, Trash2 } from 'lucide-react' -import { - ContextMenu, - ContextMenuContent, - ContextMenuItem, - ContextMenuSeparator, - ContextMenuTrigger -} from '@/components/ui/context-menu' -import { - DropdownMenu, - DropdownMenuContent, - DropdownMenuItem, - DropdownMenuSeparator, - DropdownMenuTrigger -} from '@/components/ui/dropdown-menu' -import { Button } from '@/components/ui/button' -import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' -import { AgentIcon } from '@/lib/agent-catalog' -import { cn } from '@/lib/utils' -import type { AutomationRun } from '../../../../shared/automations-types' -import { getAutomationRunRepoId } from '../../../../shared/automation-run-identity' -import { formatUiAutomationSchedule } from './automation-schedule-label' -import { - getExecutionHostLabel, - getLocalExecutionHostLabel, - getRepoExecutionHostId -} from '../../../../shared/execution-host' -import type { SshConnectionState } from '../../../../shared/ssh-types' -import type { ProjectHostSetup } from '../../../../shared/project-types' -import type { Repo } from '../../../../shared/repo-types' -import type { Worktree } from '../../../../shared/worktree/types' -import type { RuntimeStatus } from '../../../../shared/runtime-types' -import type { TaskSourceHostAvailability } from '../task-source-context-summary' -import type { AutomationRowAction } from './automation-captured-owner' -import type { AutomationHostTarget } from './automation-host-client' -import { - getAutomationRowLastRunSnapshot, - getLocalAutomationLastRunSnapshot -} from './automation-list-last-run' -import { AutomationListLastRunCell } from './AutomationListLastRunCell' -import { formatAutomationDateTimeWithRelative } from './automation-page-parts' -import { getAutomationTargetAvailability } from './automation-target-availability' -import { getAgentLabel } from './automation-draft-model' import type { AutomationListRow } from './automation-list-row-identity' -import { - formatAutomationCost, - formatAutomationTokens, - type AutomationUsageSummary -} from './automation-usage-model' -import { AUTOMATIONS_TABLE_GRID_CLASS } from './automations-table-layout' -import { - LIST_TABLE_ROW_CLASS, - LIST_TABLE_ROW_SELECTED_CLASS, - LIST_TABLE_STICKY_ROW_CELL_CLASS -} from '@/lib/list-table-layout' -import { isPortaledRowMenuClick, isRowActivationKey } from '@/lib/list-row-interaction' -import { AutomationListStatusCell } from './AutomationListStatusCell' -import { translate } from '@/i18n/i18n' +import { AutomationListLocalRow, type AutomationListLocalRowProps } from './AutomationListLocalRow' -export type AutomationListLocalRowsProps = { +export type AutomationListLocalRowsProps = Omit & { rows: readonly AutomationListRow[] - selectedRowKey: string | null | undefined - isSelectedLocal: boolean - lastRunByAutomationId: ReadonlyMap - relativeNow: number - repoMap: ReadonlyMap - worktreeMap: ReadonlyMap - repoForRow?: (row: AutomationListRow) => Repo | undefined - worktreeForRow?: (row: AutomationListRow, repo: Repo | undefined) => Worktree | undefined - projectHostSetups: readonly ProjectHostSetup[] - sshConnectionStates: ReadonlyMap> - runtimeStatusByEnvironmentId: ReadonlyMap< - string, - { status: RuntimeStatus | null; checkedAt: number } - > - hostTargetFor: (row: AutomationListRow) => AutomationHostTarget | null - automationSourceHostAvailabilityByRowKey: ReadonlyMap - hostLabelById?: ReadonlyMap - isActionEnabled?: (row: AutomationListRow, action: AutomationRowAction) => boolean - onSelect: (rowKey: string) => void - onRunNow: (row: AutomationListRow) => void - onEdit: (row: AutomationListRow) => void - onToggle: (row: AutomationListRow) => void - onDelete: (row: AutomationListRow) => void -} - -const EMPTY_HOST_LABELS: ReadonlyMap = new Map() - -function automationUsageText(summary: AutomationUsageSummary | undefined): string { - if (!summary || summary.unavailableRuns > 0) { - return summary?.knownRuns - ? usageAmountText(summary) - : translate( - 'auto.components.automations.AutomationsPage.usageUnavailable', - 'Usage unavailable' - ) - } - return summary.knownRuns > 0 - ? usageAmountText(summary) - : translate('auto.components.automations.AutomationsPage.noRunUsageYet', 'No run usage yet') -} - -function usageAmountText(summary: AutomationUsageSummary): string { - return translate( - 'auto.components.automations.AutomationsPage.runUsageSummary', - '{{cost}} est. · {{tokens}} tokens', - { - cost: formatAutomationCost(summary.estimatedCostUsd), - tokens: formatAutomationTokens(summary.totalTokens) - } - ) } export function AutomationListLocalRows({ rows, - selectedRowKey, - isSelectedLocal, - lastRunByAutomationId, - relativeNow, - repoMap, - worktreeMap, - repoForRow, - worktreeForRow, - projectHostSetups, - sshConnectionStates, - runtimeStatusByEnvironmentId, - hostTargetFor, - automationSourceHostAvailabilityByRowKey, - hostLabelById = EMPTY_HOST_LABELS, - isActionEnabled, - onSelect, - onRunNow, - onEdit, - onToggle, - onDelete + ...rowProps }: AutomationListLocalRowsProps): React.JSX.Element { - const allows = (row: AutomationListRow, action: AutomationRowAction): boolean => - isActionEnabled?.(row, action) ?? true return ( <> - {rows.map((row) => { - const { automation } = row - const automationRepo = repoForRow?.(row) ?? repoMap.get(getAutomationRunRepoId(automation)) - const automationWorktree = automation.workspaceId - ? (worktreeForRow?.(row, automationRepo) ?? worktreeMap.get(automation.workspaceId)) - : null - const automationRunAvailability = getAutomationTargetAvailability({ - automation, - repo: automationRepo, - workspace: automationWorktree, - projectHostSetups, - sshConnectionStates, - runtimeStatusByEnvironmentId, - automationHostTarget: hostTargetFor(row), - sourceHostAvailability: automationSourceHostAvailabilityByRowKey.get(row.key) - }) - const projectLabel = - automationRepo?.displayName ?? - translate('auto.components.automations.AutomationsPage.13118faadf', 'Unknown project') - const scheduleLabel = formatUiAutomationSchedule(automation.rrule) - const nextRunLabel = automation.enabled - ? formatAutomationDateTimeWithRelative(automation.nextRunAt, relativeNow) - : translate('auto.components.automations.enablement.paused', 'Paused') - const isSelected = isSelectedLocal && selectedRowKey === row.key - const agentLabel = getAgentLabel(automation.agentId) - const hostId = - automation.runContext?.hostId ?? - (automationRepo ? getRepoExecutionHostId(automationRepo) : null) - const hostLabel = - row.hostLabel || - (hostId - ? (hostLabelById.get(hostId) ?? getExecutionHostLabel(hostId)) - : getLocalExecutionHostLabel()) - const agentTooltipLabel = `${agentLabel} · ${hostLabel} · ${automationUsageText(row.usageSummary ?? undefined)}` - const canRunNow = automationRunAvailability.canRunNow && allows(row, 'run') - const lastRun = lastRunByAutomationId.get(automation.id) - // Without a fetched run, the row's projected summary carries the newest - // retained run's status — the list never downloads run history for this. - const lastRunSnapshot = lastRun - ? getLocalAutomationLastRunSnapshot(automation, lastRun) - : getAutomationRowLastRunSnapshot(row) - - const actionItems = ( - <> - onRunNow(row)} - /> - } - label={translate('auto.components.automations.AutomationsPage.f4612e3f78', 'Edit')} - onSelect={() => onEdit(row)} - /> - : - } - label={ - automation.enabled - ? translate('auto.components.automations.AutomationsPage.b457436d6a', 'Pause') - : translate('auto.components.automations.AutomationsPage.376631ef2b', 'Resume') - } - onSelect={() => onToggle(row)} - /> - - } - label={translate('auto.components.automations.AutomationsPage.15e0bfb13b', 'Delete')} - variant="destructive" - onSelect={() => onDelete(row)} - /> - - ) - - return ( - - -
    { - // Why: Radix portals menus out of the row DOM, but React still - // bubbles those clicks here — ignore so menu actions don't open detail. - if (isPortaledRowMenuClick(event)) { - return - } - onSelect(row.key) - }} - onKeyDown={(event) => { - if (!isRowActivationKey(event)) { - return - } - event.preventDefault() - onSelect(row.key) - }} - className={cn( - AUTOMATIONS_TABLE_GRID_CLASS, - LIST_TABLE_ROW_CLASS, - isSelected && LIST_TABLE_ROW_SELECTED_CLASS - )} - > - - {automation.name} - - - {scheduleLabel} - - - {projectLabel} - - - {hostLabel} - - - {nextRunLabel} - - - - - - - - - - - {agentTooltipLabel} - - - - - - - - { - if (canRunNow) { - onRunNow(row) - } - }} - > - - - {automationRunAvailability.canRunNow - ? translate( - 'auto.components.automations.AutomationsPage.2faecab10b', - 'Run Now' - ) - : automationRunAvailability.message} - - - onEdit(row)}> - - {translate('auto.components.automations.AutomationsPage.f4612e3f78', 'Edit')} - - onToggle(row)} - > - {automation.enabled ? ( - - ) : ( - - )} - {automation.enabled - ? translate( - 'auto.components.automations.AutomationsPage.b457436d6a', - 'Pause' - ) - : translate( - 'auto.components.automations.AutomationsPage.376631ef2b', - 'Resume' - )} - - - onDelete(row)} - > - - {translate( - 'auto.components.automations.AutomationsPage.15e0bfb13b', - 'Delete' - )} - - - -
    -
    - {actionItems} -
    - ) - })} + {rows.map((row) => ( + + ))} ) } - -function MenuRunItem({ - disabled, - label, - onSelect -}: { - disabled: boolean - label: string - onSelect: () => void -}): React.JSX.Element { - return ( - { - if (disabled) { - event.preventDefault() - return - } - onSelect() - }} - > - - {label} - - ) -} - -function MenuItem({ - disabled, - icon, - label, - onSelect, - variant -}: { - disabled?: boolean - icon: React.ReactNode - label: string - onSelect: () => void - variant?: 'destructive' -}): React.JSX.Element { - return ( - - {icon} - {label} - - ) -} - -function MenuSeparator(): React.JSX.Element { - return -} diff --git a/src/renderer/src/components/automations/AutomationListSortHeader.tsx b/src/renderer/src/components/automations/AutomationListSortHeader.tsx new file mode 100644 index 00000000000..2c24a344328 --- /dev/null +++ b/src/renderer/src/components/automations/AutomationListSortHeader.tsx @@ -0,0 +1,51 @@ +import React from 'react' +import { ArrowDown, ArrowUp } from 'lucide-react' +import { cn } from '@/lib/utils' +import { translate } from '@/i18n/i18n' +import type { AutomationListSort, AutomationListSortField } from './automation-list-view' + +export function AutomationListSortHeader({ + field, + label, + sort, + onSort +}: { + field: AutomationListSortField + label: string + sort: AutomationListSort | null + onSort: (field: AutomationListSortField) => void +}): React.JSX.Element { + const active = sort?.field === field + const direction = active ? sort.direction : null + // Why: one interpolated key per direction — word order and punctuation around + // the column name differ per language. + const sortedLabel = + direction === 'asc' + ? translate( + 'auto.components.automations.AutomationListSortHeader.sortedAscending', + '{{value0}}, sorted ascending', + { value0: label } + ) + : direction === 'desc' + ? translate( + 'auto.components.automations.AutomationListSortHeader.sortedDescending', + '{{value0}}, sorted descending', + { value0: label } + ) + : null + return ( + + ) +} diff --git a/src/renderer/src/components/automations/AutomationListTableHeader.test.tsx b/src/renderer/src/components/automations/AutomationListTableHeader.test.tsx index 5c5bbe8e568..638e96a23bc 100644 --- a/src/renderer/src/components/automations/AutomationListTableHeader.test.tsx +++ b/src/renderer/src/components/automations/AutomationListTableHeader.test.tsx @@ -1,7 +1,8 @@ // @vitest-environment happy-dom import { cleanup, render, screen } from '@testing-library/react' -import { afterEach, describe, expect, it } from 'vitest' +import { afterEach, describe, expect, it, vi } from 'vitest' +import userEvent from '@testing-library/user-event' import { AutomationListTableHeader } from './AutomationListTableHeader' import { LIST_TABLE_HEADER_CLASS, @@ -43,3 +44,45 @@ describe('AutomationListTableHeader', () => { expect(nameCell.className).toBe(LIST_TABLE_STICKY_HEADER_CELL_CLASS) }) }) + +describe('AutomationListTableHeader sorting', () => { + afterEach(cleanup) + + it('exposes only the orderable columns as buttons', () => { + render( {}} />) + + expect(screen.getAllByRole('button').map((button) => button.textContent)).toEqual([ + 'Name', + 'Last run' + ]) + }) + + it('reports the sorted column and direction in the accessible name', () => { + const { rerender } = render( + {}} /> + ) + expect(screen.getByRole('button', { name: 'Name, sorted ascending' })).toBeDefined() + expect(screen.getByRole('button', { name: 'Last run' })).toBeDefined() + + rerender( + {}} /> + ) + expect(screen.getByRole('button', { name: 'Last run, sorted descending' })).toBeDefined() + expect(screen.getByRole('button', { name: 'Name' })).toBeDefined() + }) + + it('requests a sort for the clicked column', async () => { + const onSort = vi.fn() + render() + + await userEvent.click(screen.getByRole('button', { name: 'Last run' })) + + expect(onSort.mock.calls).toEqual([['lastRun']]) + }) + + it('stays non-interactive when the list cannot be sorted', () => { + render() + + expect(screen.queryAllByRole('button')).toEqual([]) + }) +}) diff --git a/src/renderer/src/components/automations/AutomationListTableHeader.tsx b/src/renderer/src/components/automations/AutomationListTableHeader.tsx index dcbd107fcbc..605baf8a945 100644 --- a/src/renderer/src/components/automations/AutomationListTableHeader.tsx +++ b/src/renderer/src/components/automations/AutomationListTableHeader.tsx @@ -5,34 +5,85 @@ import { LIST_TABLE_HEADER_CLASS, LIST_TABLE_STICKY_HEADER_CELL_CLASS } from '@/lib/list-table-layout' +import { AutomationListSortHeader } from './AutomationListSortHeader' +import type { AutomationListSort, AutomationListSortField } from './automation-list-view' -export function AutomationListTableHeader(): React.JSX.Element { - const labels = [ - ['auto.components.automations.AutomationsPage.tableName', 'Name'], - ['auto.components.automations.AutomationDetail.18763ded26', 'Schedule'], - ['auto.components.automations.AutomationsPage.tableProject', 'Project'], - ['auto.components.automations.AutomationsPage.tableHost', 'Host'], - ['auto.components.automations.AutomationDetail.578ff46987', 'Next run'], - ['auto.components.automations.AutomationsPage.tableLastRun', 'Last run'], - ['auto.components.automations.AutomationsPage.tableStatus', 'Status'], - ['auto.components.automations.AutomationDetail.2df8970cd5', 'Agent'] - ] as const +type HeaderColumn = { + key: string + fallback: string + /** Absent for columns the list cannot order by. */ + sortField?: AutomationListSortField +} + +const COLUMNS: readonly HeaderColumn[] = [ + { + key: 'auto.components.automations.AutomationsPage.tableName', + fallback: 'Name', + sortField: 'name' + }, + { + key: 'auto.components.automations.AutomationDetail.18763ded26', + fallback: 'Schedule' + }, + { + key: 'auto.components.automations.AutomationsPage.tableProject', + fallback: 'Project' + }, + { + key: 'auto.components.automations.AutomationsPage.tableHost', + fallback: 'Host' + }, + { + key: 'auto.components.automations.AutomationDetail.578ff46987', + fallback: 'Next run' + }, + { + key: 'auto.components.automations.AutomationsPage.tableLastRun', + fallback: 'Last run', + sortField: 'lastRun' + }, + { + key: 'auto.components.automations.AutomationsPage.tableStatus', + fallback: 'Status' + }, + { + key: 'auto.components.automations.AutomationDetail.2df8970cd5', + fallback: 'Agent' + } +] + +export function AutomationListTableHeader({ + sort = null, + onSort +}: { + sort?: AutomationListSort | null + onSort?: (field: AutomationListSortField) => void +} = {}): React.JSX.Element { return (
    - {labels.map(([key, fallback], index) => ( - - {translate(key, fallback)} - - ))} + {COLUMNS.map((column, index) => { + const label = translate(column.key, column.fallback) + const className = + index === 0 + ? LIST_TABLE_STICKY_HEADER_CELL_CLASS + : index === COLUMNS.length - 1 + ? 'text-center' + : undefined + return ( + + {column.sortField && onSort ? ( + + ) : ( + label + )} + + ) + })} {translate('auto.components.automations.AutomationsPage.tableActions', 'Actions')} diff --git a/src/renderer/src/components/automations/AutomationsListPanel.test.tsx b/src/renderer/src/components/automations/AutomationsListPanel.test.tsx index 2f772d2a7ed..f2362b83d0f 100644 --- a/src/renderer/src/components/automations/AutomationsListPanel.test.tsx +++ b/src/renderer/src/components/automations/AutomationsListPanel.test.tsx @@ -11,7 +11,12 @@ import { createRoot, type Root } from 'react-dom/client' import { afterEach, beforeEach, describe, expect, it } from 'vitest' import { TooltipProvider } from '@/components/ui/tooltip' import { AutomationsListPanel } from './AutomationsListPanel' -import { EMPTY_AUTOMATION_LIST_FILTER } from './automation-list-view' +import { + buildAutomationListViewItems, + EMPTY_AUTOMATION_LIST_FILTER, + type AutomationListSort, + type AutomationListSortField +} from './automation-list-view' import type { AutomationHostCatalogView } from './use-automation-host-catalog' import { makeAutomation, @@ -49,7 +54,13 @@ const HOST_CATALOG = { status: 'all', announceFallback: false }, - rows: { rows: [], automations: [], capturedOwners: new Map(), groups: [], answered: true }, + rows: { + rows: [], + automations: [], + capturedOwners: new Map(), + groups: [], + answered: true + }, loadCounts: { failedHostCount: 0, totalHostCount: 1 }, selectHost: () => undefined, recover: () => undefined, @@ -70,6 +81,8 @@ function renderPanel( selectExternalKey?: (key: string | null) => void externalEntries?: readonly ExternalAutomationListEntry[] setActivePaneTab?: (tab: AutomationPaneTab) => void + listSort?: AutomationListSort | null + onListSortChange?: (field: AutomationListSortField) => void } = {} ): void { const externalEntries = options.externalEntries ?? [] @@ -95,8 +108,12 @@ function renderPanel( externalManagersUncheckedNotice={uncheckedNotice} onSelectHost={() => undefined} onRecoverHost={() => undefined} - filteredRows={rows} - filteredExternalAutomationEntries={externalEntries} + sortedListItems={buildAutomationListViewItems({ + rows, + externalEntries + })} + listSort={options.listSort ?? null} + onListSortChange={options.onListSortChange ?? (() => undefined)} selectedRowKey={options.selectedRowKey ?? null} selectedExternalKey={options.selectedExternalKey ?? null} relativeNow={0} @@ -221,7 +238,11 @@ describe('AutomationsListPanel enter key navigation', () => { const input = searchField() expect(input).not.toBeNull() - const enter = new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }) + const enter = new KeyboardEvent('keydown', { + key: 'Enter', + bubbles: true, + cancelable: true + }) input?.dispatchEvent(enter) expect(enter.defaultPrevented).toBe(true) @@ -252,7 +273,11 @@ describe('AutomationsListPanel enter key navigation', () => { const input = searchField() expect(input).not.toBeNull() - const enter = new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }) + const enter = new KeyboardEvent('keydown', { + key: 'Enter', + bubbles: true, + cancelable: true + }) input?.dispatchEvent(enter) expect(enter.defaultPrevented).toBe(true) @@ -272,7 +297,11 @@ describe('AutomationsListPanel enter key navigation', () => { const input = searchField() expect(input).not.toBeNull() - const enter = new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }) + const enter = new KeyboardEvent('keydown', { + key: 'Enter', + bubbles: true, + cancelable: true + }) input?.dispatchEvent(enter) expect(detailOpened).toBe(false) diff --git a/src/renderer/src/components/automations/AutomationsListPanel.tsx b/src/renderer/src/components/automations/AutomationsListPanel.tsx index 5943096756a..783eee57da6 100644 --- a/src/renderer/src/components/automations/AutomationsListPanel.tsx +++ b/src/renderer/src/components/automations/AutomationsListPanel.tsx @@ -23,13 +23,19 @@ import { import type { AutomationListRow } from './automation-list-row-identity' import type { AutomationPaneTab } from './automation-page-state' import { AutomationListFilterPills } from './AutomationListFilterMenu' -import { isAutomationListFilterActive, type AutomationListFilter } from './automation-list-view' +import { + isAutomationListFilterActive, + type AutomationListFilter, + type AutomationListSort, + type AutomationListSortField, + type AutomationListViewItem +} from './automation-list-view' import { automationHostFilterStableKey } from '../../../../shared/automation-host-filter' import type { AutomationTemplate } from './automation-templates' import type { ExternalAutomationListEntry } from './external-automation-list-entries' import type { ExternalAutomationScope } from './external-automation-scope-client' -import { AutomationListLocalRows } from './AutomationListLocalRows' -import { AutomationListExternalRows } from './AutomationListExternalRows' +import { AutomationListLocalRow } from './AutomationListLocalRow' +import { AutomationListExternalRow } from './AutomationListExternalRow' import { AutomationHostFilterNotice, AutomationHostLoadSummary } from './AutomationHostFilterNotice' import { AutomationListEmptyView } from './AutomationListEmptyView' import { resolveAutomationListEmptyState } from './automation-list-empty-state' @@ -63,8 +69,10 @@ type AutomationsListPanelProps = { action: AutomationHostRecoveryAction, entry?: AutomationHostCatalogEntry | null ) => void - filteredRows: readonly AutomationListRow[] - filteredExternalAutomationEntries: readonly ExternalAutomationListEntry[] + /** Both collections as one list in render order; the sort spans local and external rows. */ + sortedListItems: readonly AutomationListViewItem[] + listSort: AutomationListSort | null + onListSortChange: (field: AutomationListSortField) => void selectedRowKey: string | null selectedExternalKey: string | null selectedExternal?: ExternalAutomationListEntry | null @@ -124,8 +132,9 @@ export function AutomationsListPanel(props: AutomationsListPanelProps): React.JS externalManagersUncheckedNotice, onSelectHost, onRecoverHost, - filteredRows, - filteredExternalAutomationEntries, + sortedListItems, + listSort, + onListSortChange, selectedRowKey, selectedExternalKey, relativeNow, @@ -161,18 +170,20 @@ export function AutomationsListPanel(props: AutomationsListPanelProps): React.JS // Hosts moved into the Filters menu, so its toolbar row is the focus fallback now. const toolbarRef = useRef(null) const pendingKeyboardScrollRef = useRef(false) - const rowKeys = React.useMemo(() => filteredRows.map((row) => row.key), [filteredRows]) - const visibleItems = React.useMemo( - () => [ - ...filteredRows.map((row) => ({ kind: 'local' as const, id: row.key })), - ...filteredExternalAutomationEntries.map((entry) => ({ - kind: 'external' as const, - id: entry.key - })) - ], - [filteredExternalAutomationEntries, filteredRows] + // Why: keyboard traversal and focus recovery read render order, which the sort owns. + const rowKeys = React.useMemo( + () => sortedListItems.filter((item) => item.kind === 'local').map((item) => item.id), + [sortedListItems] ) - useAutomationListFocusRecovery({ rowKeys, containerRef: listRef, fallbackRef: toolbarRef }) + const visibleItems = React.useMemo( + () => sortedListItems.map((item) => ({ kind: item.kind, id: item.id })), + [sortedListItems] + ) + useAutomationListFocusRecovery({ + rowKeys, + containerRef: listRef, + fallbackRef: toolbarRef + }) const handleSearchArrowNavigate = React.useCallback( (key: AutomationListArrowKey) => { const next = getAutomationListArrowNavigationTarget({ @@ -331,24 +342,30 @@ export function AutomationsListPanel(props: AutomationsListPanelProps): React.JS > {hasFilteredListItems ? (
    - +
    - - { - selectAutomationRow(null) - selectExternalKey(entryKey) - setActivePaneTab('overview') - onOpenDetail() - }} - onRequestAction={requestExternalAction} - onEdit={openEditExternalDialog} - /> + {sortedListItems.map((item) => + item.kind === 'local' ? ( + + ) : ( + { + selectAutomationRow(null) + selectExternalKey(entryKey) + setActivePaneTab('overview') + onOpenDetail() + }} + onRequestAction={requestExternalAction} + onEdit={openEditExternalDialog} + /> + ) + )}
    ) : ( diff --git a/src/renderer/src/components/automations/AutomationsPage.create-destination.test.tsx b/src/renderer/src/components/automations/AutomationsPage.create-destination.test.tsx index a0778029ef9..0635ecdb6ff 100644 --- a/src/renderer/src/components/automations/AutomationsPage.create-destination.test.tsx +++ b/src/renderer/src/components/automations/AutomationsPage.create-destination.test.tsx @@ -19,7 +19,6 @@ import { addRuntimeProject, api, installAutomationsPageHarness, - listedRow, mocks, renderPage, runtimeHost, @@ -30,6 +29,7 @@ import { scopedList, settleHostQueries } from './automations-page-test-harness' +import { listedRow } from './automations-page-listed-items' import { makeAutomation, REPO_ID, WORKSPACE_ID } from './automations-page-fixtures' import type { Repo } from '../../../../shared/repo-types' import type { ProjectHostSetup } from '../../../../shared/project-types' diff --git a/src/renderer/src/components/automations/AutomationsPage.cross-authority-actions.test.tsx b/src/renderer/src/components/automations/AutomationsPage.cross-authority-actions.test.tsx index 618c092b45a..8193502cb36 100644 --- a/src/renderer/src/components/automations/AutomationsPage.cross-authority-actions.test.tsx +++ b/src/renderer/src/components/automations/AutomationsPage.cross-authority-actions.test.tsx @@ -22,6 +22,7 @@ import { SELF_PRECONDITION, settleHostQueries } from './automations-page-test-harness' +import { listedRows } from './automations-page-listed-items' import { makeAutomation } from './automations-page-fixtures' installAutomationsPageHarness() @@ -36,9 +37,7 @@ async function collidingHosts(): Promise { } function selectDesktopRow(): string { - const row = mocks.listPanel?.filteredRows.find( - (candidate) => candidate.automation.name === 'Desktop nightly' - ) + const row = listedRows().find((candidate) => candidate.automation.name === 'Desktop nightly') expect(row).toBeDefined() return row?.key ?? '' } @@ -58,9 +57,7 @@ describe('AutomationsPage row actions under a colliding automation id', () => { await renderPage() await settleHostQueries() - const remote = mocks.listPanel?.filteredRows.find( - (candidate) => candidate.automation.name === 'Remote nightly' - ) + const remote = listedRows().find((candidate) => candidate.automation.name === 'Remote nightly') await act(async () => { mocks.listPanel?.selectAutomationRow(remote?.key ?? '') }) diff --git a/src/renderer/src/components/automations/AutomationsPage.external-scope.test.tsx b/src/renderer/src/components/automations/AutomationsPage.external-scope.test.tsx index cd966dcbcd7..b4ea3413cb3 100644 --- a/src/renderer/src/components/automations/AutomationsPage.external-scope.test.tsx +++ b/src/renderer/src/components/automations/AutomationsPage.external-scope.test.tsx @@ -20,6 +20,7 @@ import { RUNTIME_SELF_FILTER, settleHostQueries } from './automations-page-test-harness' +import { listedExternalEntries } from './automations-page-listed-items' import { makeExternalManager } from './automations-page-fixtures' installAutomationsPageHarness() @@ -117,7 +118,7 @@ describe('AutomationsPage external manager probes', () => { await renderPage() await settleHostQueries() - expect(mocks.listPanel?.filteredExternalAutomationEntries).toEqual([]) + expect(listedExternalEntries()).toEqual([]) }) it('drops the previous host rows when the selection moves, not when the new probe lands', async () => { @@ -127,7 +128,7 @@ describe('AutomationsPage external manager probes', () => { const { rerender } = await renderPage() await settleHostQueries() - expect(mocks.listPanel?.filteredExternalAutomationEntries.length).toBeGreaterThan(0) + expect(listedExternalEntries().length).toBeGreaterThan(0) // The new host never answers, so anything still listed belongs to the old one. api.automations.listExternalManagerForOwner.mockImplementation( @@ -137,7 +138,7 @@ describe('AutomationsPage external manager probes', () => { await rerender() await settleHostQueries() - expect(mocks.listPanel?.filteredExternalAutomationEntries).toEqual([]) + expect(listedExternalEntries()).toEqual([]) }) it('reports a host it could not check rather than showing it as clean', async () => { diff --git a/src/renderer/src/components/automations/AutomationsPage.notice-recovery.test.tsx b/src/renderer/src/components/automations/AutomationsPage.notice-recovery.test.tsx index d99cfb91dac..69c27640be1 100644 --- a/src/renderer/src/components/automations/AutomationsPage.notice-recovery.test.tsx +++ b/src/renderer/src/components/automations/AutomationsPage.notice-recovery.test.tsx @@ -15,7 +15,6 @@ import { addRuntimeProject, api, installAutomationsPageHarness, - listedRow, mocks, renderPage, runtimeHost, @@ -26,6 +25,7 @@ import { scopedList, settleHostQueries } from './automations-page-test-harness' +import { listedRow } from './automations-page-listed-items' import { makeAutomation } from './automations-page-fixtures' installAutomationsPageHarness() diff --git a/src/renderer/src/components/automations/AutomationsPage.refresh-selection.test.tsx b/src/renderer/src/components/automations/AutomationsPage.refresh-selection.test.tsx index 523cc50df7d..f33c4d09d6c 100644 --- a/src/renderer/src/components/automations/AutomationsPage.refresh-selection.test.tsx +++ b/src/renderer/src/components/automations/AutomationsPage.refresh-selection.test.tsx @@ -22,6 +22,7 @@ import { SELF_PRECONDITION, settleHostQueries } from './automations-page-test-harness' +import { listedRows } from './automations-page-listed-items' import { makeAutomation, makeRun } from './automations-page-fixtures' installAutomationsPageHarness() @@ -69,7 +70,7 @@ describe('AutomationsPage refresh', () => { await renderPage() - expect(mocks.listPanel?.filteredRows[0]?.usageSummary).toEqual(usageSummary) + expect(listedRows()[0]?.usageSummary).toEqual(usageSummary) }) it('does not re-list through the active runtime just because one is selected', async () => { @@ -231,9 +232,7 @@ describe('AutomationsPage multi-host selection', () => { ) ).toEqual(['Desktop nightly', 'Remote nightly']) - const remote = mocks.listPanel?.filteredRows.find( - (row) => row.automation.name === 'Remote nightly' - ) + const remote = listedRows().find((row) => row.automation.name === 'Remote nightly') await act(async () => { mocks.listPanel?.selectAutomationRow(remote?.key ?? '') }) diff --git a/src/renderer/src/components/automations/AutomationsPage.run-visibility.test.tsx b/src/renderer/src/components/automations/AutomationsPage.run-visibility.test.tsx index f7ad0be65a7..5e605a4487a 100644 --- a/src/renderer/src/components/automations/AutomationsPage.run-visibility.test.tsx +++ b/src/renderer/src/components/automations/AutomationsPage.run-visibility.test.tsx @@ -16,12 +16,12 @@ import type { Automation } from '../../../../shared/automations-types' import { api, installAutomationsPageHarness, - listedRow, mocks, renderPage, scopedList, settleHostQueries } from './automations-page-test-harness' +import { listedRow, listedRows } from './automations-page-listed-items' import { makeAutomation } from './automations-page-fixtures' installAutomationsPageHarness() @@ -42,7 +42,7 @@ function desktopStoreHolds(automations: Automation[]): void { /** The next-run column reads this; the mocked list panel renders only names. */ function listedNextRunAt(): number | null | undefined { - return mocks.listPanel?.filteredRows[0]?.automation.nextRunAt + return listedRows()[0]?.automation.nextRunAt } describe('AutomationsPage run visibility', () => { diff --git a/src/renderer/src/components/automations/AutomationsPage.test.tsx b/src/renderer/src/components/automations/AutomationsPage.test.tsx index 768d3250d80..a62a433a4d1 100644 --- a/src/renderer/src/components/automations/AutomationsPage.test.tsx +++ b/src/renderer/src/components/automations/AutomationsPage.test.tsx @@ -24,13 +24,13 @@ import { api, DESKTOP_SELF_OWNER, installAutomationsPageHarness, - listedRow, mocks, renderPage, rows, scopedList, SELF_PRECONDITION } from './automations-page-test-harness' +import { listedRow, listedExternalEntries } from './automations-page-listed-items' import { makeAutomation, makeExternalManager, @@ -147,7 +147,7 @@ describe('AutomationsPage list rendering', () => { api.automations.updateExternalForOwner.mockResolvedValue(undefined) await renderPage() - const entry = mocks.listPanel?.filteredExternalAutomationEntries[0] + const entry = listedExternalEntries()[0] if (!entry) { throw new Error('no external entry to edit') } @@ -177,7 +177,7 @@ describe('AutomationsPage list rendering', () => { api.automations.runExternalActionForOwner.mockResolvedValue(undefined) await renderPage() - const entry = mocks.listPanel?.filteredExternalAutomationEntries[0] + const entry = listedExternalEntries()[0] if (!entry) { throw new Error('no external entry to act on') } @@ -217,7 +217,7 @@ describe('AutomationsPage list rendering', () => { api.automations.listExternalRunsForOwner.mockResolvedValue({ runs: [], total: 0 }) const { container } = await renderPage() - const entry = mocks.listPanel?.filteredExternalAutomationEntries[0] + const entry = listedExternalEntries()[0] if (!entry) { throw new Error('no external entry to read runs for') } diff --git a/src/renderer/src/components/automations/AutomationsPageListPanel.tsx b/src/renderer/src/components/automations/AutomationsPageListPanel.tsx index 25c7ff7b88c..7adea56c608 100644 --- a/src/renderer/src/components/automations/AutomationsPageListPanel.tsx +++ b/src/renderer/src/components/automations/AutomationsPageListPanel.tsx @@ -1,6 +1,7 @@ import React from 'react' import type { AutomationsPageController } from './use-automations-page-controller' import { AutomationsListPanel } from './AutomationsListPanel' +import { nextAutomationListSort } from './automation-list-view' export function AutomationsPageListPanel({ controller, @@ -45,8 +46,6 @@ export function AutomationsPageListPanel({ hasListItems, hasFilteredListItems, isListSearchQueryTooLarge, - filteredRows, - filteredExternalAutomationEntries, selectedRow, selectedExternal, searchCounts @@ -79,8 +78,9 @@ export function AutomationsPageListPanel({ void pageRefresh.refresh() } }} - filteredRows={filteredRows} - filteredExternalAutomationEntries={filteredExternalAutomationEntries} + sortedListItems={list.sortedListItems} + listSort={local.listSort} + onListSortChange={(field) => local.setListSort(nextAutomationListSort(local.listSort, field))} selectedRowKey={selectedRow?.key ?? null} selectedExternalKey={local.selectedExternalKey} selectedExternal={selectedExternal} diff --git a/src/renderer/src/components/automations/automation-list-view-sort.test.ts b/src/renderer/src/components/automations/automation-list-view-sort.test.ts index d3dfbe63133..8fbef8a0590 100644 --- a/src/renderer/src/components/automations/automation-list-view-sort.test.ts +++ b/src/renderer/src/components/automations/automation-list-view-sort.test.ts @@ -5,32 +5,34 @@ import { type AutomationListSort, type AutomationListViewItem } from './automation-list-view' +import { unscopedAutomationListRows } from './automation-list-row-identity' import { makeAutomation } from './automations-page-fixtures' -const locale = vi.hoisted(() => ({ value: 'en' })) -vi.mock('@/i18n/i18n', () => ({ getIntlLocale: () => locale.value })) - afterEach(() => { vi.restoreAllMocks() - locale.value = 'en' }) -function rows(count = 512): AutomationListViewItem[] { +function items(count = 512): AutomationListViewItem[] { const names = ['Alpha', 'álpha', 'Ångström', 'Zebra', 'Örebro', 'I', 'ı', 'İ', 'job 10', 'job 2'] return buildAutomationListViewItems({ - automations: Array.from({ length: count }, (_, index) => - makeAutomation({ id: `job-${index}`, name: names[(index * 7) % names.length] }) + rows: unscopedAutomationListRows( + Array.from({ length: count }, (_, index) => + makeAutomation({ + id: `job-${index}`, + name: names[(index * 7) % names.length] + }) + ) ), - externalEntries: [], - runs: [] + externalEntries: [] }) } -function previousOrder(items: AutomationListViewItem[], sort: AutomationListSort) { +/** The pre-collator comparator, resolving options on every comparison. */ +function previousOrder(list: AutomationListViewItem[], sort: AutomationListSort, locale: string) { function compare(left: AutomationListViewItem, right: AutomationListViewItem) { const value = sort.field === 'name' - ? left.name.localeCompare(right.name, locale.value, { sensitivity: 'base' }) + ? left.name.localeCompare(right.name, locale, { sensitivity: 'base' }) : (left.lastRunAt ?? 0) - (right.lastRunAt ?? 0) return value !== 0 ? sort.direction === 'asc' @@ -38,37 +40,35 @@ function previousOrder(items: AutomationListViewItem[], sort: AutomationListSort : -value : left.id.localeCompare(right.id) } - return [...items].sort(compare) + return [...list].sort(compare) } describe('automation list collation', () => { it.each(['en', 'sv', 'tr', 'ja'])( 'preserves %s ordering, tie-breaks and input identity', - (language) => { - locale.value = language - const items = rows() - const original = [...items] + (locale) => { + const list = items() + const original = [...list] for (const direction of ['asc', 'desc'] as const) { const sort = { field: 'name', direction } as const - const expected = previousOrder(items, sort) - const result = sortAutomationListViewItems(items, sort) + const expected = previousOrder(list, sort, locale) + const result = sortAutomationListViewItems(list, sort, locale) expect(result).toEqual(expected) expect(result.every((row, index) => row === expected[index])).toBe(true) } - expect(items).toEqual(original) + expect(list).toEqual(original) } ) - it('resolves collation once per name sort and responds to locale changes', () => { - const items = rows() + it('resolves collation once per name sort and follows the locale it is given', () => { + const list = items() const OriginalCollator = Intl.Collator const construct = vi.spyOn(Intl, 'Collator').mockImplementation(function (locales, options) { return new OriginalCollator(locales, options) }) const compare = vi.spyOn(String.prototype, 'localeCompare') - sortAutomationListViewItems(items, { field: 'name', direction: 'asc' }) - locale.value = 'sv' - sortAutomationListViewItems(items, { field: 'name', direction: 'desc' }) + sortAutomationListViewItems(list, { field: 'name', direction: 'asc' }, 'en') + sortAutomationListViewItems(list, { field: 'name', direction: 'desc' }, 'sv') expect(construct.mock.calls).toEqual([ ['en', { sensitivity: 'base' }], ['sv', { sensitivity: 'base' }] @@ -76,16 +76,39 @@ describe('automation list collation', () => { expect(compare.mock.calls.filter((args) => args.length >= 3)).toHaveLength(0) }) + it('orders by row key, not the bare automation ID, so hosts cannot collapse', () => { + const duplicate = makeAutomation({ id: 'shared', name: 'Same' }) + const list = buildAutomationListViewItems({ + rows: [ + { + key: 'row|host-b|shared', + automation: duplicate, + hostLabel: 'b', + usageSummary: null + }, + { + key: 'row|host-a|shared', + automation: duplicate, + hostLabel: 'a', + usageSummary: null + } + ], + externalEntries: [] + }) + const sorted = sortAutomationListViewItems(list, { field: 'name', direction: 'asc' }, 'en') + expect(sorted.map((item) => item.id)).toEqual(['row|host-a|shared', 'row|host-b|shared']) + }) + it('does not construct collation for unsorted, time-sorted or trivial lists', () => { - const items = rows() + const list = items() const construct = vi.spyOn(Intl, 'Collator') - expect(sortAutomationListViewItems(items, null)).toEqual(items) + expect(sortAutomationListViewItems(list, null, 'en')).toEqual(list) const sort = { field: 'lastRun', direction: 'desc' } as const - expect(sortAutomationListViewItems(items, sort)).toEqual(previousOrder(items, sort)) - expect(sortAutomationListViewItems([], { field: 'name', direction: 'asc' })).toEqual([]) + expect(sortAutomationListViewItems(list, sort, 'en')).toEqual(previousOrder(list, sort, 'en')) + expect(sortAutomationListViewItems([], { field: 'name', direction: 'asc' }, 'en')).toEqual([]) expect( - sortAutomationListViewItems(items.slice(0, 1), { field: 'name', direction: 'asc' }) - ).toEqual(items.slice(0, 1)) + sortAutomationListViewItems(list.slice(0, 1), { field: 'name', direction: 'asc' }, 'en') + ).toEqual(list.slice(0, 1)) expect(construct).not.toHaveBeenCalled() }) }) diff --git a/src/renderer/src/components/automations/automation-list-view.test.ts b/src/renderer/src/components/automations/automation-list-view.test.ts index 188c94f3db7..169fbdd360a 100644 --- a/src/renderer/src/components/automations/automation-list-view.test.ts +++ b/src/renderer/src/components/automations/automation-list-view.test.ts @@ -1,7 +1,6 @@ import { describe, expect, it } from 'vitest' import type { Automation, - AutomationRun, AutomationRunStatus, ExternalAutomationJob, ExternalAutomationManager @@ -48,31 +47,6 @@ function makeAutomation(overrides: Partial = {}): Automation { } } -function makeRun(overrides: Partial = {}): AutomationRun { - return { - id: 'run-1', - automationId: 'automation-1', - title: 'Zebra job', - scheduledFor: 10, - status: 'completed', - trigger: 'scheduled', - workspaceId: 'worktree-1', - sessionKind: 'terminal', - chatSessionId: null, - terminalSessionId: null, - terminalPaneKey: null, - terminalPtyId: null, - outputSnapshot: null, - precheckResult: null, - usage: null, - error: null, - startedAt: 20, - dispatchedAt: 30, - createdAt: 10, - ...overrides - } -} - function makeExternalEntry( overrides: Partial = {} ): ExternalAutomationListEntry { @@ -118,22 +92,69 @@ function makeExternalEntry( } } +/** A catalog row with an optional projected last-run status, keyed like a real host row. */ +function makeCatalogRow( + id: string, + overrides: Partial = {}, + lastRunStatus?: AutomationRunStatus +): AutomationListRow { + return { + key: `row|host|${id}`, + automation: makeAutomation({ id, ...overrides }), + hostLabel: 'This computer', + usageSummary: lastRunStatus + ? { + knownRuns: 1, + unavailableRuns: 0, + inputTokens: 0, + outputTokens: 0, + cacheTokens: 0, + reasoningOutputTokens: 0, + totalTokens: 0, + estimatedCostUsd: null, + lastRunStatus, + lastRunAt: 111 + } + : null + } +} + +const rowKey = (id: string): string => `row|host|${id}` + describe('automation-list-view', () => { it('counts and detects active filters', () => { - expect(isAutomationListFilterActive({ status: 'all', lastRun: 'all', agentIds: [] })).toBe( - false - ) - expect(isAutomationListFilterActive({ status: 'paused', lastRun: 'all', agentIds: [] })).toBe( - true - ) - expect(countAutomationListFilters({ status: 'paused', lastRun: 'failed', agentIds: [] })).toBe( - 2 - ) + expect( + isAutomationListFilterActive({ + status: 'all', + lastRun: 'all', + agentIds: [] + }) + ).toBe(false) + expect( + isAutomationListFilterActive({ + status: 'paused', + lastRun: 'all', + agentIds: [] + }) + ).toBe(true) + expect( + countAutomationListFilters({ + status: 'paused', + lastRun: 'failed', + agentIds: [] + }) + ).toBe(2) }) it('toggles sort direction and defaults last run to newest first', () => { - expect(nextAutomationListSort(null, 'name')).toEqual({ field: 'name', direction: 'asc' }) - expect(nextAutomationListSort(null, 'lastRun')).toEqual({ field: 'lastRun', direction: 'desc' }) + expect(nextAutomationListSort(null, 'name')).toEqual({ + field: 'name', + direction: 'asc' + }) + expect(nextAutomationListSort(null, 'lastRun')).toEqual({ + field: 'lastRun', + direction: 'desc' + }) expect(nextAutomationListSort({ field: 'name', direction: 'asc' }, 'name')).toEqual({ field: 'name', direction: 'desc' @@ -146,82 +167,62 @@ describe('automation-list-view', () => { it('filters by enabled state and last-run outcome', () => { const items = applyAutomationListView({ - automations: [ - makeAutomation({ id: 'paused', name: 'Paused', enabled: false }), - makeAutomation({ id: 'ok', name: 'Healthy' }) + rows: [ + makeCatalogRow('paused', { name: 'Paused', enabled: false }, 'completed'), + makeCatalogRow('ok', { name: 'Healthy' }, 'dispatch_failed') ], externalEntries: [makeExternalEntry()], - runs: [ - makeRun({ automationId: 'paused', status: 'completed' }), - makeRun({ automationId: 'ok', status: 'dispatch_failed' }) - ], filter: { status: 'enabled', lastRun: 'failed', agentIds: [] }, - sort: null + sort: null, + locale: 'en' }) - expect(items.map((item) => item.id)).toEqual(['ok', 'manager-1:job-1']) + expect(items.map((item) => item.id)).toEqual([rowKey('ok'), 'manager-1:job-1']) }) it('filters local rows by multiple agents and leaves external rows out of agent scopes', () => { const items = applyAutomationListView({ - automations: [ - makeAutomation({ id: 'codex-job', agentId: 'codex' }), - makeAutomation({ id: 'claude-job', agentId: 'claude' }) + rows: [ + makeCatalogRow('codex-job', { agentId: 'codex' }), + makeCatalogRow('claude-job', { agentId: 'claude' }) ], externalEntries: [makeExternalEntry()], - runs: [], filter: { status: 'all', lastRun: 'all', agentIds: ['codex', 'claude'] }, - sort: null + sort: null, + locale: 'en' }) - expect(items.map((item) => item.id)).toEqual(['codex-job', 'claude-job']) + expect(items.map((item) => item.id)).toEqual([rowKey('codex-job'), rowKey('claude-job')]) }) it('counts an agent filter alongside status and last-run filters', () => { - expect(isAutomationListFilterActive({ status: 'all', lastRun: 'all', agentIds: [] })).toBe( - false - ) expect( - countAutomationListFilters({ status: 'paused', lastRun: 'failed', agentIds: ['codex'] }) + isAutomationListFilterActive({ + status: 'all', + lastRun: 'all', + agentIds: [] + }) + ).toBe(false) + expect( + countAutomationListFilters({ + status: 'paused', + lastRun: 'failed', + agentIds: ['codex'] + }) ).toBe(3) }) it('sorts by name across local and external rows', () => { const items = applyAutomationListView({ - automations: [makeAutomation({ name: 'Zebra job' })], + rows: [makeCatalogRow('zebra', { name: 'Zebra job' })], externalEntries: [makeExternalEntry({ name: 'Alpha digest' })], - runs: [], filter: { status: 'all', lastRun: 'all', agentIds: [] }, - sort: { field: 'name', direction: 'asc' } + sort: { field: 'name', direction: 'asc' }, + locale: 'en' }) expect(items.map((item) => item.name)).toEqual(['Alpha digest', 'Zebra job']) }) it('filters catalog rows by status, agent, and the projected last-run status', () => { - function makeCatalogRow( - id: string, - overrides: Partial, - lastRunStatus?: AutomationRunStatus - ): AutomationListRow { - return { - key: `row|host|${id}`, - automation: makeAutomation({ id, ...overrides }), - hostLabel: 'This computer', - usageSummary: lastRunStatus - ? { - knownRuns: 1, - unavailableRuns: 0, - inputTokens: 0, - outputTokens: 0, - cacheTokens: 0, - reasoningOutputTokens: 0, - totalTokens: 0, - estimatedCostUsd: null, - lastRunStatus, - lastRunAt: 111 - } - : null - } - } const rows = [ makeCatalogRow('paused-codex', { enabled: false, agentId: 'codex' }), makeCatalogRow('failed-claude', { agentId: 'claude' }, 'dispatch_failed'), @@ -229,9 +230,10 @@ describe('automation-list-view', () => { makeCatalogRow('never-codex', { agentId: 'codex' }) ] const ids = (filter: Partial) => - filterAutomationListRows(rows, { ...EMPTY_AUTOMATION_LIST_FILTER, ...filter }).map( - (row) => row.automation.id - ) + filterAutomationListRows(rows, { + ...EMPTY_AUTOMATION_LIST_FILTER, + ...filter + }).map((row) => row.automation.id) expect(ids({ status: 'paused' })).toEqual(['paused-codex']) expect(ids({ agentIds: ['claude'] })).toEqual(['failed-claude']) @@ -249,7 +251,10 @@ describe('automation-list-view', () => { catalogRef: targetId === null ? null - : { authority: { kind: 'desktop' }, selector: { kind: 'ssh', targetId } }, + : { + authority: { kind: 'desktop' }, + selector: { kind: 'ssh', targetId } + }, hostLabel: targetId ?? '', usageSummary: null }) @@ -257,9 +262,10 @@ describe('automation-list-view', () => { const keyOf = (row: AutomationListRow): string => row.catalogRef ? hostStableKey(row.catalogRef) : '' const ids = (hostStableKeys: readonly string[]) => - filterAutomationListRows(rows, { ...EMPTY_AUTOMATION_LIST_FILTER, hostStableKeys }).map( - (row) => row.automation.id - ) + filterAutomationListRows(rows, { + ...EMPTY_AUTOMATION_LIST_FILTER, + hostStableKeys + }).map((row) => row.automation.id) // Multi-select is any-of; a pre-catalog row names no host and is excluded. expect(ids([keyOf(rows[0]), keyOf(rows[1])])).toEqual(['on-a', 'on-b']) @@ -290,15 +296,22 @@ describe('automation-list-view', () => { it('sorts by last run newest first and keeps never-run rows last', () => { const items = applyAutomationListView({ - automations: [ - makeAutomation({ id: 'old', name: 'Old' }), - makeAutomation({ id: 'never', name: 'Never' }) + rows: [ + makeCatalogRow('old', { + name: 'Old', + lastRunAt: Date.parse('2026-08-11T09:00:00Z') + }), + makeCatalogRow('never', { name: 'Never' }) ], externalEntries: [makeExternalEntry({ lastRunAt: '2026-08-12T09:00:00Z' })], - runs: [makeRun({ automationId: 'old', dispatchedAt: Date.parse('2026-08-11T09:00:00Z') })], filter: { status: 'all', lastRun: 'all', agentIds: [] }, - sort: { field: 'lastRun', direction: 'desc' } + sort: { field: 'lastRun', direction: 'desc' }, + locale: 'en' }) - expect(items.map((item) => item.id)).toEqual(['manager-1:job-1', 'old', 'never']) + expect(items.map((item) => item.id)).toEqual([ + 'manager-1:job-1', + rowKey('old'), + rowKey('never') + ]) }) }) diff --git a/src/renderer/src/components/automations/automation-list-view.ts b/src/renderer/src/components/automations/automation-list-view.ts index 459d4b0f588..cedd394ed23 100644 --- a/src/renderer/src/components/automations/automation-list-view.ts +++ b/src/renderer/src/components/automations/automation-list-view.ts @@ -1,5 +1,3 @@ -import { getIntlLocale } from '@/i18n/i18n' -import type { Automation, AutomationRun } from '../../../../shared/automations-types' import type { TuiAgent } from '../../../../shared/tui-agent' import { hostStableKey } from '../../../../shared/automation-owner-key' import type { AutomationListRow } from './automation-list-row-identity' @@ -7,8 +5,6 @@ import type { ExternalAutomationListEntry } from './external-automation-list-ent import { getAutomationRowLastRunSnapshot, getExternalAutomationLastRunSnapshot, - getLocalAutomationLastRunSnapshot, - indexLatestAutomationRuns, type AutomationLastRunSnapshot } from './automation-list-last-run' @@ -22,6 +18,13 @@ export type AutomationListSort = { direction: AutomationListSortDirection } +/** + * A row and an external job flattened to what the shared list renders and sorts. + * + * `id` is the row's own key, never the bare automation ID: under All hosts two + * authorities can return the same ID, and the sort tie-break decides render + * order, so a bare ID would collapse them. See `automation-list-row-identity`. + */ export type AutomationListViewItem = | { kind: 'local' @@ -31,7 +34,7 @@ export type AutomationListViewItem = lastRunAt: number | null lastRun: AutomationLastRunSnapshot agentId: TuiAgent - automation: Automation + row: AutomationListRow } | { kind: 'external' @@ -117,30 +120,26 @@ function matchesLastRunFilter( return snapshot.tone === filter } +/** Flattens the two rendered collections into one sortable list, preserving row identity. */ export function buildAutomationListViewItems({ - automations, - externalEntries, - runs + rows, + externalEntries }: { - automations: readonly Automation[] + rows: readonly AutomationListRow[] externalEntries: readonly ExternalAutomationListEntry[] - runs: readonly AutomationRun[] }): AutomationListViewItem[] { - const lastRunByAutomationId = indexLatestAutomationRuns(runs) - const locals: AutomationListViewItem[] = automations.map((automation) => { - const lastRun = getLocalAutomationLastRunSnapshot( - automation, - lastRunByAutomationId.get(automation.id) - ) + const locals: AutomationListViewItem[] = rows.map((row) => { + // Why: the same snapshot the row cell renders, so the sort matches the column. + const lastRun = getAutomationRowLastRunSnapshot(row) return { kind: 'local', - id: automation.id, - name: automation.name, - enabled: automation.enabled, + id: row.key, + name: row.automation.name, + enabled: row.automation.enabled, lastRunAt: lastRun.at, lastRun, - agentId: automation.agentId, - automation + agentId: row.automation.agentId, + row } }) const externals: AutomationListViewItem[] = externalEntries.map((entry) => { @@ -217,34 +216,21 @@ export function filterExternalAutomationListEntries( ) } -export function filterAutomationListViewItems( - items: readonly AutomationListViewItem[], - filter: AutomationListFilter -): AutomationListViewItem[] { - if (!isAutomationListFilterActive(filter)) { - return [...items] - } - return items.filter( - (item) => - matchesStatusFilter(item.enabled, filter.status) && - matchesLastRunFilter(item.lastRun, filter.lastRun) && - (filter.agentIds.length === 0 || - (item.agentId !== null && filter.agentIds.includes(item.agentId))) - ) -} - +/** + * `locale` is a parameter, not a `getIntlLocale()` read, so callers memoizing this + * can declare it — a hidden read is invisible to a dependency array. + */ export function sortAutomationListViewItems( items: readonly AutomationListViewItem[], - sort: AutomationListSort | null + sort: AutomationListSort | null, + locale: string ): AutomationListViewItem[] { if (!sort || items.length < 2) { return [...items] } const next = [...items] const compareNames = - sort.field === 'name' - ? new Intl.Collator(getIntlLocale(), { sensitivity: 'base' }).compare - : null + sort.field === 'name' ? new Intl.Collator(locale, { sensitivity: 'base' }).compare : null next.sort((left, right) => { const compared = compareNames ? compareNames(left.name, right.name) @@ -257,24 +243,26 @@ export function sortAutomationListViewItems( return next } +/** The rendered list: filter each collection with its own rules, then sort as one. */ export function applyAutomationListView({ - automations, + rows, externalEntries, - runs, filter, - sort + sort, + locale }: { - automations: readonly Automation[] + rows: readonly AutomationListRow[] externalEntries: readonly ExternalAutomationListEntry[] - runs: readonly AutomationRun[] filter: AutomationListFilter sort: AutomationListSort | null + locale: string }): AutomationListViewItem[] { return sortAutomationListViewItems( - filterAutomationListViewItems( - buildAutomationListViewItems({ automations, externalEntries, runs }), - filter - ), - sort + buildAutomationListViewItems({ + rows: filterAutomationListRows(rows, filter), + externalEntries: filterExternalAutomationListEntries(externalEntries, filter) + }), + sort, + locale ) } diff --git a/src/renderer/src/components/automations/automations-page-listed-items.ts b/src/renderer/src/components/automations/automations-page-listed-items.ts new file mode 100644 index 00000000000..d87ae62b4a8 --- /dev/null +++ b/src/renderer/src/components/automations/automations-page-listed-items.ts @@ -0,0 +1,32 @@ +/** + * What the page actually listed, read back from the mocked list panel. + * + * Tests act through the same authority-qualified keys and render order the + * user's click carries, rather than synthesizing either. + */ + +import type { AutomationListRow } from './automation-list-row-identity' +import type { ExternalAutomationListEntry } from './external-automation-list-entries' +import { mocks } from './automations-page-test-harness' + +function listedItems() { + return mocks.listPanel?.sortedListItems ?? [] +} + +/** Local rows the page listed, in render order. */ +export function listedRows(): readonly AutomationListRow[] { + return listedItems().flatMap((item) => (item.kind === 'local' ? [item.row] : [])) +} + +/** External entries the page listed, in render order. */ +export function listedExternalEntries(): readonly ExternalAutomationListEntry[] { + return listedItems().flatMap((item) => (item.kind === 'external' ? [item.entry] : [])) +} + +export function listedRow(automationId: string): AutomationListRow { + const row = listedRows().find((entry) => entry.automation.id === automationId) + if (!row) { + throw new Error(`no listed row for ${automationId}`) + } + return row +} diff --git a/src/renderer/src/components/automations/automations-page-test-harness.tsx b/src/renderer/src/components/automations/automations-page-test-harness.tsx index d9fd1088bf7..e34ae0640bc 100644 --- a/src/renderer/src/components/automations/automations-page-test-harness.tsx +++ b/src/renderer/src/components/automations/automations-page-test-harness.tsx @@ -27,6 +27,7 @@ import type { AutomationHostCatalogView } from './use-automation-host-catalog' import type { AutomationCreateDestinationControl } from './use-automation-create-destination' import type { ExternalAutomationListEntry } from './external-automation-list-entries' import type { AutomationListRow } from './automation-list-row-identity' +import type { AutomationListViewItem } from './automation-list-view' import { resetAutomationCapabilityProbes } from './automation-scoped-list-client' import { addRuntimeProject as addRuntimeProjectFixture, @@ -39,7 +40,7 @@ export const RUNTIME_REPO_ID = RUNTIME_REPO_ID_FIXTURE export const RUNTIME_WORKSPACE_ID = RUNTIME_WORKSPACE_ID_FIXTURE export type ListPanelProps = { - filteredExternalAutomationEntries: ExternalAutomationListEntry[] + sortedListItems: readonly AutomationListViewItem[] selectedExternal: ExternalAutomationListEntry | null openEditExternalDialog: ( manager: ExternalAutomationListEntry['manager'], @@ -55,7 +56,6 @@ export type ListPanelProps = { ) => void hasListItems: boolean hasFilteredListItems: boolean - filteredRows: readonly AutomationListRow[] selectedRowKey: string | null selectedExternalKey: string | null hostCatalog: AutomationHostCatalogView @@ -211,30 +211,31 @@ vi.mock('./AutomationsListPanel', () => ({ return (
    - ))} - {props.filteredExternalAutomationEntries.map((entry) => ( - - ))} + {props.sortedListItems.map((item) => + item.kind === 'local' ? ( + + ) : ( + + ) + )} {props.hasListItems ? null :
    }
    ) @@ -407,18 +408,6 @@ export async function refreshOnFocus(): Promise { }) } -/** - * The row the page actually listed for an ID, so tests act through the same - * authority-qualified key the user's click carries rather than a synthesized one. - */ -export function listedRow(automationId: string): AutomationListRow { - const row = mocks.listPanel?.filteredRows.find((entry) => entry.automation.id === automationId) - if (!row) { - throw new Error(`no listed row for ${automationId}`) - } - return row -} - export function rows(container: HTMLElement, testId: string): string[] { return [...container.querySelectorAll(`[data-testid="${testId}"]`)].map( (node) => node.textContent ?? '' diff --git a/src/renderer/src/components/automations/use-automations-page-list-state.ts b/src/renderer/src/components/automations/use-automations-page-list-state.ts index 65cce90ffc2..b13c8a689ec 100644 --- a/src/renderer/src/components/automations/use-automations-page-list-state.ts +++ b/src/renderer/src/components/automations/use-automations-page-list-state.ts @@ -4,9 +4,12 @@ import { buildExternalAutomationListEntries } from './external-automation-list-e import { externalAutomationScopeEntries } from './external-automation-scope-gating' import { externalAutomationUncheckedNotice } from './external-automation-unchecked-hosts' import { + buildAutomationListViewItems, filterAutomationListRows, - filterExternalAutomationListEntries + filterExternalAutomationListEntries, + sortAutomationListViewItems } from './automation-list-view' +import { getIntlLocale } from '@/i18n/i18n' import { unscopedAutomationListRows } from './automation-list-row-identity' import { useAutomationHostCatalog } from './use-automation-host-catalog' import { useAutomationListSearch } from './use-automation-list-search' @@ -28,6 +31,7 @@ export function useAutomationsPageListState({ failedAuthorityKeys, listSearchQuery, listFilter, + listSort, selectedRowKey, selectedExternalKey, selectedAutomationRuns, @@ -129,6 +133,21 @@ export function useAutomationsPageListState({ () => externalAutomationUncheckedNotice(scopedExternal.failures, hostCatalog.entries), [hostCatalog.entries, scopedExternal.failures] ) + // Why: a language switch changes collation without touching rows, so the locale + // has to reach the memo as a value. + const sortLocale = getIntlLocale() + const sortedListItems = useMemo( + () => + sortAutomationListViewItems( + buildAutomationListViewItems({ + rows: filteredRows, + externalEntries: filteredExternalAutomationEntries + }), + listSort, + sortLocale + ), + [filteredExternalAutomationEntries, filteredRows, listSort, sortLocale] + ) return { hostCatalog, @@ -146,6 +165,7 @@ export function useAutomationsPageListState({ isListSearchQueryTooLarge, filteredRows, filteredExternalAutomationEntries, + sortedListItems, hasListItems, hasFilteredListItems, searchCounts, diff --git a/src/renderer/src/components/automations/use-automations-page-local-state.ts b/src/renderer/src/components/automations/use-automations-page-local-state.ts index e92f666cb22..7a097b144c3 100644 --- a/src/renderer/src/components/automations/use-automations-page-local-state.ts +++ b/src/renderer/src/components/automations/use-automations-page-local-state.ts @@ -12,7 +12,11 @@ import type { AutomationActionNotice } from './automation-row-action-dispatch' import type { AutomationHostCatalogEntry } from './automation-host-catalog-types' import type { AutomationCreateDestination } from './automation-create-destination' import type { AutomationListRow } from './automation-list-row-identity' -import { EMPTY_AUTOMATION_LIST_FILTER, type AutomationListFilter } from './automation-list-view' +import { + EMPTY_AUTOMATION_LIST_FILTER, + type AutomationListFilter, + type AutomationListSort +} from './automation-list-view' import type { AutomationPaneTab, AutomationRunPageOrigin, @@ -54,6 +58,7 @@ export function useAutomationsPageLocalState(store: AutomationsPageStoreState) { const [isSaving, setIsSaving] = useState(false) const [listSearchQuery, setListSearchQuery] = useState('') const [listFilter, setListFilter] = useState(EMPTY_AUTOMATION_LIST_FILTER) + const [listSort, setListSort] = useState(null) const [createOpen, setCreateOpen] = useState(false) const [createTarget, setCreateTarget] = useState('orca') const [editingAutomationId, setEditingAutomationId] = useState(null) @@ -178,6 +183,8 @@ export function useAutomationsPageLocalState(store: AutomationsPageStoreState) { setListSearchQuery, listFilter, setListFilter, + listSort, + setListSort, createOpen, setCreateOpen, createTarget, diff --git a/src/shared/pane-agent-identity-inventory.test.ts b/src/shared/pane-agent-identity-inventory.test.ts index ee868bfcc16..d493dec1aef 100644 --- a/src/shared/pane-agent-identity-inventory.test.ts +++ b/src/shared/pane-agent-identity-inventory.test.ts @@ -56,7 +56,7 @@ const INVENTORY: readonly InventoryGroup[] = [ 'src/renderer/src/components/agent-session-continuation/AgentSessionContinuationDialog.tsx', 2 ], - ['src/renderer/src/components/automations/AutomationListLocalRows.tsx', 2], + ['src/renderer/src/components/automations/AutomationListLocalRow.tsx', 2], 'src/renderer/src/components/automations/automation-draft-model.ts', ['src/renderer/src/components/automations/automation-list-search-rows.ts', 2], ['src/renderer/src/components/dashboard-popout/AgentMapSnapshotWorkspaceMenu.tsx', 2], From 55dcc5ceeeac04dc515ce7d14c084fdee5b82c69 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:26:36 -0700 Subject: [PATCH 14/23] test: pin terminal Codex home to an explicit managed account (#18935) --- tests/e2e/terminal-codex-home.spec.ts | 52 +++++++++++++++++++++------ 1 file changed, 41 insertions(+), 11 deletions(-) diff --git a/tests/e2e/terminal-codex-home.spec.ts b/tests/e2e/terminal-codex-home.spec.ts index 1f85a4f4c9b..3364152d38c 100644 --- a/tests/e2e/terminal-codex-home.spec.ts +++ b/tests/e2e/terminal-codex-home.spec.ts @@ -1,3 +1,5 @@ +import { mkdirSync, writeFileSync } from 'node:fs' +import path from 'node:path' import { test, expect } from './helpers/orca-app' import { execInTerminal, @@ -27,7 +29,42 @@ test.describe('Terminal Codex runtime home', () => { await ensureTerminalVisible(orcaPage) }) - test('terminal process receives the Orca-managed Codex home', async ({ orcaPage }) => { + test('terminal process receives the selected account Codex home', async ({ + electronApp, + orcaPage + }) => { + const userData = await electronApp.evaluate(({ app }) => app.getPath('userData')) + const accountId = 'e2e-terminal-home' + const managedHomePath = path.join(userData, 'codex-accounts', accountId, 'home') + mkdirSync(managedHomePath, { recursive: true }) + writeFileSync(path.join(managedHomePath, '.orca-managed-home'), `${accountId}\n`) + writeFileSync( + path.join(managedHomePath, 'auth.json'), + JSON.stringify({ OPENAI_API_KEY: 'e2e-placeholder' }) + ) + await orcaPage.evaluate( + async ({ accountId, managedHomePath }) => { + const state = window.__store!.getState() + await state.updateSettings({ + codexManagedAccounts: [ + { + id: accountId, + email: 'terminal-home@example.invalid', + managedHomePath, + createdAt: 1, + updatedAt: 1, + lastAuthenticatedAt: 1 + } + ], + activeCodexManagedAccountId: accountId, + activeCodexManagedAccountIdsByRuntime: { host: accountId, wsl: {} } + }) + const tab = state.createTab(state.activeWorktreeId!) + state.setActiveTab(tab.id) + state.setActiveTabType('terminal') + }, + { accountId, managedHomePath } + ) await waitForActiveTerminalManager(orcaPage) const ptyId = await waitForActivePanePtyId(orcaPage) const marker = `__ORCA_CODEX_HOME_E2E_${Date.now()}__` @@ -43,17 +80,10 @@ test.describe('Terminal Codex runtime home', () => { .poll( async () => { probe = readCodexHomeProbe(await getTerminalContent(orcaPage), marker) - return Boolean( - probe?.codexHome && - probe.orcaCodexHome && - probe.codexHome === probe.orcaCodexHome && - /[\\/]codex-runtime-home[\\/]home$/.test(probe.codexHome) - ) + return probe }, - { timeout: 15_000, message: 'Terminal did not expose Orca-managed Codex home env' } + { timeout: 15_000, message: 'Terminal did not expose the selected Codex account home' } ) - .toBe(true) - - expect(probe?.codexHome).toBe(probe?.orcaCodexHome) + .toEqual({ codexHome: managedHomePath, orcaCodexHome: managedHomePath }) }) }) From 59756b8a1cec8b266a1461056f18c768454bfcb0 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:31:11 -0700 Subject: [PATCH 15/23] test: deliver real terminal input and preserve setup reports (#18939) --- .../e2e/terminal-scroll-intent-follow.spec.ts | 44 +++++++++++-------- .../terminal-send-agent-prompt-submit.spec.ts | 1 + tests/tools/repro-terminal-send-submit.mjs | 4 +- 3 files changed, 30 insertions(+), 19 deletions(-) diff --git a/tests/e2e/terminal-scroll-intent-follow.spec.ts b/tests/e2e/terminal-scroll-intent-follow.spec.ts index c4d8b171fed..ae2dfa15466 100644 --- a/tests/e2e/terminal-scroll-intent-follow.spec.ts +++ b/tests/e2e/terminal-scroll-intent-follow.spec.ts @@ -168,6 +168,7 @@ async function injectQueuedWriteThenType(page: Page, paneKey: string): Promise { const injectionTarget = window as Window & { __terminalPtyDataInjection?: { inject: (paneKey: string, data: string) => boolean } + __releaseScrollIntentTestWrite?: () => void } const state = window.__store?.getState() const worktreeId = state?.activeWorktreeId @@ -184,38 +185,44 @@ async function injectQueuedWriteThenType(page: Page, paneKey: string): Promise void } | null } = { write: null } + const heldWrites: { data: string; callback?: () => void }[] = [] terminal.write = ((data: string, callback?: () => void) => { - holder.write = { data, callback } + heldWrites.push({ data, callback }) }) as typeof terminal.write + injectionTarget.__releaseScrollIntentTestWrite = () => { + terminal.write = originalWrite + delete injectionTarget.__releaseScrollIntentTestWrite + for (const held of heldWrites) { + originalWrite.call(terminal, held.data, held.callback) + } + } try { const payload = '\x1b[?2026h\r\x1b[2KWorking in-flight\x1b[?2026l' if (!injectionTarget.__terminalPtyDataInjection?.inject(targetPaneKey, payload)) { throw new Error('PTY injector unavailable') } + if (heldWrites.length === 0) { + throw new Error('Foreground terminal write was not captured') + } const textarea = pane.container.querySelector('.xterm-helper-textarea') if (!textarea) { throw new Error('xterm helper textarea unavailable') } textarea.focus() - const event = new KeyboardEvent('keydown', { - bubbles: true, - cancelable: true, - key: 'x', - code: 'KeyX' - }) - Object.defineProperty(event, 'keyCode', { configurable: true, value: 88 }) - Object.defineProperty(event, 'which', { configurable: true, value: 88 }) - textarea.dispatchEvent(event) - } finally { - terminal.write = originalWrite + } catch (error) { + injectionTarget.__releaseScrollIntentTestWrite() + throw error } - const heldWrite = holder.write - if (!heldWrite) { - throw new Error('Foreground terminal write was not captured') - } - originalWrite.call(terminal, heldWrite.data, heldWrite.callback) }, paneKey) + try { + await page.keyboard.press('x') + } finally { + await page.evaluate(() => { + ;( + window as Window & { __releaseScrollIntentTestWrite?: () => void } + ).__releaseScrollIntentTestWrite?.() + }) + } } async function startStreamingFixturePhase1(page: Page): Promise { @@ -308,5 +315,6 @@ test.describe('terminal scroll intent keeps following output', () => { { timeout: 5_000, intervals: [25] } ) .toBe(0) + await waitForMarkerAtBottom(orcaPage, 'STREAM_PHASE2_DONE') }) }) diff --git a/tests/e2e/terminal-send-agent-prompt-submit.spec.ts b/tests/e2e/terminal-send-agent-prompt-submit.spec.ts index 3567cb1d8a2..c1a602dd9d1 100644 --- a/tests/e2e/terminal-send-agent-prompt-submit.spec.ts +++ b/tests/e2e/terminal-send-agent-prompt-submit.spec.ts @@ -58,6 +58,7 @@ async function createFakeCodexTerminal( if (!worktree) { throw new Error(`runtime did not register ${testRepoPath}`) } + rmSync(fixtureReport, { force: true }) const created = await client.call<{ terminal: { handle: string } }>('terminal.create', { worktree: `id:${worktree.id}`, command: [fakeCodexCommand, ...args].join(' '), diff --git a/tests/tools/repro-terminal-send-submit.mjs b/tests/tools/repro-terminal-send-submit.mjs index 7e1c0152012..41d2464cbbc 100644 --- a/tests/tools/repro-terminal-send-submit.mjs +++ b/tests/tools/repro-terminal-send-submit.mjs @@ -186,7 +186,9 @@ async function parentMain() { const expectBlocked = hasFlag('expect-blocked') const providedHandle = argValue('terminal') await mkdir(tempDir, { recursive: true }) - await rm(reportPath, { force: true }) + if (!providedHandle) { + await rm(reportPath, { force: true }) + } let handle = providedHandle if (!handle) { From ab8e10e298df8be5b3294553d593a1420349c662 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:43:41 -0700 Subject: [PATCH 16/23] test: isolate skill cloud fixture ports across workers (#18942) --- .../e2e/helpers/remote-skill-cloud-fixture.ts | 41 ++++++++++++------- .../remote-skill-cloud-fixture.unit.test.ts | 41 +++++++++++++++++++ tests/e2e/paired-skill-installation.spec.ts | 8 ++-- tests/e2e/ssh-skill-installation.spec.ts | 21 ++++++---- 4 files changed, 85 insertions(+), 26 deletions(-) create mode 100644 tests/e2e/helpers/remote-skill-cloud-fixture.unit.test.ts diff --git a/tests/e2e/helpers/remote-skill-cloud-fixture.ts b/tests/e2e/helpers/remote-skill-cloud-fixture.ts index f1d28d926b7..8e1650947dd 100644 --- a/tests/e2e/helpers/remote-skill-cloud-fixture.ts +++ b/tests/e2e/helpers/remote-skill-cloud-fixture.ts @@ -8,13 +8,12 @@ import { } from '../../../src/main/skills/skill-package-creation' import { SKILL_PACKAGE_CONTENT_TYPE } from '../../../src/shared/skill-package-manifest' -export const REMOTE_SKILL_CLOUD_PORT = Number(process.env.ORCA_E2E_SKILL_CLOUD_PORT ?? '43961') -export const REMOTE_SKILL_CLOUD_ORIGIN = `http://127.0.0.1:${REMOTE_SKILL_CLOUD_PORT}` export const REMOTE_SKILL_PACKAGE_ID = 'package_remote_e2e' export const REMOTE_SKILL_VERSION_ID = 'version_remote_e2e' export const REMOTE_SKILL_NAME = 'remote-e2e-skill' export type RemoteSkillCloudFixture = { + origin: string archive: CreatedSkillPackage bytes: Buffer requests: { method: string; path: string; body: unknown }[] @@ -39,19 +38,30 @@ export async function startRemoteSkillCloudFixture(): Promise { - void handleRemoteSkillCloudRequest({ request, response, archive, bytes, requests }).catch( - (error) => { - response.writeHead(500, { 'content-type': 'application/json' }) - response.end(JSON.stringify({ code: 'fixture_failed', message: String(error) })) - } - ) + void handleRemoteSkillCloudRequest({ + request, + response, + archive, + bytes, + requests, + origin + }).catch((error) => { + response.writeHead(500, { 'content-type': 'application/json' }) + response.end(JSON.stringify({ code: 'fixture_failed', message: String(error) })) + }) }) await new Promise((resolve, reject) => { server.once('error', reject) - server.listen(REMOTE_SKILL_CLOUD_PORT, '127.0.0.1', resolve) + server.listen(Number(process.env.ORCA_E2E_SKILL_CLOUD_PORT ?? 0), '127.0.0.1', resolve) }) - return { archive, bytes, requests, root, server } + const address = server.address() + if (!address || typeof address === 'string') { + throw new Error('Skill fixture has no TCP address') + } + origin = `http://127.0.0.1:${address.port}` + return { archive, bytes, requests, root, server, origin } } export async function stopRemoteSkillCloudFixture(fixture: RemoteSkillCloudFixture): Promise { @@ -60,13 +70,14 @@ export async function stopRemoteSkillCloudFixture(fixture: RemoteSkillCloudFixtu } async function handleRemoteSkillCloudRequest(input: { + origin: string request: IncomingMessage response: ServerResponse archive: CreatedSkillPackage bytes: Buffer requests: RemoteSkillCloudFixture['requests'] }): Promise { - const path = new URL(input.request.url ?? '/', REMOTE_SKILL_CLOUD_ORIGIN).pathname + const path = new URL(input.request.url ?? '/', input.origin).pathname if (input.request.method === 'GET' && path === '/package.tar.gz') { input.requests.push({ method: 'GET', path, body: null }) input.response.writeHead(200, { @@ -84,17 +95,19 @@ async function handleRemoteSkillCloudRequest(input: { const body = JSON.parse(await readRequestBody(input.request)) as unknown input.requests.push({ method: 'POST', path, body }) input.response.writeHead(200, { 'content-type': 'application/json' }) - input.response.end(JSON.stringify(downloadGrant(input.archive, input.bytes.length))) + input.response.end( + JSON.stringify(downloadGrant(input.archive, input.bytes.length, input.origin)) + ) return } input.response.writeHead(404, { 'content-type': 'application/json' }) input.response.end(JSON.stringify({ code: 'not_found', message: 'Not found' })) } -function downloadGrant(archive: CreatedSkillPackage, compressedBytes: number) { +function downloadGrant(archive: CreatedSkillPackage, compressedBytes: number, origin: string) { return { grant: { - url: `${REMOTE_SKILL_CLOUD_ORIGIN}/package.tar.gz`, + url: `${origin}/package.tar.gz`, expiresAt: '2099-01-01T00:00:00.000Z' }, version: { diff --git a/tests/e2e/helpers/remote-skill-cloud-fixture.unit.test.ts b/tests/e2e/helpers/remote-skill-cloud-fixture.unit.test.ts new file mode 100644 index 00000000000..70291cf0e8f --- /dev/null +++ b/tests/e2e/helpers/remote-skill-cloud-fixture.unit.test.ts @@ -0,0 +1,41 @@ +import { expect, it, vi } from 'vitest' +import { + REMOTE_SKILL_PACKAGE_ID, + REMOTE_SKILL_VERSION_ID, + startRemoteSkillCloudFixture, + stopRemoteSkillCloudFixture +} from './remote-skill-cloud-fixture' + +it('serves concurrent skill fixtures from independent bound origins', async () => { + vi.stubEnv('ORCA_E2E_SKILL_CLOUD_PORT', undefined) + const results = await Promise.allSettled([ + startRemoteSkillCloudFixture(), + startRemoteSkillCloudFixture() + ]) + const fixtures = results.flatMap((result) => + result.status === 'fulfilled' ? [result.value] : [] + ) + try { + expect(results.every((result) => result.status === 'fulfilled')).toBe(true) + expect(new Set(fixtures.map((fixture) => fixture.origin)).size).toBe(2) + for (const fixture of fixtures) { + const response = await fetch( + `${fixture.origin}/v1/skill-packages/${REMOTE_SKILL_PACKAGE_ID}/versions/${REMOTE_SKILL_VERSION_ID}/download-grants`, + { + method: 'POST', + body: '{}', + headers: { 'content-type': 'application/json' } + } + ) + expect(response.status).toBe(200) + const result = (await response.json()) as { grant: { url: string } } + expect(result.grant.url).toBe(`${fixture.origin}/package.tar.gz`) + const archive = await fetch(result.grant.url) + expect(Buffer.from(await archive.arrayBuffer())).toEqual(fixture.bytes) + expect(fixture.requests).toHaveLength(2) + } + } finally { + await Promise.all(fixtures.map(stopRemoteSkillCloudFixture)) + vi.unstubAllEnvs() + } +}) diff --git a/tests/e2e/paired-skill-installation.spec.ts b/tests/e2e/paired-skill-installation.spec.ts index ac31a81e2f1..0268dbb84a3 100644 --- a/tests/e2e/paired-skill-installation.spec.ts +++ b/tests/e2e/paired-skill-installation.spec.ts @@ -14,7 +14,6 @@ import { type HeadlessPairedRuntimeHost } from './helpers/headless-paired-runtime-host' import { - REMOTE_SKILL_CLOUD_ORIGIN, REMOTE_SKILL_NAME, REMOTE_SKILL_PACKAGE_ID, REMOTE_SKILL_VERSION_ID, @@ -119,13 +118,14 @@ test('installs on a headless serve runtime through the same contract', async ({ }) function cloudClientEnvironment(): Record { + const { origin } = requireCloudFixture() return { - ORCA_ARTIFACTS_API_URL: REMOTE_SKILL_CLOUD_ORIGIN, - ORCA_CLOUD_API_URL: REMOTE_SKILL_CLOUD_ORIGIN, + ORCA_ARTIFACTS_API_URL: origin, + ORCA_CLOUD_API_URL: origin, ORCA_CLOUD_CLIENT_ID: 'skills-e2e-client', ORCA_CLOUD_DEV_AUTH: '1', ORCA_CLOUD_ALLOW_PLAINTEXT_SESSION: '1', - ORCA_SKILL_PACKAGE_DOWNLOAD_ORIGINS: REMOTE_SKILL_CLOUD_ORIGIN + ORCA_SKILL_PACKAGE_DOWNLOAD_ORIGINS: origin } } diff --git a/tests/e2e/ssh-skill-installation.spec.ts b/tests/e2e/ssh-skill-installation.spec.ts index a102478fb62..794883b2cd1 100644 --- a/tests/e2e/ssh-skill-installation.spec.ts +++ b/tests/e2e/ssh-skill-installation.spec.ts @@ -10,7 +10,6 @@ import { import { connectDockerSshRelayTarget } from './helpers/docker-ssh-relay-connection' import { waitForActiveWorktree, waitForSessionReady } from './helpers/store' import { - REMOTE_SKILL_CLOUD_ORIGIN, REMOTE_SKILL_NAME, REMOTE_SKILL_PACKAGE_ID, REMOTE_SKILL_VERSION_ID, @@ -25,13 +24,19 @@ const REMOTE_FOLDER = '/tmp/orca-skill-folder-workspace' let cloud: RemoteSkillCloudFixture | null = null test.use({ - orcaAppExtraEnv: { - ORCA_ARTIFACTS_API_URL: REMOTE_SKILL_CLOUD_ORIGIN, - ORCA_CLOUD_API_URL: REMOTE_SKILL_CLOUD_ORIGIN, - ORCA_CLOUD_CLIENT_ID: 'skills-e2e-client', - ORCA_CLOUD_DEV_AUTH: '1', - ORCA_CLOUD_ALLOW_PLAINTEXT_SESSION: '1', - ORCA_SKILL_PACKAGE_DOWNLOAD_ORIGINS: REMOTE_SKILL_CLOUD_ORIGIN + // oxlint-disable-next-line no-empty-pattern -- The server starts in beforeAll before this test fixture runs. + orcaAppExtraEnv: async ({}, provideEnv) => { + if (!cloud) { + throw new Error('Skill cloud fixture unavailable') + } + await provideEnv({ + ORCA_ARTIFACTS_API_URL: cloud.origin, + ORCA_CLOUD_API_URL: cloud.origin, + ORCA_CLOUD_CLIENT_ID: 'skills-e2e-client', + ORCA_CLOUD_DEV_AUTH: '1', + ORCA_CLOUD_ALLOW_PLAINTEXT_SESSION: '1', + ORCA_SKILL_PACKAGE_DOWNLOAD_ORIGINS: cloud.origin + }) } }) From 22a7bfd3804717898e82b30ddbf880c5511b8c5f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:45:48 -0700 Subject: [PATCH 17/23] test: align Source Control AI fixtures with current settings (#18941) --- tests/e2e/helpers/source-control-ai-generation.ts | 15 +++++++++++++-- tests/e2e/helpers/source-control-ai-generators.ts | 6 +++--- 2 files changed, 16 insertions(+), 5 deletions(-) diff --git a/tests/e2e/helpers/source-control-ai-generation.ts b/tests/e2e/helpers/source-control-ai-generation.ts index f6be16d7342..c93a20e37a1 100644 --- a/tests/e2e/helpers/source-control-ai-generation.ts +++ b/tests/e2e/helpers/source-control-ai-generation.ts @@ -67,7 +67,7 @@ export async function seedCreatePrComposer(page: Page): Promise<{ prWorktreePath: string primaryBranch: string }> { - return page.evaluate(async () => { + const seeded = await page.evaluate(async () => { const store = window.__store ?? (() => { @@ -101,6 +101,7 @@ export async function seedCreatePrComposer(page: Page): Promise<{ const eligibility = { provider: 'github' as const, review: null, + reviewLookupOutcome: 'not_found' as const, canCreate: true, blockedReason: null, nextAction: null, @@ -121,7 +122,7 @@ export async function seedCreatePrComposer(page: Page): Promise<{ ...current.remoteStatusesByWorktree, [prWorktree.id]: { hasUpstream: true, - upstreamName: `origin/${branch}`, + upstreamName: primaryBranch, ahead: 0, behind: 0 } @@ -130,6 +131,10 @@ export async function seedCreatePrComposer(page: Page): Promise<{ args.branch === branch ? eligibility : { ...eligibility, canCreate: false }, fetchHostedReviewForBranch: async () => null, fetchPRForBranch: async () => null, + enqueueGitHubPRRefresh: () => undefined, + // Ignore provider work queued before this generation-only fixture was installed. + getEffectiveGitHubPRRefreshState: () => undefined, + prRefreshStates: {}, fetchUpstreamStatus: async () => undefined, setUpstreamStatus: () => undefined })) @@ -141,6 +146,12 @@ export async function seedCreatePrComposer(page: Page): Promise<{ primaryBranch } }) + // Checks reads fresh Git state instead of the seeded store cache. + execFileSync('git', ['branch', '--set-upstream-to', seeded.primaryBranch], { + cwd: seeded.prWorktreePath, + stdio: 'pipe' + }) + return seeded } export async function seedCommitMessageComposer(page: Page): Promise<{ diff --git a/tests/e2e/helpers/source-control-ai-generators.ts b/tests/e2e/helpers/source-control-ai-generators.ts index be3f1b43247..8c09bb8b556 100644 --- a/tests/e2e/helpers/source-control-ai-generators.ts +++ b/tests/e2e/helpers/source-control-ai-generators.ts @@ -14,13 +14,13 @@ async function setCustomGenerator(page: Page, scriptPath: string): Promise } await store.getState().updateSettings({ activeRuntimeEnvironmentId: null, - commitMessageAi: { - ...currentSettings.commitMessageAi, + sourceControlAi: { enabled: true, agentId: 'custom' as const, selectedModelByAgent: {}, selectedThinkingByModel: {}, - customPrompt: '', + instructionsByOperation: {}, + actions: {}, customAgentCommand: `node ${JSON.stringify(scriptPath)}` } }) From 7bb54cc2f73c08a3df026c28766afd48b0e24471 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:56:57 -0700 Subject: [PATCH 18/23] ci: reduce runner overhead and disposable package compression (#18948) * ci: reduce PR runner overhead and package compression time * ci: validate mobile when its dependency action changes --- .github/workflows/mobile.yml | 27 +--- .github/workflows/pr.yml | 108 ++++++-------- .github/workflows/skill-update-roundtrip.yml | 4 + config/scripts/pr-code-change-scope.test.mjs | 7 +- config/scripts/pr-e2e-gate-contract.test.mjs | 30 ++-- docs/reference/ci-runner-efficiency.md | 99 +++++++++++++ docs/reference/windows-signing-runner-time.md | 137 ++++++++++++++++++ 7 files changed, 311 insertions(+), 101 deletions(-) create mode 100644 docs/reference/ci-runner-efficiency.md create mode 100644 docs/reference/windows-signing-runner-time.md diff --git a/.github/workflows/mobile.yml b/.github/workflows/mobile.yml index 59f6cf20bf4..6dbfc02aa3c 100644 --- a/.github/workflows/mobile.yml +++ b/.github/workflows/mobile.yml @@ -15,8 +15,13 @@ on: # Why: this job holds the only checks that load the Fastfile, so edits to # it or to the release workflow it guards must re-run them. - '.github/workflows/mobile.yml' + - '.github/actions/install-node-dependencies/**' - '.github/workflows/mobile-ios-release.yml' +concurrency: + group: mobile-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + jobs: verify: runs-on: ubuntu-latest @@ -35,10 +40,7 @@ jobs: - name: Checkout uses: actions/checkout@v6 - - name: Setup Node.js - uses: actions/setup-node@v6 - with: - node-version-file: package.json + - uses: ./.github/actions/install-node-dependencies # bundler-cache installs mobile/Gemfile.lock, so this job is also what # proves the pinned fastlane the release workflow depends on still @@ -50,23 +52,6 @@ jobs: bundler-cache: true working-directory: mobile - - name: Setup pnpm - uses: pnpm/setup@v2 - with: - install: false - - # Why: the mobile typecheck imports shared types from ../src/shared, and - # some of those files import runtime deps (tweetnacl, ws) resolved from - # the repo-root node_modules. Without a root install, tsc fails with - # "Cannot find module 'tweetnacl'/'ws'". Mobile is a separate pnpm project - # (not in the root workspace), so this is a distinct install. - # --ignore-scripts skips the root postinstall (Electron native-module - # rebuild) which is irrelevant to a type-only check and would only add - # time and failure surface on this ubuntu mobile runner. - - name: Install root dependencies - working-directory: . - run: pnpm install --frozen-lockfile --ignore-scripts - - name: Install dependencies run: pnpm install --frozen-lockfile diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index a749214e232..ba2eaf83192 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -41,6 +41,10 @@ jobs: managed_hook_node18: ${{ steps.filter.outputs.managed_hook_node18 }} package: ${{ steps.filter.outputs.package }} package_windows: ${{ steps.filter.outputs.package_windows }} + e2e_should_run: ${{ steps.e2e_filter.outputs.should_run }} + test_files: ${{ steps.e2e_filter.outputs.test_files }} + ssh_source_changed: ${{ steps.e2e_filter.outputs.ssh_source_changed }} + native_ime_source_changed: ${{ steps.e2e_filter.outputs.native_ime_source_changed }} steps: - name: Checkout uses: actions/checkout@v6 @@ -66,6 +70,37 @@ jobs: printf '%s\n' "$CHANGED" printf '%s\n' "$CHANGED" | node config/scripts/pr-code-change-scope.mjs | tee -a "$GITHUB_OUTPUT" + # Reuse the path-detector checkout instead of queuing another runner. + - name: Filter changed E2E specs + id: e2e_filter + if: github.event.pull_request.draft != true && steps.filter.outputs.should_run == 'true' + run: | + set -euo pipefail + BASE="${{ github.event.pull_request.base.sha }}" + HEAD="${{ github.event.pull_request.head.sha }}" + CHANGED="$(git diff --name-only --diff-filter=AMCR --merge-base "$BASE" "$HEAD")" + # Source routes are executable contracts so a test can prove exact + # authorities, exclusions, and sentinels without evaluating workflow shell. + TEST_FILES_JSON="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs)" + echo "test_files=$TEST_FILES_JSON" >> "$GITHUB_OUTPUT" + # Why a separate signal: the Docker-SSH lane must trigger on SSH source, not on a + # spec name surviving in a route's list. Same routes, so the two cannot drift. + SSH_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --ssh-source)" + echo "ssh_source_changed=$SSH_SOURCE_CHANGED" >> "$GITHUB_OUTPUT" + echo "SSH source changed: $SSH_SOURCE_CHANGED" + # Why its own signal: the real-IME lane is a whole ibus session, not a spec, so it must + # trigger on IME source rather than on a spec name in some route's list. + NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)" + echo "native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED" >> "$GITHUB_OUTPUT" + echo "Native IME source changed: $NATIVE_IME_SOURCE_CHANGED" + if [ "$TEST_FILES_JSON" != '[]' ]; then + echo "should_run=true" >> "$GITHUB_OUTPUT" + echo "Changed E2E specs: $TEST_FILES_JSON" + else + echo "should_run=false" >> "$GITHUB_OUTPUT" + echo "No changed E2E specs" + fi + static_analysis: name: static analysis needs: [code_paths] @@ -712,7 +747,11 @@ jobs: - name: Package unpacked app env: ORCA_REUSE_PREPARED_NATIVE_RUNTIME: '1' - run: pnpm exec electron-builder --config config/electron-builder.config.cjs --linux AppImage deb rpm --x64 --publish never + # PR artifacts are only inspected locally; gzip avoids release-size xz compression. + run: >- + pnpm exec electron-builder --config config/electron-builder.config.cjs + --linux AppImage deb rpm --x64 --publish never + --config.deb.compression=gz --config.rpm.compression=gzip - name: Verify root-package marker payloads run: | @@ -861,65 +900,10 @@ jobs: - name: Smoke packaged CLI run: node config/scripts/smoke-packaged-cli.mjs --app-dir=dist/win-unpacked - # Why: PR E2E is advisory and only validates changed specs; scheduled and - # release runs retain full-suite coverage. - e2e-paths: - name: detect changed e2e specs - needs: [code_paths] - runs-on: ubuntu-latest - if: github.event.pull_request.draft != true && needs.code_paths.outputs.should_run == 'true' - # Why: detector only needs to read the checkout; do not inherit repo defaults. - permissions: - contents: read - outputs: - should_run: ${{ steps.filter.outputs.should_run }} - test_files: ${{ steps.filter.outputs.test_files }} - ssh_source_changed: ${{ steps.filter.outputs.ssh_source_changed }} - native_ime_source_changed: ${{ steps.filter.outputs.native_ime_source_changed }} - steps: - - name: Checkout - uses: actions/checkout@v6 - with: - # Why blob:none: full history is needed for the merge-base diff, but historical - # file contents are not. Blobs are ~89% of this repo's pack, and Git fetches the - # few this job actually reads on demand. - fetch-depth: 0 - filter: blob:none - persist-credentials: false - - - name: Filter changed E2E specs - id: filter - run: | - set -euo pipefail - BASE="${{ github.event.pull_request.base.sha }}" - HEAD="${{ github.event.pull_request.head.sha }}" - CHANGED="$(git diff --name-only --diff-filter=AMCR --merge-base "$BASE" "$HEAD")" - # Source routes are executable contracts so a test can prove exact - # authorities, exclusions, and sentinels without evaluating workflow shell. - TEST_FILES_JSON="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs)" - echo "test_files=$TEST_FILES_JSON" >> "$GITHUB_OUTPUT" - # Why a separate signal: the Docker-SSH lane must trigger on SSH source, not on a - # spec name surviving in a route's list. Same routes, so the two cannot drift. - SSH_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --ssh-source)" - echo "ssh_source_changed=$SSH_SOURCE_CHANGED" >> "$GITHUB_OUTPUT" - echo "SSH source changed: $SSH_SOURCE_CHANGED" - # Why its own signal: the real-IME lane is a whole ibus session, not a spec, so it must - # trigger on IME source rather than on a spec name in some route's list. - NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)" - echo "native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED" >> "$GITHUB_OUTPUT" - echo "Native IME source changed: $NATIVE_IME_SOURCE_CHANGED" - if [ "$TEST_FILES_JSON" != '[]' ]; then - echo "should_run=true" >> "$GITHUB_OUTPUT" - echo "Changed E2E specs: $TEST_FILES_JSON" - else - echo "should_run=false" >> "$GITHUB_OUTPUT" - echo "No changed E2E specs" - fi - e2e: name: e2e - needs: e2e-paths - if: needs.e2e-paths.outputs.should_run == 'true' + needs: code_paths + if: needs.code_paths.outputs.e2e_should_run == 'true' # Why: reusable e2e.yml only checkouts, builds, and uploads artifacts. permissions: contents: read @@ -928,8 +912,8 @@ jobs: # The synthetic pull-request merge ref can disappear while this reusable # workflow is queued. The head SHA is immutable and works for every PR. ref: ${{ github.event.pull_request.head.sha }} - test_files: ${{ needs.e2e-paths.outputs.test_files }} - ssh_source_changed: ${{ needs.e2e-paths.outputs.ssh_source_changed }} + test_files: ${{ needs.code_paths.outputs.test_files }} + ssh_source_changed: ${{ needs.code_paths.outputs.ssh_source_changed }} # Why this is not in verify's needs: it is the first PR-gate run of a harness whose reliability # is only known from nightly main runs (20/20 green, 2026-08-09..2026-08-29, p50 3m25s). It @@ -939,8 +923,8 @@ jobs: # require `success || skipped` outside the strict loop — see the note on `e2e`. terminal_ime_native: name: real IME - needs: e2e-paths - if: needs.e2e-paths.outputs.native_ime_source_changed == 'true' + needs: code_paths + if: needs.code_paths.outputs.native_ime_source_changed == 'true' # Why: the reusable workflow only checks out, builds, and uploads artifacts. permissions: contents: read diff --git a/.github/workflows/skill-update-roundtrip.yml b/.github/workflows/skill-update-roundtrip.yml index 71fcf264f69..239f1b2f27c 100644 --- a/.github/workflows/skill-update-roundtrip.yml +++ b/.github/workflows/skill-update-roundtrip.yml @@ -22,6 +22,10 @@ on: - main paths: *skill-roundtrip-paths +concurrency: + group: skill-roundtrip-${{ github.event_name }}-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + jobs: roundtrip: strategy: diff --git a/config/scripts/pr-code-change-scope.test.mjs b/config/scripts/pr-code-change-scope.test.mjs index 4642372135c..f31822e5b93 100644 --- a/config/scripts/pr-code-change-scope.test.mjs +++ b/config/scripts/pr-code-change-scope.test.mjs @@ -414,10 +414,11 @@ describe('PR Checks skip wiring', () => { }) it('skips e2e detection on docs-only PRs without dropping the draft gate', () => { - expect(prWorkflow.jobs['e2e-paths'].needs).toEqual(['code_paths']) - expect(prWorkflow.jobs['e2e-paths'].if).toBe( - "github.event.pull_request.draft != true && needs.code_paths.outputs.should_run == 'true'" + const filter = prWorkflow.jobs.code_paths.steps.find((step) => step.id === 'e2e_filter') + expect(filter.if).toBe( + "github.event.pull_request.draft != true && steps.filter.outputs.should_run == 'true'" ) + expect(prWorkflow.jobs['e2e-paths']).toBeUndefined() }) it('lets verify pass skipped jobs the classifier turned off', () => { diff --git a/config/scripts/pr-e2e-gate-contract.test.mjs b/config/scripts/pr-e2e-gate-contract.test.mjs index ceac6b8cc6e..67e271868df 100644 --- a/config/scripts/pr-e2e-gate-contract.test.mjs +++ b/config/scripts/pr-e2e-gate-contract.test.mjs @@ -39,7 +39,7 @@ const nativeImeSpec = readFileSync( 'utf8' ) -const filterStep = prWorkflow.jobs['e2e-paths'].steps.find( +const filterStep = prWorkflow.jobs.code_paths.steps.find( (step) => step.name === 'Filter changed E2E specs' ) const rollbackStep = prWorkflow.jobs.static_analysis.steps.find( @@ -106,16 +106,16 @@ describe('PR E2E gate contract', () => { // Why: without this the job could lose its filter and run on every PR — the // cost the path filter exists to avoid — while the gate assertions above // stay green. - expect(prWorkflow.jobs.e2e.needs).toBe('e2e-paths') - expect(prWorkflow.jobs.e2e.if).toBe("needs.e2e-paths.outputs.should_run == 'true'") - expect(prWorkflow.jobs['e2e-paths'].outputs.should_run).toBe( - '${{ steps.filter.outputs.should_run }}' + expect(prWorkflow.jobs.e2e.needs).toBe('code_paths') + expect(prWorkflow.jobs.e2e.if).toBe("needs.code_paths.outputs.e2e_should_run == 'true'") + expect(prWorkflow.jobs.code_paths.outputs.e2e_should_run).toBe( + '${{ steps.e2e_filter.outputs.should_run }}' ) - expect(prWorkflow.jobs['e2e-paths'].outputs.test_files).toBe( - '${{ steps.filter.outputs.test_files }}' + expect(prWorkflow.jobs.code_paths.outputs.test_files).toBe( + '${{ steps.e2e_filter.outputs.test_files }}' ) expect(prWorkflow.jobs.e2e.with.ref).toBe('${{ github.event.pull_request.head.sha }}') - expect(prWorkflow.jobs.e2e.with.test_files).toBe('${{ needs.e2e-paths.outputs.test_files }}') + expect(prWorkflow.jobs.e2e.with.test_files).toBe('${{ needs.code_paths.outputs.test_files }}') }) it('enforces every job verify depends on', () => { @@ -360,11 +360,11 @@ describe('PR E2E gate contract', () => { expect(sshLaneCondition).toContain("inputs.ssh_source_changed == 'true' ||") expect(e2eWorkflow.on.workflow_call.inputs.ssh_source_changed.type).toBe('string') - expect(prWorkflow.jobs['e2e-paths'].outputs.ssh_source_changed).toBe( - '${{ steps.filter.outputs.ssh_source_changed }}' + expect(prWorkflow.jobs.code_paths.outputs.ssh_source_changed).toBe( + '${{ steps.e2e_filter.outputs.ssh_source_changed }}' ) expect(prWorkflow.jobs.e2e.with.ssh_source_changed).toBe( - '${{ needs.e2e-paths.outputs.ssh_source_changed }}' + '${{ needs.code_paths.outputs.ssh_source_changed }}' ) expect(filterStep.run).toContain('pr-e2e-source-routing.mjs --ssh-source') expect(filterStep.run).toContain('ssh_source_changed=$SSH_SOURCE_CHANGED') @@ -565,12 +565,12 @@ describe('PR E2E gate contract', () => { expect(prWorkflow.jobs.terminal_ime_native.uses).toBe( './.github/workflows/terminal-ime-e2e.yml' ) - expect(prWorkflow.jobs.terminal_ime_native.needs).toBe('e2e-paths') + expect(prWorkflow.jobs.terminal_ime_native.needs).toBe('code_paths') expect(prWorkflow.jobs.terminal_ime_native.if).toBe( - "needs.e2e-paths.outputs.native_ime_source_changed == 'true'" + "needs.code_paths.outputs.native_ime_source_changed == 'true'" ) - expect(prWorkflow.jobs['e2e-paths'].outputs.native_ime_source_changed).toBe( - '${{ steps.filter.outputs.native_ime_source_changed }}' + expect(prWorkflow.jobs.code_paths.outputs.native_ime_source_changed).toBe( + '${{ steps.e2e_filter.outputs.native_ime_source_changed }}' ) expect(filterStep.run).toContain('pr-e2e-source-routing.mjs --native-ime-source') expect(filterStep.run).toContain('native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED') diff --git a/docs/reference/ci-runner-efficiency.md b/docs/reference/ci-runner-efficiency.md new file mode 100644 index 00000000000..e569f749102 --- /dev/null +++ b/docs/reference/ci-runner-efficiency.md @@ -0,0 +1,99 @@ +# CI efficiency and runner capacity + +Audit date: September 5, 2026. No paid capacity or provider configuration changed. + +## Measurements and changes + +Three recent successful PR runs used 54.6–64.9 aggregate runner minutes: +[33998366568](https://github.com/stablyai/orca/actions/runs/33998366568), +[33998220287](https://github.com/stablyai/orca/actions/runs/33998220287), and +[33998181502](https://github.com/stablyai/orca/actions/runs/33998181502). +These are sums of active job durations, excluding skipped jobs; they are not +billing minutes or queue time. This small sample is not a historical average. + +- Consolidate E2E routing into the existing code-path detector. The removed + detector occupied 20–22 seconds and required another runner allocation and + full-history checkout per nondraft code PR. The same routing commands remain, + including SSH and native IME selection; actual E2E results remain advisory. + A routing-script error now fails the required code-path detector. +- Use gzip for PR-only Debian/RPM artifacts. The two sampled Linux packaging + jobs took 8m10s and 8m19s overall; one spent 3m47s in electron-builder. Its + default Debian/RPM compression is xz. PR artifacts are inspected on the same + runner, so their download size offers no benefit. Keep all AppImage, Debian, + RPM, payload, launcher, and shutdown checks. Release compression is unchanged. + Compression savings need a hosted run; do not equate the full packaging step + with removable compression time. +- Cancel superseded Mobile Checks and Skill update round-trip PR runs. The + skill matrix has 13 jobs. Preserve non-cancelling main/merge-group skill runs, + with separate concurrency groups per event. +- Reuse the existing script-free root dependency action in Mobile Checks, + including the pnpm cache keyed by both root and mobile lockfiles. The root + install remains necessary because mobile types import root dependencies. + +The repository already has eight unit shards, path-scoped platform checks, +native caches, one shared E2E build, PR cancellation, incremental TypeScript +caching, and changed-spec E2E routing. Increasing shards would increase setup +work and simultaneous runner demand. Do not adjust the count without comparing +critical-path time and aggregate job time on the same commit. + +## Runner recommendations + +The repository is **public**, verified using the GitHub API. Standard +GitHub-hosted Linux, Windows, and macOS runners have free compute minutes for +public repositories. Queue pressure and third-party provider allowances still +matter; artifact storage and larger runners have separate billing rules. +See [GitHub Actions billing](https://docs.github.com/en/billing/concepts/product-billing/github-actions). + +1. Keep standard GitHub-hosted runners as the default. Ask GitHub Support for a + higher concurrent-job limit before paying for more capacity. The documented + standard limits depend on the account plan (Free: 20 total/5 macOS; Team: + 60/5; Enterprise: 500/50), and increases are subject to approval. The actual + account entitlement was not verified. See [limits](https://docs.github.com/en/actions/reference/limits). +2. Reserve existing Blacksmith allowance for macOS if that is the priority. + Blacksmith documents 3,000 free x64 2-vCPU-equivalent minutes per organization; + a 6-vCPU Mac minute consumes 20 equivalents, or 150 actual Mac minutes if + it uses the entire free pool. Cloud workflows also use Blacksmith Linux. + Moving Linux to hosted GitHub saves shared allowance, but does not necessarily + free Mac hardware capacity. Account-specific contracts and usage were not + inspected. See [Blacksmith runners](https://docs.blacksmith.sh/blacksmith-runners/overview). +3. Treat Ubicloud as an optional small Linux overflow trial. Its documented + $2.50 monthly credit buys 1,250 premium 2-vCPU minutes at $0.002/minute, or + 2,000 standard 2-vCPU minutes at $0.00125/minute. New accounts default to + premium and require a credit card. No enforceable hard spending cap was + verified, so changing runner labels cannot guarantee the no-spend constraint. + One PR's roughly 55–65 runner minutes also makes clear how small this pool + is relative to repository activity (hardware speeds differ). + See [pricing](https://ubicloud.com/docs/about/pricing) and + [setup](https://ubicloud.com/docs/github-actions-integration/quickstart). + +## Machines that also run coding agents + +Do not register the credentialed host directly as a public-PR runner. A PR can +execute arbitrary build/test code, and a persistent host lets it access local +credentials or affect subsequent jobs. Docker alone is not adequate isolation +when it exposes the host home, Docker socket, SSH agent, or office network. + +A possible no-new-hardware experiment is a disposable VM per job, preferably on +a dedicated spare machine, with a just-in-time single-job runner, no shared +home/keychain/SSH agent or host mounts, restricted network access, and CPU/RAM +limits that leave room for coding agents. Destroy the VM after every job; +ephemeral runner registration by itself does not clean the machine. Start with +trusted branch/manual workloads and keep public fork PRs on hosted runners. +Provisioning and ongoing patching are real operational costs even when the +machine is already owned. See GitHub's +[self-hosted runner security guidance](https://docs.github.com/en/actions/security-for-github-actions/security-guides/security-hardening-for-github-actions). + +## Release waits + +The latest successful sampled Windows release used 13m59s of a 21m56s job in +signing wait/download steps. The same release held an Ubuntu job for 11m38s +polling the isolated Mac build. These are stronger occupancy opportunities than +small checkout savings, especially when approval takes hours. + +[Windows signing without occupying a runner](windows-signing-runner-time.md) +describes a staged, same-run design, required protected environments, and +rehearsal criteria. No callback integration or protected Windows signing +environments currently exist. An environment-gated design adds a GitHub +approval after each SignPath approval and changes the current automatic inner +signing timeout fallback; those are explicit release-policy decisions, so this +PR leaves production signing behavior unchanged. diff --git a/docs/reference/windows-signing-runner-time.md b/docs/reference/windows-signing-runner-time.md new file mode 100644 index 00000000000..fb02ba6bdca --- /dev/null +++ b/docs/reference/windows-signing-runner-time.md @@ -0,0 +1,137 @@ +# Windows signing without occupying a runner during approval + +Status: implementation proposal; production signing behavior is unchanged. + +## Measured cost + +In [release run 33821033674](https://github.com/stablyai/orca/actions/runs/33821033674) +(September 4, 2026), the Windows job took 21m56s. The inner-binary download step +took 13m19s and the installer download step took 40s: 13m59s, or 64% of the job, +was spent in the signing download/wait steps. These durations include the +download itself, so they are an upper bound on removable idle time, not a +prediction of net savings after transferring state between jobs. + +`release-cut.yml` submits both requests with `wait-for-completion: false`, but +then invokes `Get-SignedArtifact` on the same Windows runner with one-hour and +four-hour completion timeouts. The six-hour job timeout accommodates both +waits. Changing the submission flag again, polling less often, or running the +wait inside a container does not release the runner slot. + +This is runner occupancy, not a billing estimate. Standard GitHub-hosted +runners in a public repository may be free; removing the waits still releases +concurrency for other work. Check actual billing before assigning dollar savings. + +The same release also occupied an Ubuntu runner for 11m38s while +`run-release-mac-build-workflow.mjs` waited on the isolated macOS workflow. +That is a separate orchestration optimization. Windows development-channel +builds deliberately ship unsigned and have no SignPath wait to remove. + +## Proposed execution graph + +Keep all Windows stages in the original `release-cut.yml` run to preserve the +current SignPath GitHub artifact provenance boundary: + +1. `build-windows` builds and uploads the unpacked app, original installer, + updater metadata, and inner-signing manifest. It submits the inner request, + sends the existing notification, exposes the request ID, and finishes. +2. `package-windows` depends on that job and uses a protected environment named + `windows-inner-signing`. Its runner is allocated only after GitHub approval. + It restores the exact build, downloads the signed binaries with a short, + bounded completion wait, applies the existing signature restoration and + signed `elevate.exe` cache replacement, builds the NSIS installer, uploads it, + submits the second signing request, notifies approvers, and finishes. +3. `finalize-windows` depends on packaging and uses a second protected environment + named `windows-installer-signing`. After approval it downloads the signed + installer, regenerates its blockmap and `latest.yml`, runs existing outer and + inner signature checks, uploads evidence, and uploads the assets to the draft. +4. `publish-release` depends on finalization as well as the existing Linux, macOS, + and blocking release gates. It remains the only job that publishes the draft. + +The approver signs in SignPath, waits for that request to finish, and then +approves the corresponding pending GitHub job. Each notification should link +to both places and explain the order. GitHub approval is an extra action; +approving in SignPath alone does not release an environment gate. + +## Required configuration + +The repository environments were inspected through the GitHub API on +September 5, 2026. Neither Windows environment exists. `adhoc-mac-build` has no +protection rules; it cannot be reused as an approval gate. No SignPath callback +handler was found in the repository's workflows, scripts, application, or cloud +code. + +Before enabling the graph: + +1. Create both environments in repository Settings → Environments. +2. Add the release approvers as required reviewers for each environment. Decide + whether a release initiator may approve their own job, and configure that + consistently with the existing SignPath policy. +3. Restrict deployment branches to the trusted refs used to dispatch release + workflows, and check that the release workflow's ref passes the restriction. + The workflow ref and the checked-out release tag are different concepts. +4. Read back both environments through the API and verify that + `required_reviewers` rules exist before changing the release graph. Merely + referring to a new environment name in YAML can create an unprotected + environment and silently leave the wait on the runner. +5. Add a preflight assertion for those rules so accidental removal fails before + any signing request is submitted. Verify the API access required for this + assertion using the release workflow's token; do not assume an administrator's + local `gh` access proves workflow-token access. + +An automatic alternative requires a SignPath completion callback and an +authenticated integration that releases the corresponding deployment gate. +Confirm the Foundation plan supports the necessary callback before choosing +that architecture. Do not introduce a long-running GitHub polling job as the +callback substitute: it would continue occupying a slot. + +## State and failure contracts + +- Use artifacts from this exact run and attempt, with a manifest containing the + tag, tag commit SHA, workflow SHA, request IDs, artifact IDs, and SHA-256 hashes. + Artifact names alone are insufficient. Preserve the original unsigned + installer for the existing inner-signing fallback. +- Restore `dist/win-unpacked`, the staging list, the installer, and updater + metadata as one checkpoint. Use an archive to preserve the tree. Do not ship + a fresh rebuild of the app after approving a different binary tree. +- Each new Windows runner needs the pinned Node/pnpm toolchain, build + dependencies, SignPath module, and electron-builder tool cache. The second + runner must populate the NSIS cache before replacing `elevate.exe`; the old + code assumes the first installer build already populated that cache. +- Retain checkout-from-tag behavior and the existing support for release tags + that predate the composite action. Explicitly restore new orchestration code + from the workflow SHA when necessary. +- Preserve the rule that rerunning a workflow never submits a new signing + request. A resume must consume the recorded request and artifacts. Test failed + stage reruns, whole-workflow reruns, and missing/expired checkpoints separately. +- Keep installer signature checks blocking. Keep inner verification evidence + and its current warning-only policy unless changed in a separate decision. +- Resolve the current one-hour inner-signing fallback deliberately: an + environment approval can remain pending longer than one hour and rejection + skips dependent jobs. It cannot reproduce the existing automatic timeout + fallback by itself. A first migration should explicitly document the new + manual release/cancellation behavior; silently treating rejected approval as + permission to ship is not acceptable. +- Keep the release-wide concurrency lock while the graph waits, preventing + another release from overtaking this draft. This saves worker occupancy, but + does not shorten the serialized release queue's human approval time. + +## Validation before production + +First adapt `windows-signing-rehearsal.yml` to exercise the same staged code +using the auto-approved test-signing policy. Then run a manual rehearsal with +the protected environments and confirm that pending approval has no allocated +Windows runner. Verify signed bytes through the existing extraction-based +installer checks, not only the outer installer signature. + +Cover approval before SignPath completion, rejected approval, missing signed +files, changed checkpoint hashes, lost checkpoints, expired artifacts, failed +packaging, and stage reruns without duplicate submissions. Confirm no release +becomes public until all platform and signature gates pass. Compare transferred +artifact/setup time with the original 13m59s wait sample to measure net savings. + +A separate `workflow_dispatch` continuation can avoid environment provisioning, +but changes this design substantially: the original release run finishes, +workflow-level concurrency no longer protects the pending draft, and SignPath +must accept artifacts assembled from a prior run. That option needs a durable +release state machine and provenance validation before production use; it is +not a drop-in replacement for the two download steps. From 71f2c5d3f9bd29d13c93c43b6a09105648001cea Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 17:15:57 -0700 Subject: [PATCH 19/23] test: keep artifact share fixtures unexpired across calendar dates (#18955) --- src/main/artifacts/artifact-cloud-recovery.test.ts | 2 +- src/main/artifacts/artifact-cloud-service-races.test.ts | 2 +- src/main/artifacts/artifact-cloud-service.test.ts | 5 ++++- 3 files changed, 6 insertions(+), 3 deletions(-) diff --git a/src/main/artifacts/artifact-cloud-recovery.test.ts b/src/main/artifacts/artifact-cloud-recovery.test.ts index f57b53c2b02..5690a37b94c 100644 --- a/src/main/artifacts/artifact-cloud-recovery.test.ts +++ b/src/main/artifacts/artifact-cloud-recovery.test.ts @@ -336,7 +336,7 @@ function createResponseBody(slug: string): object { renderedContentType: 'text/html', createdAt: '2026-08-06T00:00:00.000Z', updatedAt: '2026-08-06T00:00:00.000Z', - expiresAt: '2026-09-06T00:00:00.000Z', + expiresAt: new Date(Date.now() + 30 * 24 * 60 * 60 * 1000).toISOString(), byteSize: 17, deletedAt: null }, diff --git a/src/main/artifacts/artifact-cloud-service-races.test.ts b/src/main/artifacts/artifact-cloud-service-races.test.ts index c31c2a23e3e..8606dce4bec 100644 --- a/src/main/artifacts/artifact-cloud-service-races.test.ts +++ b/src/main/artifacts/artifact-cloud-service-races.test.ts @@ -33,7 +33,7 @@ function createResponse(slug: string): Response { renderedContentType: 'text/html', createdAt: '2026-08-06T00:00:00.000Z', updatedAt: '2026-08-06T00:00:00.000Z', - expiresAt: '2026-09-06T00:00:00.000Z', + expiresAt: new Date(Date.now() + 30 * 24 * 60 * 60 * 1000).toISOString(), byteSize: 12, deletedAt: null }, diff --git a/src/main/artifacts/artifact-cloud-service.test.ts b/src/main/artifacts/artifact-cloud-service.test.ts index 75da3922fc8..8a02478feb2 100644 --- a/src/main/artifacts/artifact-cloud-service.test.ts +++ b/src/main/artifacts/artifact-cloud-service.test.ts @@ -43,7 +43,10 @@ const cloudB: OrcaProfileCloudSummary = { linkedAt: 2 } -function createResponse(slug = 'artifact-a', expiresAt = '2026-09-06T00:00:00.000Z'): Response { +function createResponse( + slug = 'artifact-a', + expiresAt = new Date(Date.now() + 30 * 24 * 60 * 60 * 1000).toISOString() +): Response { return new Response( JSON.stringify({ artifact: { From 3bb038a1851922b75ab15e7e4b1631e11a36f32f Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:20:59 -0400 Subject: [PATCH 20/23] docs(relay): 2026-09 reconnect findings, improvement checklist, roadmap, and Roll 2 plan (#18958) Operator record for the 2026-09-04 relay reconnect incident and the Roll 1 same-cap cell image roll (complete 2026-09-05, selector gen 148), plus the follow-up checklist, roadmap, and the Roll 2 implementation plan. Docs only; split out of #18565 so the record merges independently of the code. --- .../relay-improvement-checklist-2026-09.md | 189 ++++ .../docs/relay-improvement-roadmap-2026-09.md | 67 ++ .../docs/relay-reconnect-2026-09-findings.md | 991 ++++++++++++++++++ cloud/docs/relay-roll2-plan-2026-09.md | 154 +++ 4 files changed, 1401 insertions(+) create mode 100644 cloud/docs/relay-improvement-checklist-2026-09.md create mode 100644 cloud/docs/relay-improvement-roadmap-2026-09.md create mode 100644 cloud/docs/relay-reconnect-2026-09-findings.md create mode 100644 cloud/docs/relay-roll2-plan-2026-09.md diff --git a/cloud/docs/relay-improvement-checklist-2026-09.md b/cloud/docs/relay-improvement-checklist-2026-09.md new file mode 100644 index 00000000000..91f1cc742ef --- /dev/null +++ b/cloud/docs/relay-improvement-checklist-2026-09.md @@ -0,0 +1,189 @@ +# Relay improvement: implementation checklist, lanes, and disruption + +Companion to [`relay-improvement-roadmap-2026-09.md`](./relay-improvement-roadmap-2026-09.md) (item numbers +match). This file answers three questions per item: what are the concrete steps, what can run in parallel, +and will a user notice. + +## Status as of 2026-09-04 22:30Z + +Three buckets. "Merged" means the code is on `main` and nothing in production has changed yet. "Deployed" means users are already getting it. "Awaiting owner" means I will not touch production without a go. + +**Deployed to production** +- Auth instance cap 20 + dead-family audit fix (orca-cloud #474) as revision `orca-cloud-auth-00031-tox`. +- Dynamic NAT ports in both regions (stablyai/orca #18693). Zero drops and zero proxy dial errors since. +- Nine alert policies with log metrics: 4 auth (#475), 3 relay Cloud SQL/NAT (#18693), 1 cell process-exit (#18717), all on the relay Slack channel. + +**Merged, ships with the next relay cell image roll (Roll 1 carries `519f4914`; Roll 2 needs a fresh image build)** +- Per-cell inventory locks, delta counters, pool `statement_timeout` (#18722). Roll 2. +- Cells dial Cloud SQL with `--private-ip` when configured (#18720). Inert until 2.1 applies. +- Phone shows a clear "sign in on the desktop again" state when the desktop is signed out (#18698). + +**Merged, ships with the next auth deploy** +- Refresh rotation grace window (orca-cloud #478). Startup adds one nullable column (brief exclusive lock on `refresh_tokens`). +- Pruning job code (orca-cloud #476) is in the image; the job itself is Terraform-disabled until 1.2. + +**Merged, ships with the next desktop release** +- Never replay a refresh token after a timeout; ±10 % jitter on relay lease renewal (#18719). +- Renderer learns when a cloud session is revoked (#18694). + +**Merged, not applied** +- Incident dashboard (#18717) blocked behind the runtime-metric label drift (5.x first item). +- Monitor probe fix (#18723) is live in the workflow; the same-cap roll gate has not yet produced a green dry-run since. + +**Awaiting owner go (production mutations)** +1. Roll 1 cell image roll (1.1): dry-run gate, then c8 canary, then batches. +2. Auth deploy carrying #478 (3.1): quiet minute for the column add. +3. orca-cloud #477 private IP (2.1): merge arms an instance restart and a one-way door. Recommendation: hold. +4. Runtime-metric `region` label drift (5.x): intentional replacement of 21 metrics, or drop the label. +5. Enable pruning (1.2): first budget 20k rows; needs a Terraform apply. +6. Paging channel for auth alerts (5.2): needs the destination from you. + +**Open code follow-ups (no gate, nobody assigned)** +- Monitor summary Markdown does not render `tolerated: true` continuity events (added by #18798); the state artifact has them, the checkpoint table does not. +- Relay container boot races the `cloud-sql-proxy` sidecar: c13's fresh container exited twice (`applyPostgresSchema` connection timeout, 2 s each) before the proxy was listening. Make schema apply wait for the proxy or order the containers. +- `cloud-deploy-relay-production-capacity-job.yml` (~line 416) has the same wave-0 single-shot preflight carve-out that #18778 removes from the same-cap job; its single-evidence path never retries freshness-only failures. +- `cloud/package.json` `test` names every dev-script test file explicitly; an unregistered `*.test.mjs` is silently never run in CI (found by #18769). Needs a glob or a ratchet that fails on an unlisted test file. +- Same-cap job's verify step uses bare `curl --fail-with-body` against the just-rolled cell; one 503 at the LB warm-up edge failed c8 canary #2 (run 33935407461) after the transition verifier had already passed. Needs a bounded retry, same rule as #18723/#18740. +- `verify-mutation` in `cloud-deploy-relay-production.yml`, the multi-target workflow, and the capacity workflow still binds to an exact commit; same exposure #18754 fixed for the same-cap and rehome paths. +- `incident-live-preflight-cli.ts` reports only `source/code` (`active-probe/threshold_max`) with no signal name or observed value, so a failed mutation preflight (c27 recovery #3, run 33986948522) cannot be attributed to an endpoint without an out-of-band probe. Print the signal and observed/threshold pair. Related: the 2 000 ms `endpointLatencyMs` bar is shared by US and Asia cells while Asia /health round trips from a US runner sit at 0.7–1.3 s idle; consider a per-region bar or the p50 of the gate window instead of one shot. Gates #44 and #45 (2026-09-05) both froze on `cell.production-gce-c27.latency_ms` at 2.6–2.7 s with c28 showing the identical tail under operator probes; the bar is now blocking Asia rolls. **Fix: stablyai/orca #18877** (per-region `cellEndpointLatencyMs`, us-central1 2 000 / asia-east2 4 000, plus signal/observed/threshold in preflight messages). Residual: `probeEndpointHealth` in `resource-inventory.ts` still uses the flat 2 000 bar to decide whether to retry after the 10 s readiness-cache wait, so a healthy Asia cell over 2 s costs one extra probe per sample (latency, not verdict); thread the region bar into the retry decision. +- The root oxlint config ignores `cloud/**`, so `check:code-quality:changed` never inspects relay-ops or the cloud dev scripts; typecheck + vitest is the only gate there. +- Monitor bars that froze on non-health today: `directorInstancesMin: 5` with `latest-sum` (one-minute instance recycle), `endpointLatencyMs: 2000` on a US-runner probe to asia-east2, `cloudDataMaxAgeMs: 180000` vs Cloud Monitoring publish lag up to 255 s. Recalibrate with a week of data. +- `parsed()` in `resource-inventory.ts` still returns null on a 200 with a malformed MIG body; a second path to `runtime_power_unknown`. +- Deploy script strips `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` on every release (3.1 first item). +- `assignOnce` placement lock still global (4.1 remainder). +- Region preference (4.2), retries-bar recalibration after a week of Roll 2 data (4.4), pruner `stopReason` alert (1.5). +- Full apps-root apply for 4 unrelated drifts (1.4), from a host with the 1Password account. + +## Uplift ranking (reliability gained per unit of effort) + +| Rank | Item | Why it ranks here | +|---|---|---| +| 1 | 1.1 cell image roll | Removes the only crash mode we have seen in production. 22 of 23 cells still have it. One afternoon. | +| 2 | 3.1 refresh rotation grace window | Turns the entire "slow auth → mass sign-out" class into a slowdown. One day. | +| 3 | 4.1 inventory lock contention | The floor under every 503 and slow phone accept, every day, not just incidents. One week. | +| — | 2.2 relay/auth database split | **Deferred 2026-09-04** to ~2026-11-01. Biggest structural fix, but the concrete cause is fixed and alerts now page; see roadmap 2.2 for re-open triggers. | +| 4 | 1.2 + 1.3 pruning and reclaim | Defuses the 63 M-row time bomb. Low effort, mostly waiting. | +| 5 | 5.1 + 5.2 crash alert, page a human | Cheapest detection uplift; today's incident ran 4 h unpaged. | +| 6 | 2.1 private IP | Durable version of a fix that already landed (dynamic NAT ports). Do it on the existing instance. | +| 7 | 4.3 + 3.2 desktop hardening | Small, ride the normal desktop release. | +| 8 | 4.2, 4.4, 5.4, 1.4, 1.5 | Housekeeping and quality-of-life. | + +## The shared bottleneck: cell rolls + +Every change to what runs on a cell (image, proxy flag, env, relay code) needs a same-cap roll: drain → +recreate → verify, one wave at a time, gated by the 15-minute monitor, about an afternoon. Each wave forces +the desktops on that cell to re-dial (c7 canary: 807 controls re-dialed in ~10 s) and phones on those +desktops reconnect on their normal retry. Users see a few seconds of "reconnecting" per wave. + +So batch. Two rolls, not five: + +- **Roll 1 (now):** current image only (1.1). Do not wait for anything else. +- **Roll 2 (week 2–3):** proxy `--private-ip` (2.1) + relay pool `statement_timeout` (2.3) + lock-contention + fix (4.1), all in one image/template. Prerequisite: 2.1's peering and private IP exist first. + +## Lanes (independent; different people can own them) + +``` +Lane A data plane 1.1 roll ──────────────────► Roll 2 (2.1 flag + 2.3 + 4.1) ──► 4.4 recalibrate +Lane B auth/DB 1.2 enable pruning ──(10 d)──► 1.3 reclaim 3.1 grace window (any time) +Lane C network 2.1 peering + private IP ─────┐ (feeds Roll 2) (2.2 DB split deferred) +Lane D desktop 3.2 no same-token retry, 4.3 lease jitter (any release; wire-compatible) +Lane E observability 1.5, 5.1, 5.2, 5.4 (Terraform only, any time) +Lane F director 4.2 region preference (Cloud Run deploy, any time) +Misc 1.4 full apps-root apply (any time; see its check) +``` + +Hard dependencies: Roll 2 waits on 2.1's network work; 1.3 waits on 1.2 finishing. Everything else is +independent. (2.2 deferred; if revived, do it after 2.1 so the new instance is private from day one.) + +## Disruption summary + +| Item | User-visible? | What they see | Mitigation | +|---|---|---|---| +| 1.1 / Roll 2 | **Yes, transient** | Per wave, desktops on that cell reconnect within seconds; phones follow on retry. | Waves gated by the monitor; run in the US night. Already rehearsed on c7. | +| 1.2 pruning | No | Background deletes, 5k rows per batch. | Small first budget; watch `stopReason` and Cloud SQL write throughput. Stop the scheduler if checkpoint alerts fire. | +| 1.3 reclaim | **Depends on tool** | `VACUUM FULL` takes an exclusive lock on `refresh_tokens`: sign-in and refresh block for its duration (minutes to tens of minutes on 16 GB). `pg_repack` holds only brief locks. | Use `pg_repack`. If VACUUM FULL, announce a maintenance window. | +| 1.4 full apps apply | Should be none, **verify** | Terraform will create a new auth revision (env added). Traffic is pinned to `00031-tox` by name, so the new revision should receive 0 %. | Confirm in the plan that no `traffic` change appears. If it does, stop: the Terraform image variable is not the serving image. | +| 1.5, 5.x alerts | No | | | +| 2.1 private IP | **Yes, certain** | Google: "Configuring an existing Cloud SQL instance to use private IP causes the instance to restart, resulting in downtime." No in-place path, HA does not avoid it. Expect 1–2 min DB unavailability: sign-in fails, relay renewals retry. **One-way door**: private IP cannot be disabled and the VPC link cannot be removed once set. The proxy flag change rides Roll 2. | Off-peak; only after Roll 1 (old image dies on a 2 min DB blip). Owner decision required before the foundation apply. | +| 2.2 DB split (deferred) | **Yes, scheduled** | Relay unavailable for the cutover (drain all cells → copy relay tables → flip `DATABASE_URL` → restart). Minutes if rehearsed. Desktops and phones reconnect automatically after. | Rehearse on staging; do it in the US night; announce. | +| 2.3 statement timeout | No beyond Roll 2 | | | +| 3.1 grace window | No | Auth deploys are no-traffic candidate → smoke → promote. | Security trade-off: a stolen token replayed inside the window is served once instead of revoking. 60 s is the usual choice. | +| 3.2, 4.3 desktop | No | Normal app update. | | +| 4.1 lock fix | No beyond Roll 2 | | Verify against real Postgres on 55440 with concurrent probes before shipping. | +| 4.2 region preference | **Minor, Asia users** | Phones that start being placed in Asia reconnect once to a nearer cell. | Roll out behind the existing region-preference flag. | +| 4.4 | No | | | + +## Checklists + +### 1.1 Cell image roll (Roll 1) +- [x] Confirm fleet is quiet: 15-min monitor dry-run passes. #19 green 23:07:53Z (run 33927238469). Canary then failed the evidence provenance check because main moved during the gate; re-gating with a same-commit chain. +- [x] Confirm director is on 519f4914 and c7 on 85bf6799 (confirmed 2026-09-04 via instance-template census; 20 serving cells still on `5aedbca5`) (`verify` mode of the same-cap workflow). +- [x] Dispatch `cloud-deploy-relay-production-same-cap` waves per the plan in the findings doc; one wave, verify, next. Done 2026-09-05 01:14Z–22:27Z: c8 canary, US batches c9–c10, c13–c16, c19–c26 at protocol 1, then Asia c27 (recovered via `mode=rollback` re-entry after gate freezes on the flat latency bar, fixed by #18877), c28, c29 as single-cell canaries at protocol 0. +- [x] After each wave: the transition verifier passed at migration-only and again at general on every cell (assignments carried, heartbeat fresh, hard cap 3 000); no `container die` fleet-wide across the whole roll. The 4408/1006 burst per wave was not measured separately; the verifier's assignment count before and after each restart is the recovery evidence recorded. +- [x] Record image census in the findings doc. 2026-09-05 22:27Z: all 19 general cells on `519f4914` except c7 on `85bf6799`; existing-only c1–c6, c11, c12 and migration-only c17, c18 untouched on their older images by design. Selector at gen 148. + +### 1.2 Enable pruning +- [x] `auth_token_pruner_image` = digest of `orca-cloud-auth-00031-tox` (`343a0915…`; it contains the entrypoint). orca-cloud #479 merged. +- [x] `auth_token_pruner_enabled = true`, `auth_token_pruner_max_rows_per_run = 20000` for the first day (orca-cloud #479). +- [x] Targeted plan asserted 9 create / 0 change / 0 destroy. Applied 2026-09-05 02:06Z. +- [x] Trigger one run by hand; read the summary event. 02:18Z: `time-budget`, 73 batches, 365k scanned, 1 040 deleted (1 021 revoked, 19 expired), no errors. Scan-bound. +- [ ] Raise the budget to the default 200k after a clean day; watch Cloud SQL write MB/s and the checkpoint alert. +- [ ] 1.5: log metric + policy on `stopReason != complete`. + +### 1.3 Reclaim +- [ ] Wait for steady-state runs deleting ~0 rows. +- [ ] `pg_repack -t refresh_tokens` off-peak (needs the extension; check `pg_available_extensions`). Not `VACUUM FULL` without a window. +- [ ] Confirm table + index size and `disk/utilization` dropped. + +### 1.4 Full apps-root apply +- [ ] Run from CI or a host with the 1Password account (local plan fails on the Cloudflare data source). +- [ ] Plan shows exactly the four known drifts and **no traffic change** on `google_cloud_run_v2_service.auth`. +- [ ] Apply; confirm `status.traffic` still pins `00031-tox` at 100 %. + +### 2.1 Private IP (PRs open: orca-cloud #477 foundation, stablyai/orca #18720 relay flag) +- [ ] **Owner decision**: the foundation apply restarts the instance and is irreversible on Google's side. Merging #477 arms the next foundation apply; hold the merge until the window is chosen. +- [ ] Director is out of scope: it uses the Cloud Run built-in connector (managed Google path, not the relay VPC NAT), so it consumed none of the exhausted ports; moving it needs Direct VPC egress + a separate DSN secret. Own PR if ever wanted. +- [ ] Step 7 (`ipv4_enabled=false`) is blocked until humans have IAP/bastion access and the director is moved; it breaks both today. +- [ ] Allocate a `/24` private services range on the relay VPC; `google_service_networking_connection`. +- [ ] Add `ip_configuration.private_network` to `google_sql_database_instance.auth` (foundation root). Plan must show update, not replace. +- [ ] Apply off-peak; expect a possible restart. Watch auth 5xx alert and relay `sqlFailures`. +- [ ] Cell template: proxy args add `--private-ip` (code merged #18720; flag not set). Director: Direct VPC egress or connector, then the same flag. Both ride Roll 2. +- [ ] After Roll 2: NAT `port_usage` for relay gateways drops to ~0; then consider `ipv4_enabled = false` (removes the public IP; breaks the local `cloud-sql-proxy --token` workflow unless it also goes private). + +### 2.2 Database split (deferred to ~2026-11-01; checklist kept for when it is revived) +- [ ] New `google_sql_database_instance.relay` (private IP from day one, its own size and flags). Staging first. +- [ ] Relay schema applies cleanly to an empty instance (it does at startup). +- [ ] Rehearsal on staging: drain → `pg_dump` relay tables → restore → flip `relay_database_url` secret → restart director + cells → phones/desktops reconnect. Time it. +- [ ] Production: announce a window; same steps; verify `orca_relay_runtime_metrics` controls recover to pre-cutover count. +- [ ] Update `production-cloud-sql-app-consumers` budget test and both alert policies' `database_id`. + +### 2.3 Relay pool statement timeout (merged stablyai/orca #18722; ships Roll 2) +- [x] `statement_timeout` on the relay `pg.Pool` (5 s, env-configurable; schema pool untimed; `57014` retryable), below the control-renewal deadline; DDL on an untimed connection (same pattern as auth #476). +- [x] Postgres test on 55440: a held lock fails the query fast and the bounded retry takes over. + +### 3.1 Refresh rotation grace window (orca-cloud #478 merged 2026-09-04; deploy pending owner go) +- [ ] Fix the deploy-script env strip for `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` (pre-existing; found by #478). +- [x] `rotateRefreshToken`: if `rotated_at` within 60 s and not revoked, return the existing successor (idempotent), no revoke, no audit. +- [x] Outside the window or a third presentation: unchanged (revoke + audit). +- [x] Tests: replay inside window returns same successor; outside revokes; concurrent double-present yields one successor. +- [x] Deploy via `deploy-auth-production` (candidate → smoke → promote). Deployed 2026-09-04 23:15Z as `orca-cloud-auth-00035-gos`, cap 20 kept, 0 5xx; `successor_material` column present; sealed successors being written. (candidate → smoke → promote). + +### 3.2 / 4.3 Desktop (merged stablyai/orca #18719; ships next desktop release) +- [x] 3.2: on refresh timeout, re-read stored session before retrying; do not re-send a token already rotated locally. +- [x] 4.3: ±10 % jitter on control lease renewal; unit test on the distribution; wire-compatible (server accepts early renewals already). + +### 4.1 Lock contention (partial: stablyai/orca #18722 merged; ships Roll 2) +- [x] Replace the global `FOR UPDATE` over `relay_cells` with per-cell row locks; counters delta-only. Remaining: `assignOnce` placement lock is still global (optimistic snapshot follow-up). with per-cell row locks or `pg_advisory_xact_lock(cell)`; counters delta-only. +- [x] Postgres tests on 55440 with concurrent probes (in #18722). Staging load run still owed; `postgres_retries` per hour drops in staging load run. +- [ ] Ships in Roll 2; then 4.4 recalibrates the retries bar from a week of data. + +### 4.2 Region preference +- [ ] Director: honor requested region when the preferred region has headroom, else sticky. Behind the existing flag. +- [ ] Measure with `orca_relay_runtime_metrics` region counters before/after. + +### 5.x Observability +- [x] **Relay-root runtime-metric drift**: resolved by dropping the `region` label to match live state (stablyai/orca #18734). Applied 2026-09-04 23:11Z: 8 never-applied `control_*` renewal metrics + the incident dashboard created, 0 destroyed, 21 live metrics untouched. +- [x] 5.1 `container die` log metric per cell (`relay_cell_process_exit`, applied 2026-09-04 via #18717), > 3 / 15 min, relay channel. +- [ ] 5.2 Add a paging channel (**needs owner input**: destination) to `auth_alert_notification_channels` for refresh rejections + latency. +- [x] 5.4 One dashboard (applied 2026-09-04 23:11Z): `orca_relay_cloud_sql_wal_checkpoint`, NAT drops, `orca_auth_refresh_401`, summed `controls`. diff --git a/cloud/docs/relay-improvement-roadmap-2026-09.md b/cloud/docs/relay-improvement-roadmap-2026-09.md new file mode 100644 index 00000000000..64f33c69a70 --- /dev/null +++ b/cloud/docs/relay-improvement-roadmap-2026-09.md @@ -0,0 +1,67 @@ +# Relay improvement roadmap (written 2026-09-04, after the auth/relay outage) + +Owner-facing list of what is left to make the relay more robust, in priority order. Evidence and history +for every item is in [`relay-reconnect-2026-09-findings.md`](./relay-reconnect-2026-09-findings.md) +(Findings 1–13). Everything already landed on 2026-09-04 is listed at the end so this file is complete on +its own. + +## 1. Finish what 2026-09-04 started (this week) + +| # | Item | Why | How | Size | +|---|---|---|---|---| +| 1.1 | **Roll all 23 cells onto the current relay image** | Every cell still runs the image that exits the whole process on a Postgres connect timeout (Finding 6). The fixed image runs only on the director and c7. Any future DB stall repeats the 200-crashes-in-48h pattern. | `cloud-deploy-relay-production-same-cap` waves, gated by the 15-min monitor. Roll inputs and canary results are in the findings doc ("Roll inputs", "Canary blast radius"). | one afternoon | +| 1.2 | **Enable the refresh_tokens pruning job** (orca-cloud #476, merged, off) | `refresh_tokens` is 63 M rows / 26 GB and grows forever; its size is what turned a slow disk into a sign-out storm (Finding 13). | Build an auth image from main (the 21:04Z deploy already contains the entrypoint: `orca-cloud-auth-00031-tox`, digest `343a0915…`), set `auth_token_pruner_enabled = true` and the image digest in `infra/terraform-apps/environments/production.tfvars`, apply targeted. First run with a small `auth_token_pruner_max_deleted_rows`. Watch the run summary's `stopReason`, not the exit code. ~48 M rows drain in ~10 days at 200k/hour. | 1 hour + 10 days of watching | +| 1.3 | **Reclaim the disk after pruning** | Deletes leave dead tuples; the 16 GB table does not shrink on its own. | `pg_repack` (or `VACUUM FULL` in a maintenance window; it takes an exclusive lock) on `refresh_tokens` off-peak, after 1.2 finishes. | 1 evening | +| 1.4 | **Full Terraform apply of the orca-cloud apps root** | The production plan carries four drifts from other merged work: `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` env on the auth service (#476), a skill-share log exclusion filter change, skill pressure threshold 16→8, an artifacts bucket lifecycle rule. Locally it also fails on the 1Password Cloudflare data source. | Run from CI or a machine with the 1Password account; review the four drifts as ordinary changes. | 30 min | +| 1.5 | **Alert on the pruning job** | A run that only ever times out exits 0 and reads as green. | Log metric on the job's summary event where `stopReason != "complete"`, policy on the relay channel. | 1 hour | + +## 2. Remove the shared fate between auth and relay (2.1 and 2.3 this quarter; 2.2 deferred) + +| # | Item | Why | How | Size | +|---|---|---|---|---| +| 2.1 | **Private IP for Cloud SQL, `--private-ip` on the cell proxies** (do this on the existing shared instance; do not wait for 2.2) | Cells reach the database's public IP through Cloud NAT. Dynamic port allocation (landed) raised the ceiling from 64 to 4096 ports per VM, but the NAT is still in the path and its logs are still the only place port exhaustion shows up (Finding 11). | Add a private IP to `orca-cloud-auth-db` (foundation root, orca-cloud), peer the relay VPC, switch the proxy flag in the cell template, roll. | 1–2 days | +| 2.2 | **Split the relay database from the auth database** — *DEFERRED 2026-09-04 (owner decision): revisit ~2026-11-01 once pruning is done and there is a month of alert history* | One Cloud SQL instance serves `orca_auth`, `orca_relay`, `orca_push`, `orca_skills`. The auth table's growth stalled the relay for a day (Findings 10, 13). Deferral rationale: the concrete cause is fixed (disk 250 GB, WAL 16 GB, index, pruning), 2.3 + 1.1 turn a future stall into retries, and the checkpoint/disk/headroom alerts now page. Re-open if the checkpoint-loop or connection-headroom alert fires, or a large new auth-side table is planned. | New instance for `orca_relay`; migrate with a short relay drain. Relay state is small so the cutover is minutes. | 1–2 weeks incl. rehearsal on staging | +| 2.3 | **Statement timeouts on the relay pool** (the auth pool got one in #476) | A relay query stuck behind a checkpoint fsync should fail fast and let the bounded retry take over rather than hold a pool slot for seconds. | `statement_timeout` on the relay `pg.Pool` in `cloud/apps/relay`, tuned under the lease renewal deadline. | half a day | + +## 3. Make the desktop refresh path forgiving (next 2 weeks) + +| # | Item | Why | How | Size | +|---|---|---|---|---| +| 3.1 | **Refresh-token rotation grace window** | The server revokes the whole family the first time a just-rotated token is presented again. On 2026-09-04 that turned a 30 s server slowdown into 21,605 sign-outs. A short window (e.g. 60 s) where the immediately-previous token is still accepted, returning the same new token, is standard practice. | In `apps/auth/src/tokens/refresh-tokens.ts`: accept `rotated_at` within the window, return the successor instead of revoking. Keep true reuse (outside the window, or a third presentation) as revocation. | 1 day incl. tests | +| 3.2 | **Do not retry `/refresh` with the same token on timeout** | Desktop's 30 s `CLOUD_REQUEST_TIMEOUT_MS` expiring is treated like a network error and retried with a token the server may already have rotated. | In `src/main/orca-profiles/profile-cloud-session-refresh.ts`: on timeout, re-read the stored session first, and prefer a longer single attempt for the refresh call specifically. | half a day | +| 3.3 | **Un-revoke is impossible; make sign-out recovery obvious instead** | Server-side un-revoke does not help because the desktop deletes its local token on the 401. Landed: desktop notices immediately (#18694) and the phone says "desktop signed out" (#18698). | Nothing more unless we want a re-auth deep link from the phone to the desktop. | — | + +## 4. Chronic relay issues already characterised + +| # | Item | Why | How | Size | +|---|---|---|---|---| +| 4.1 | **Cell-inventory lock contention** (partial: PR #18722 narrowed the remaining non-placement sites; `assignOnce` placement lock is the follow-up) | `postgres_retries` is a global `FOR UPDATE` over the 23-row `relay_cells` table with a 1 s `lock_timeout`; it is the floor under every 503 and every slow phone accept (Findings 2, 5; memory `relay-cell-inventory-lock-contention`). | Per-cell row locks or an advisory lock keyed by cell; move capacity counters to delta writes. Verify against real Postgres on 55440. | 1 week | +| 4.2 | **Region preference is mostly inert** | Phones request an Asia cell on ~19 % of attempts and get one ~6 % of the time; the sticky lane wins silently, so Asia users ride the US path more than intended (memory `relay-region-preference-mostly-inert`). | Let a region preference override stickiness when the preferred region has headroom; measure with `orca_relay_runtime_metrics` region counters. | 2–3 days | +| 4.3 | **Desktop lease-rotation waves** | A cell recreate seeds a fleet-wide 1006/4408 reconnect burst ~54 min later, every ~54 min (Finding 3). | Jitter the desktop control lease renewal by ±10 % so the cohort spreads out. | half a day, desktop + wire-compatible | +| 4.4 | **Raise `postgres_retries` gate calibration** | The 300 bar was recalibrated (PR #18580) but should track the post-lock-fix baseline once 4.1 lands. | Re-derive from a week of `orca_relay_postgres_transaction_retry` counts. | 1 hour | + +## 5. Observability still missing + +| # | Item | Why | How | +|---|---|---|---| +| 5.1 | **Cell crash-rate alert** | 201 process exits in 48 h with no page (Finding 6). | Log metric on `container die` for `resource.type="gce_instance"` relay cells, > 3 per 15 min per cell. In `cloud/infra/terraform/relay-observability.tf`. | +| 5.2 | **Page a person for auth alerts** | Today's four auth policies (orca-cloud #475) route to the relay Slack channel only. A repeat of 2026-09-04 deserves a page. | Add a PagerDuty/phone notification channel to `auth_alert_notification_channels` for refresh rejections and latency. | +| 5.3 | **Pruning job alert** | See 1.5. | | +| 5.4 | **Dashboard that puts the four signals side by side** | Diagnosis took hours because checkpoint state, NAT drops, auth 401 rate, and fleet controls live in four consoles. | One Cloud Monitoring dashboard: `orca_relay_cloud_sql_wal_checkpoint`, NAT `dropped_sent_packets_count`, `orca_auth_refresh_401`, summed `controls`. | + +## Landed on 2026-09-04 (for completeness) + +- Auth service cap 2 → 20 (service-level manual scaling removed); Cloud SQL disk 49 → 250 GB PD-SSD; + `max_wal_size` 16384; partial index `refresh_tokens_family_unrevoked` built concurrently by hand. +- orca-cloud #474: the above in Terraform + deploy workflow; replayed dead token answers 401 without + re-revoking or re-auditing. Deployed as `orca-cloud-auth-00031-tox` 21:04Z. +- orca-cloud #475: auth alerts (refresh 401 > 100/5 min, 429 > 20/5 min, 5xx > 10/5 min, p99 > 10 s). Applied. +- orca-cloud #476: batched `refresh_tokens` pruner (disabled), auth pool `statement_timeout` 10 s, schema + DDL on an untimed connection. +- stablyai/orca #18693: both relay NATs on dynamic port allocation 64..4096 (applied US 21:01Z, Asia 21:05Z); + alerts for Cloud SQL WAL-checkpoint loop, disk > 70 %, NAT `OUT_OF_RESOURCES` drops. Applied. +- stablyai/orca #18694: desktop learns of a revoked session immediately, panes re-fetch on mount, pairing + notice says "Sign in again to use Orca Relay". +- stablyai/orca #18698: phone shows "Desktop signed out — sign in to Orca on your desktop to reconnect" via + the WebSocket close reason (only additive slot old phones tolerate). +- Director on image 519f4914; c7 on 85bf6799; other 22 cells still on the old image (see 1.1). diff --git a/cloud/docs/relay-reconnect-2026-09-findings.md b/cloud/docs/relay-reconnect-2026-09-findings.md new file mode 100644 index 00000000000..426a120c251 --- /dev/null +++ b/cloud/docs/relay-reconnect-2026-09-findings.md @@ -0,0 +1,991 @@ +# Relay reconnect investigation: findings and evidence + +Working notes for the 2026-09-04 mobile relay reconnect incident and the cell roll that follows. +Kept current across context compactions. Newest section first. All times UTC. Host ids are log digests, +never raw ids. Nothing here is a production mutation record unless the "Mutations" section says so. + +## Status board + +| Item | State | Where | +|---|---|---| +| PR #18565 relay accept abandonment + lease jitter + desktop rotation spread + phone probe fail-fast | Open, CI fully green again after the doc move (05:45Z), CodeRabbit + Pullfrog cleared, 3 review rounds; not merged (owner has not asked) | https://github.com/stablyai/orca/pull/18565 | +| PR #18569 monitor `relayPostgresRetryExhausted` 0 -> 300 | **Merged** 2026-09-04 ~04:20Z as 4101505b6b | https://github.com/stablyai/orca/pull/18569 | +| Same-cap `verify` of c7 (read-only) | **Passed** run 33836527159 | confirms identities, selector gen 110, rehome gen 12, protocol 1, digests | +| Monitor dry-run #1 | Froze min 5: `relay.postgres_retries` 380 > 300 | run 33836470590 | +| Monitor dry-run #2 | Green to min 13, froze 04:49Z: `director.concurrency` 76.7 > 64 (six-cell crash storm, Finding 6) | run 33837160275 | +| Monitor dry-run #3 | Froze min 3 at 05:01Z: `relay.postgres_retries` 339 > 300; no crash, concurrency 5–8 | run 33838698725 | +| Owner decision 2026-09-04 ~05:10Z | **Option B approved**: "you can raise the bar. or remove it altogether ... whats the most logical move". Kept the bar (removal would leave contention unwatched during the roll) and recalibrated from measured data. | this thread | +| PR #18580 monitor `relayPostgresRetries` 300 -> 2000 | Open, awaiting CI; mutation-checked (300 fails the new test) | https://github.com/stablyai/orca/pull/18580 | +| PR #18565 CI | Was red on `root directory guard` because this findings file sat at repo root; moved to `cloud/docs/` in 8ebff89106 | | +| PR #18580 | **Merged** 2026-09-04 05:23Z as 79d5fb469a (Pullfrog cancelled by the merge; independent Opus review requested instead, per owner) | | +| Monitor dry-run #4 | Froze min 12 at 05:37:35Z: `cell.production-gce-c27.health`/`.ready` = 0. Retries green all 12 samples under the new 2000 bar. Cause: c27 (asia-east2) container died 3x 05:37:00–05:38:01Z, Finding 6 crash class. | run 33840364323 | +| Monitor dry-run #5 | Froze at sample 1 (05:41Z): c27 health/ready still 0. MIG autoheal `recreateInstance` on c27 fired 05:38:12Z after the 3 crashes; instance RECREATING, process up with 0 controls (was ~395). Second c27 recreate in 7 h (Finding 3 seed pattern). Waiting for c27 to settle before dry-run #6. | run 33841327879 | +| Monitor dry-run #6 | **Passed** 06:06:31Z: 16 samples, no freeze (started 05:47:42Z) | run 33841783747 attempt 1 | +| c7 `canary-apply` | **Succeeded.** Dispatched 06:07:15Z; drain 06:10Z; MIG recreate 06:16–06:23Z; new image listening 06:23:42Z; verify + trust proof passed; restored to `admission=general` 06:25:21Z; canary authority sealed. c7 is on `85bf6799…`. | run 33843071283 | +| PR #18581 doc reconcile (Aug 23 figure: 2,200–3,000 raw log lines vs 1,510 on the gate metric) | **Merged** | https://github.com/stablyai/orca/pull/18581 | +| Same-cap `verify` c7 target=519f4914 rollback=85bf6799, gen 112 | **Passed** (read-only) | run 33856355648 | +| Monitor dry-run #7 (gen 112) | Froze at sample 1 (09:05:31Z): `director.errors` 4 > 0, the four 2.0 s pg-connect 500s from the 09:00 cascade still inside the 5-min delta window. Dispatched 4 min too early. | run 33856521278 | +| Monitor dry-run #8 (gen 112) | Green for 15 of 16 samples (09:09:38–09:24), froze on the final sample 09:25:22Z: `director.errors` 1 > 0. The one 500 was `/v1/admin/evacuation-status` at 09:23:50Z, 2.01 s latency = director pg-connect timeout, called by **the monitor's own collector** (`incident-monitor-sources.ts:492`). First evacuation-status 500 since Sep 1. The gate froze on a request it made itself. | run 33856905229 | +| Monitor dry-run #9 (gen 112) | Froze: c13/c23 crashed 50 s after dispatch, then c14/c20/c9 at 09:34. | run 33858650691 | +| Monitor dry-run #10 | Dispatched 09:46:13Z; froze at sample 5 (09:56:59Z): `director.errors` 12. All twelve at 09:55:17–21Z, 0.8–2.1 s latency, 10 on `/v1/regions` + 2 on `/v1/assign`; c16 and c8 crashed at 09:55:19 in the same second. A single 4-second Postgres connect stall hit director and cells together. | run 33859947207 | +| Monitor dry-run #11 | Froze at sample 2 (10:08:07Z): `director.concurrency` 79.8 > 64, the c8/c20 re-dial. They crashed 10:05:54, 3 s before the waiter's quiet check passed (log ingestion lag). | run 33861578009 | +| Monitor dry-run #12 | Dispatched 10:17:38Z after 10 quiet min; froze at sample 2 (10:19:24Z): `cell.production-gce-c16.health` 0. c16 did **not** crash (no container die, MIG NONE/HEALTHY, readiness=true throughout, `/health` 200 in 230 ms at 10:21). At 10:19:07–16 it logged "control activity renewal failed" x4 and a burst of 1006 closes, sqlFailures 1 -> 14, sqlLatencyMsMax 2588: a pg stall on the old image that did not reach the unhandled path. The probe's single fetch (30 s timeout) came back unavailable during that stall and `unavailableIsZero` turned it into health=0. | run 33862504601 | +| Monitor dry-run #13 | Green 14 of 16 samples (10:48:38–11:03), froze 11:04:43Z: c9 crashed 11:04:23, c28 11:04:25 (then looped 11:05:04, 11:05:41); c15 probe also read 0 (stall, no crash). Missed by ~90 s. **Dispatched by hand 10:48:15Z** into a 43-min crash lull (last die 10:05:54; last director 500 10:31:49). The re-armed waiter never fired: its MIG-stable check used `grep -vc True`, which exits 1 when nothing matches, so `&&` short-circuited on the *healthy* case. Waiter armed 10:20Z: 10-min quiet + every MIG stable + 60 s recheck, then dispatch, then canary c7 on green. Held at 10:24 and 10:31 by lone director `/v1/assign` 500s (2 s pg-connect stalls, no cell crash). Director 500 events since 08:46: 6 (gaps 2.7/21/31/29/7.6 min). At 10:39 the waiter was re-armed with a 6-min director-500 window (the monitor's own delta is 5 min) instead of 10, since the gate only needs the 15 min *after* dispatch to be clean. Cell crashes have stopped since 10:05 (33+ min, longest gap since 08:40). 12 dry-runs: 1 pass (#6), 11 freezes, none on a real fleet-health regression. | Cascade gaps since 09:00: 31, 2.9, 5.1, 16.1, 4.0 min (median 5); a 15-min clean window is ~28% per attempt at this rate. | | +| Monitor dry-run #14 | Dispatched 11:26:53Z by the fixed waiter (first autonomous dispatch); c14, c23, c25, c15, c24, c19 died 11:30:59–11:31:08 (six cells, 13 min after the last cascade). Froze on c8 (and others) health/ready probes. Waiter re-armed 11:06Z (grep bug fixed: `grep -c` under `|| true`), same chain; held through the 11:17 cascade and c14/c28 recreates. 13 dry-runs: 1 pass, 12 freezes. Since 08:40: 10 cascades, 75 container dies, gaps 20/31/3/5/16/4/6.5/58/13 min; only 3 windows of >=17 clean minutes existed in 2.6 h, and dry-runs hit two of them (#6 passed, #13 lost the third by 90 s). | +| Monitor dry-run #15 | Waiter armed 11:33Z (6-min director-500 window, 8-min crash window, all MIGs stable), chained canary; still holding at 12:04Z. Since 11:00: 8 cascades, 98 dies, gaps 13/13.6/3.6/14.5/4.4/6.1/3.0 min, **max gap 14.5 min**, so no 15-min clean window has existed in the last hour. 14 dry-runs: 1 pass, 13 freezes. | +| Monitor dry-run #15 verdict | Dispatched 12:28:49Z; froze at sample 2 (12:30:41Z): **12 cells** health/ready = 0 at once (c4, c5, c7, c10, c15, c16, c18, c20, c22, c25, c27, c28), including c4/c5 (0 controls all day, `/health` 200 in 190 ms a minute later) and c7 (new image). Six old-image cells also crashed 12:30:02–21. This was a fleet-wide SQL stall, not a cascade: every cell's `sqlLatencyMsMax` hit 4–6 s (c7 4865, director 5140), director pool waiting 1258, 15 cell pg-connect timeouts, director sqlFailures 92. Cloud SQL CPU 0.73, backends 160, new connections normal, memory 0.46, so the *instance* was not saturated; something held the database for ~5 s. Postgres log 12:31:23–28 shows a burst of `could not obtain lock on row in relation "relay_cells"` from NOWAIT (single-row and full-inventory) sweeps, i.e. the row locks were held during recovery. Cloud SQL transactions/min flat (~30k), reads flat, +network flat: the database was neither busy nor saturated, it was *waiting*. The stall bracket +(12:30:02–12:30:41) is where every cell's SQL max hit 4–6 s at once. Lock retries in that window were +ordinary (49/29/13 per min). Best reading: a ~5 s Postgres-side wait event shared by every session +(lock on a hot row held across a long transaction, or an instance-level pause), not CPU/IO. Cell +`sqlLatencyMsMax` was already 1.5–2.2 s fleet-wide in the four minutes before, i.e. the old cells' 1 s +`lock_timeout` plus queueing. | run 33872946111 | +| Monitor dry-run #16 | Dispatched 12:38:57Z; froze at sample 1 (12:40:11Z): `cell.production-gce-c27.latency_ms` 2071 > 2000, a fifth distinct freeze signal, the probe's own round-trip absorbing a checkpoint sync. **Loop stopped by me at 12:41Z**: with the disk in the checkpoint loop (Finding 10) no bar can hold for 15 min, so further dry-runs only burn the shared rollout lease. 16 dry-runs: 1 pass, 15 freezes. Re-arm after the disk change lands. | +| Cloud SQL checkpoint loop | **Broke on its own 12:39–12:45Z**: disk writes 48 -> 4 MB/s at 12:39 with transactions and network flat and no Cloud SQL operation; 12:40:17 checkpoint synced 0.047 s; 12:45:53 checkpoint was `time`-triggered again (first since 11:55) with sync 0.096 s and write spread over 269 s. Cause of the break unknown (most likely WAL fell back under `max_wal_size` once a burst of full-page writes aged out). It can re-enter the loop on the next large checkpoint; the disk-size fix remains the durable one. | +| Monitor dry-run #17 | Dispatched ~12:49Z (all guards clean); froze at sample 1 (12:52:05Z): `director.errors` 4, from the c9/c22 crash loop that began 12:50:34, ~90 s after dispatch. Checkpoints stayed healthy (85 ms), so this is the old image's baseline crash rate, not the disk. 17 dry-runs: 1 pass, 16 freezes. | +| Monitor dry-run #18 | **Dispatched by mistake 13:48:56Z into the outage**: my gcloud credentials expired ~13:45Z, every guard query returned empty, and the waiter's `grep -c . || true` read empty as "quiet". Froze at sample 1 (13:49:43Z) on `director.ready=0`, `auth.health=0`, and cell probes; no canary dispatched, no production mutation. All waiter loops killed at 13:51Z. Lesson: a quiet-window check must fail closed when its data source errors. Waiter had been re-armed 12:53Z. | +| Gate decision | Owner asked at 09:36Z to choose: A keep looping / B recalibrate `directorErrors` 0 -> small n / C human bypass. Ten dry-runs, four froze on this bar. Recommendation B+A. Note: B alone would not have passed #9 or #10 (cell health probes and a 12-error burst); it fixes the single-500 false freezes (#7, #8) only. | | +| Batch roll | **Deferred by plan**: roll once with the lock-fix image instead of twice. | | +| PR #18606 lock removal (root cause) | **Merged** 09:2xZ as 7b108abf71 after review, fix, re-verify; CI green | https://github.com/stablyai/orca/pull/18606 | +| Image publish for 7b108abf71 | **Done** 08:36:49Z run 33854111305: `sha256:519f4914217f08cabcdcd34825965db8473ec37c6591553a3af0d65dcdeeb183` | | +| Director deploy on 519f4914 | **Succeeded** 08:45Z run 33854355791; serving `orca-cloud-relay-00570-siv`, rollback tag on 00569-ret (also 519f4914), 00565-fes (85bf6799) still deployable. Dispatched 08:37:45Z (blue/green; prior revision 00565-fes on 85bf6799 kept as rollback). Note: `predecessor-image-digest` is a required input even with bootstrap=false; pass the serving digest. | `cloud-deploy-relay-production-director.yml` | +| c7 on new image, 2 h in | 817 controls, **0 container die** since restore (was ~1 per 15 min on old image); `sqlLatencyMsMax` still 1.0 s = lock wait unchanged, which #18606 targets | | +| Terraform alert `relay_postgres_retry_exhausted` at `> 0` | Firing continuously since #18521; recalibration not done (own change) | `cloud/infra/terraform/relay-observability.tf:447,469` | + +## Mutations performed (complete list) + +1. Merged PR #18569 to main (code/docs only). +2. Merged PR #18580 and #18581 to main (monitor bar + docs). +2b. Merged PR #18606 to main (relay lock change; no serving effect until the image is deployed). +2c. Dispatched `cloud-publish-relay-production` for 7b108abf71 (builds and pushes an image; changes nothing serving). Done: 519f4914. +2d. Dispatched `cloud-deploy-relay-production-director` on 519f4914 (preserve placement, no prune, rehome gen 12). Succeeded 08:45Z; serving revision 00570-siv. Rollback: `gcloud run services update-traffic orca-cloud-relay --region us-central1 --to-revisions orca-cloud-relay-00565-fes=100` (85bf6799, still Ready). Not needed so far. +3. 2026-09-04 06:07:15Z: dispatched `cloud-deploy-relay-production-same-cap` `canary-apply` for production-gce-c7 only (run 33843071283). Completed successfully 06:26Z: c7 isolated, drained (807 controls re-dialed), template + MIG rolled to 85bf6799, verified, restored to general admission. Selector generation advanced 110 -> 112 (isolate + restore). +4. Nothing else. Both monitor dispatches were `mode=dry-run` (read-only). The same-cap dispatch was `mode=verify` (read-only, confirmed by step gates `if: inputs.mode != 'verify'` on every mutating step). + +## Finding 6 (2026-09-04 ~05:00Z): the old cell image crashes the whole process on a Postgres connect timeout + +**This is the most important open finding.** The 23 GCE cells run image `sha256:5aedbca5…` = orca-cloud +commit e3e92d95d3 (2026-08-14). In that build `beginProof` is called as `void this.beginProof(...)`. +When `verifyCellAssignment` inside it throws (pg-pool `timeout exceeded when trying to connect`, 2 s +`connectionTimeoutMillis`), the rejection is unhandled and Node exits 1. Docker restarts the container +in ~1 s, but every control on that cell (~800 hosts) drops and re-dials `/v1/assign` at once. + +Evidence, cell c7 instance 4545742188814054238, 2026-09-04: + +``` +04:46:47.951 stderr [orca-relay] control activity renewal failed (x5) +04:46:49.527 stderr Error: timeout exceeded when trying to connect + at pg-pool/index.js:45:11 + at async PostgresPoolPressure.connect (postgres-pool-pressure.js:30:20) + at async PostgresDatabase.query (database.js:645:24) + at async RelayAssignmentStore.verifyCellAssignment (assignment-store.js:2024:22) + at async HostSessionRegistry.beginProof (host-session-registry.js:376:15) +04:46:49.527 stderr Node.js v24.19.0 +04:46:49.835 dockerd: container die … exitCode=1 image=…relay@sha256:5aed… +04:46:50.258 dockerd: container start +04:46:52.761 stdout [orca-relay] listening on https://c7.relay.onorca.dev +``` + +2026-09-04 05:36:59–05:38:01Z: c27 died 3x in 62 s plus one other instance (5464389947731541178); this froze dry-run #4 on c27's health probe. + +Fleet-wide `container die … exitCode=1` on the relay image, last 48 h: **201 events on 19 instances** +(c28 x38, c29 x37, c27 x19). Hourly counts track the lock-contention curve (peak 23/h at 21Z Sep 3). +Every one has the same `Node.js v24…` crash banner. On 2026-09-04 04:46:35–04:47:41Z six cells +(c7, c8, c19, c21, c22, c25) died within 66 s: ~4,800 hosts re-dialed, `/v1/assign` returned 16,321 +503s in one minute (baseline ~20), director concurrency hit 85 (Cloud Run cap 80), Cloud SQL +`new_connection_count` 119 -> 287/min. Fleet recovered by 04:51Z. That is what froze dry-run #2. + +Fix status: `guardSessionTask` wrapping `beginProof` landed in orca-cloud #436 (2026-08-27) and is in +the target image `sha256:85bf6799…` (main 11aace8dec). The roll is the fix. Not caused by anything in +this session: the same-cap verify finished ~04:25Z and never reached a mutating step; no compute +operations exist for those instances; heap/event-loop were flat before the crash. + +Autoheal amplifier: MIG health check is `/health` every 10 s, timeout 5 s, unhealthy after 3, so a +crash loop of ~30 s+ triggers `compute.instances.repair.recreateInstance`. All ~20 recreates in the +48 h to 2026-09-04 05:40Z were the three Asia cells (c27 x6, c28 x7, c29 x8; gcloud prints local +-07:00 times). c27 recreated 05:38:12Z after 3 crashes in 62 s; its ~395 controls went to 0 and the +monitor's `cell.production-gce-c27.health/ready` probe read 0 for the whole recreate (~several min), +freezing dry-runs #4 and #5. Each recreate also seeds a Finding 3 rotation cohort. Rolling the Asia +cells early in the batch phase should be weighed against the canary-first rule; c7 stays the canary. + +Implication for the gate: the monitor's `director.concurrency` freeze is *correctly* detecting these +crash storms. A dry-run only passes in a 15-minute window with no cell crash, roughly 1 in 3 windows +at current rates. Retrying in quiet hours is legitimate; the bar is not wrong. + +## Finding 5: `relay.postgres_retries` at 300 is 3x under today's baseline + +Retries per 5 min, cells + director, last 24 h: p50 579, p90 1039, p99 1398, max 1505; **65% of +windows over 300**. Quiet hours (03–08Z) p50 235, max 512. When the 300 bar was set (2026-08-26) +healthy bursts reached 234. Baseline has roughly tripled in 10 days. Skill notes say do not raise this +bar; I have not. Best odds for a clean 15 min are 02–04Z and 17–18Z (9/12 five-minute windows under +300 in each). + +## Finding 4: exhausted-retry bar was the wrong single blocker (fixed) + +`relayPostgresRetryExhausted: 0` never cleared after #18521 reached the director (22:12Z Sep 3): 236/236 +five-minute windows non-zero; post-#18521 p50 42 / p90 147 / max 220; Aug 23 incident peak 467. +Recalibrated to 300 in #18569 (merged). Dry-run #1 immediately revealed Finding 5 behind it. + +## Finding 3: the 00:50Z control-close wave was desktop lease rotation, not a rollout + +2026-09-04 00:49–00:51Z: 2,745 control closes on 19 instances; 1157/1632 code 1006 and 973/1030 code +4408 `control rebound` had ageMs in the 53-minute bin. Relay grants a flat 55 min lease; desktops +rebind 60–120 s early; so every host that (re)connected in the same minute rebinds as one cohort +forever. Seed: c27 MIG autoheal recreate 23:23Z (`compute.instances.repair.recreateInstance`) dumped +~420 controls. Harmonics at 23:55, 00:04, 00:25, 00:49Z. Each rebind is an `activateControl` +transaction that can take the inventory lock. Fix in #18565: relay lease 55 min ± 5 min (symmetric, +so mean rebind rate unchanged), desktop early window 1–6 min. + +## Finding 2: fleet-wide lock contention, worse on Sep 3 + +| window | 55P03 retries/h (cells) | cell sqlFailures/h | +|---|---|---| +| Sep 2 18Z – Sep 3 07Z | 660–1470 | 680–1620 | +| Sep 3 08Z–16Z | 3600–7100 | 3700–7700 | +| Sep 3 23Z | 7468 | 7585 | + +100% of sampled retries are 55P03; director phase is `cell-inventory`. Every cell pins +`sqlLatencyMsMax` at 1.0–1.2 s = the pre-#18521 1 s pool `lock_timeout`. Not load (controls flat +~26k, Cloud SQL CPU 46–53%). No `cloud-*` workflow explains the 08Z step. The lock is a global +`SELECT * FROM relay_cells FOR UPDATE` (23 rows) taken by assignment, control activation, activity +acquire, and sweeps, held to COMMIT. + +## Finding 1: root cause of the phone's 24 s hang (the original symptom) + +`acceptClient` runs four serialized Postgres calls; the fourth (`acquireActivity`) contends for the +global lock. Under contention the cell finishes after the phone's 12 s bound, then +`PendingHostDataReservation.bind` throws `host_data_reservation_already_bound` because the phone's +close already released the reservation. Every "first frame handler failed already_bound" line is that +post-mortem (31 events 23:06–01:01Z across 12 instances). Fix in #18565: abandon the accept after each +DB step once the socket is closed; new event `orca_relay_client_accept_abandoned {stage, elapsedMs}` +and metric fields `clientAcceptsAbandonedByStageDelta` / `clientAcceptAbandonedMsMax`. Phone side: +direct probe now fails fast on `reconnecting` so relay recovery is not queued behind three doomed +LAN redials (~3.5 s saved per foreground). #18518 (merged, not yet on the phone) covers the +stage-aware dial bound. + +Host 666077865f2e: stable throughout. 4408 rotation 00:27:45Z; 1006 quit 00:52:24Z on old adhoc; +sticky reassignment to c27 00:52:35Z on new build; rotation closes 01:44:55Z and 02:23:15Z with +splices intact. No drain/4404/wrong-cell. + +## Finding 7 (2026-09-04 ~05:10Z): retries bar recalibration basis (PR #18580) + +Chose 2000 over removal. The metric is the gate's own source (`orca_relay_postgres_retries` +log metric, director + cells summed per five minutes, ALIGN_DELTA 300 s): + +| window | p50 | p90 | p99 | max | > 300 | +|---|---|---|---|---|---| +| 2026-09-01 | 56 | 105 | 206 | 456 | 0% | +| 2026-09-02 | 109 | 186 | 294 | 377 | 1% | +| 2026-09-03 | 430 | 924 | 1320 | 1504 | 55% | +| 2026-09-04 to 05Z | 285 | 1012 | 1211 | 1211 | 44% | + +15-minute pass rate, last 24 h: bar 300 -> 22%, 800 -> 66%, 1000 -> 86%, 1500 -> 99%, 2000 -> 100%. +Aug 23 incident on this metric: 1510 then 646 (single windows), so retries no longer separate an +incident from baseline; exhausted (467 vs bar 300; healthy 72 h max 184), director concurrency, +and pool bars carry that role. Note: my earlier "p99 1398 / 65% over 300" in Finding 5 came from +raw log line counts; the metric-based numbers above are what the gate actually evaluates. +Baseline tripled between Sep 2 and Sep 3 with no deploy; still unexplained (Finding 2). + +## Decision needed from the owner (resolved: B) + +The same-cap roll is blocked only by the monitor gate, and the gate is blocked by `relayPostgresRetries: 300` +(Finding 5: 65% of windows breach it; even the 04:55Z quiet window hit 339). Three options: + +- A. Keep waiting for a naturally quiet 15 min. Odds per attempt ~1 in 3 in quiet hours, lower by day. + Each attempt is free and read-only. Could take hours. +- B. Recalibrate `relayPostgresRetries` from measured data, same method as #18569: 24 h p99 is 1398, the + Aug 23 incident ran 2200–3000, so ~1500 clears healthy windows with ~1.5–2x incident separation + (less margin than the exhausted bar had). Overrides the "do not raise" note in the skill facts. + Argument for: the roll being gated is the thing that reduces retries. Argument against: the bar is + doing its job of saying contention is high. +- C. A human dispatches the roll with a different gate policy. Not something I can or should do. + +My recommendation: B, with the number chosen from the table in Finding 5 and the roll following +immediately so the bar can be re-tightened after the fleet is on the 500 ms lock wait. + +## Finding 12 (2026-09-04 13:12Z): **INCIDENT IN PROGRESS. The auth service is at its 2-instance cap and rejecting 90% of desktop token calls with 429; the relay fleet has emptied.** + +Timeline: 13:04–13:06 the old-image cascades and NAT stalls drove ~1,400 desktops to re-dial. Their relay +JWTs (5-min TTL) expired mid-storm, so they hit `orca-cloud-auth` `/v1/desktop/auth/refresh` and +`/v1/desktop/auth/relay-token` together. The auth service is Cloud Run `maxScale=2`, `concurrency=80`, +1 vCPU throttled (`auth_max_instances = 2` in orca-cloud `infra/terraform-apps/environments/production.tfvars`, +applied by `deploy-auth-production.yml`). Both instances pinned at concurrency 85 from 13:02; from 13:07 +Cloud Run's front door returns **429 "no available instance"** (0 s latency, never reaches the container): +12,045 at 13:07, 54,292 at 13:08, 46,025 at 13:08, 42,529 at 13:09. Sep 3 total auth 429s: **0**. +Without a fresh relay token every desktop's `/v1/assign` gets 401 (1,433 distinct hosts 401'd, 0 got 200 +since 13:07) and every cell closes its control with `4401 relay authorization expired`. Fleet controls: +13,375 (12:55) -> 7,633 (13:08) -> **249 (13:12)**, splices 1. Auth container CPU 0.15–0.5, so the cap is +the limit, not the code. Every desktop is now in its refresh-retry loop hammering the same 2 instances: +this is a self-sustaining thundering herd and will not clear on its own. At 13:14Z: fleet **30 controls** +across 23 cells; successful relay-token issuance 5,000–6,500/min until 13:05, then 1,059 / 734 / 733 / +443 / 220 / 214 / 148 / **4** per minute through 13:13; auth 429s 54k -> 25k/min only because desktops +are backing off, not because the service recovered. Note `AUTH_MAX_INSTANCES: 2` is also hardcoded in +orca-cloud `.github/workflows/deploy-auth-production.yml` (lines 33–34), so a redeploy would re-pin it; +change both the workflow env and the tfvars. + +**Immediate mitigation (owner action, not applied):** raise the auth service's max instances. Fastest: +`gcloud run services update orca-cloud-auth --region us-central1 --max-instances 20` (or `10`, matching +the other apps' `max_instances = 10`), then land the same in `auth_max_instances` so Terraform does not +revert it. Auth is stateless behind Cloud SQL (`refresh_tokens` table); backends 210 of 400, so 20 +instances x a small pool is within budget. Also consider the desktop's refresh backoff: it re-dials on +401 immediately with no jitter, so a 429 storm sustains itself. + +**13:51Z status: my gcloud session lost auth at ~13:45Z; all production monitoring from this session is +blind until re-authenticated (`gcloud auth login`, interactive). Last confirmed state 13:40Z: fleet 0 +controls, auth maxScale 2, 7,600 auth 429/min. All autonomous dispatch loops are stopped.** + +**17:19Z–17:21Z MITIGATION APPLIED (owner said "fix it NOW").** State at 17:19Z, four hours in: all 23 +cells at 0 controls, auth 429 ~2,000/min, auth 2xx ~40/min, and the 2xx that got through took 13–28 s +(both instances saturated). Mutation 1: `gcloud run services update orca-cloud-auth --max-instances 20` +created revision `orca-cloud-auth-00018-4jc` (same image `auth@sha256:1710ff6c`, same env/concurrency, +only maxScale 2 -> 20) but the service pins traffic to `00023-qud` **by revision name**, so the new revision +was immediately `Retired` and nothing changed. Mutation 2 (17:21:30Z): `gcloud run services update-traffic +--to-revisions orca-cloud-auth-00018-4jc=100`. Lesson: the auth service's traffic block is name-pinned +(the deploy workflow does an explicit traffic switch), so a bare `services update` never reaches users. +Terraform still says `auth_max_instances = 2`; the next `deploy-auth-production.yml` run will revert this +unless the tfvars and the workflow's `AUTH_MAX_INSTANCES` are changed first. + +## Finding 13 (2026-09-04 17:19Z–18:10Z): **the auth outage is a database problem, not (only) a Cloud Run cap; `refresh_tokens` has 63 M rows and reuse-revokes scan whole families** + +Mutations this window (all online, no restarts, all by hand in project onorca-cloud): +1. 17:19Z `gcloud run services update orca-cloud-auth --max-instances 20` → new revision `00018-4jc`, but traffic is + pinned by revision name so it was `Retired`; 17:21:30Z `update-traffic --to-revisions 00018-4jc=100`. +2. Still 2 instances at 17:31Z: the SERVICE has its own `scaling.maxInstanceCount=2` in **manual scaling mode** + (`run.googleapis.com/maxScale: '2'` on service metadata, set by Terraform `infra/terraform-apps/auth.tf`), which + overrides the revision cap. `--scaling=auto` then `--max 20` at 17:31:45Z. Instances 2→20 by 17:38Z; 429s fell + 6,000/2 min → 60/2 min at 17:36Z and controls briefly reached 11. +3. Then latency, not capacity, became the wall: every refresh took 100+ s inside Postgres (desktop client timeout + is 30 s, `CLOUD_REQUEST_TIMEOUT_MS`), so 20 instances × 80 concurrency filled again with requests nobody was + waiting for, and 429s returned (~1,500/2 min from 17:40Z). +4. 17:27Z Cloud SQL disk 62 GB → 250 GB (IOPS ceiling 1,470 → ~7,500). 18:00Z `max_wal_size` 1.5 GB → 16 GB + (the checkpoint loop: `checkpoint starting: wal` every 45–60 s since 13:06Z). +5. 18:07Z `CREATE INDEX CONCURRENTLY refresh_tokens_family_unrevoked ON refresh_tokens(family_id) WHERE + revoked_at IS NULL` (an earlier attempt with `AND rotated_at IS NULL` was wrong for the revoke predicate; its + invalid remnant `refresh_tokens_family_live` was dropped). + +Evidence: `refresh_tokens` = 63.3 M live tuples, 16 GB table + 10 GB indexes; every refresh inserts a row and +nothing ever deletes (30-day TTL rows are never pruned). Query Insights 17:33–17:39Z: `UPDATE refresh_tokens SET +revoked_at = $1 WHERE family_id = $2 AND revoked_at IS NULL` = 21,000 s of execution per 6 min, ~90–120 k rows +updated per minute; io_time 15,000 s read; pg_stat_activity 180+ backends in `IO/DataFileRead` on that statement, +200 backends total for orca_auth (20 instances × pool max 10). `session-refresh-reuse-detected` audit events per +hour: ~100 all day → 8,805 (13Z), 15,511, 19,486, 24,897, 26,935 (17Z). Mechanism: a desktop's refresh times out +client-side at 30 s, the server had already rotated the token, the desktop retries with the same token, the +server calls that reuse and revokes the family (Bitmap scan on `refresh_tokens_family` + heap filter over every +row the family ever had), then the desktop retries the dead token again, and each retry re-runs the same +full-family scan (already-revoked families short-circuit nowhere). Reuse-detected 401 also **signs the user out** +on the desktop (`isOrcaCloudAuthFailure` → `clearCloudSessionIfUnchanged`), so every user who hit this during the +outage must sign in again. + +Durable fixes (orca-cloud PR in preparation on branch `auth-revoke-only-live-tokens`): `AUTH_MAX_INSTANCES` and +`auth_max_instances` → 20; Terraform disk 250 + `max_wal_size=16384`; the partial index in the schema; an +`already-revoked` short-circuit in `rotateRefreshToken` that skips the family UPDATE and the audit insert. Still +open after that: prune `refresh_tokens` (expired or revoked rows older than N days), a server-side statement +timeout shorter than the desktop's 30 s so the client and server agree on failure, and an alert on auth 429s. + +**19:11Z RESOLVED at the database layer.** `refresh_tokens_family_unrevoked` went valid at 19:11:17Z (build +18:07–19:11, two full table scans of 2.1 M blocks under load). Within 60 s: refresh latency 100 s → 0.1 s, auth 429 +→ 0, active orca_auth backends 200 → 2, checkpoints back on the 5-min timer (`checkpoint starting: time` at 18:35, +18:41, 19:00, 19:11). Director `/v1/assign` returning 200. Fleet controls 0 → 17 by 19:14Z. + +**Residual: mass sign-out.** 19:11–19:14Z: 3,857 refresh 401s from 3,829 distinct IPs, then near zero. Every one is +a desktop whose family was revoked by reuse-detection during the outage; the desktop clears its cloud session on +401 (`clearCloudSessionIfUnchanged`) and stops retrying. Those users must sign in again before the relay sees +them. Fresh `/session` sign-ins: 1, 5, 3 per minute at 19:10–19:12. Recovery of controls is now paced by users +signing in, not by infrastructure. Total `session-refresh-reuse-detected` events 13:00–19:00Z ≈ 100k, against a +~100/hour baseline. +**Affected-user count (19:22Z, from `refresh_tokens`):** 23,318 live token families revoked in the window, +**21,605 distinct users**. Only ~3,800 desktops had seen their 401 by 19:15Z; the rest were closed or asleep +and will find themselves signed out on next launch, so sign-ins will trickle for days. + +**Desktop UX finding (owner's own Mac, 19:22Z):** a revoked desktop keeps showing the account card as +"Connected" and the pairing pane as "Orca Relay: Unavailable" / `relay_control_not_active` indefinitely; the +local trace writes no relay events. Only quit + relaunch surfaced the sign-out prompt, after which sign-in → +relay-token → `/v1/assign` 200 (0.15 s) → working pairing, all within 10 s. Follow-ups: the relay coordinator's +401 path should flip the account card to reconnect-required immediately, and the pairing error should say "Sign +in again to use Relay" when the cause is an auth failure. Announcement wording: "If Relay shows Unavailable, quit +and reopen Orca, then sign in when prompted." + +orca-cloud PR #474 (branch `auth-revoke-only-live-tokens`): caps → 20, disk 250 / max_wal_size 16384 in +Terraform, partial index in the schema, `already-revoked` short-circuit. Do not deploy auth to any environment +with a large `refresh_tokens` before building the index concurrently there. + +**Wave 1 of the roadmap (2026-09-04 21:35Z onward):** five Opus agents in isolated worktrees: 3.1 grace window +(orca-cloud), 4.1+2.3 relay locks + pool timeout, 3.2+4.3 desktop refresh/jitter, 5.1+5.4 observability, +2.1 private IP (plan only, both repos). First back: stablyai/orca PR #18717 (crash alert + dashboard). Its key +finding: cell exits log to `cos_system` with uppercase `jsonPayload.MESSAGE` and `SYSLOG_IDENTIFIER=docker`, +so every earlier `jsonPayload.message:"container die"` count in this doc that read 0 was querying the wrong +field. Verified: 87 exits 12–13Z on the agent's filter, 0 in the last 6 h. Monitor dry-run 33922255205 +dispatched 21:41Z as the Roll 1 gate. +Dry-run 33922255205 froze at 21:46Z on `signal_missing cloud_sql.backends`. Cause: Cloud Monitoring published +no `num_backends` point for the auth instance between 21:40 and 21:46 (every other minute of the last 100 has +one; measured directly via the timeSeries API). A Google-side publish gap, not a database or monitor defect; +the monitor's freeze-on-missing rule is correct. The 12–13Z monitor failures were a different cause (active +probes reading 0 during the crash cascade). Re-dispatched at 21:50Z. +Dry-run #2 (33922844671) froze at 21:52:21Z on `auth.health observed 0` — verdict read from the state.json +artifact, not the log (the log only prints checkpoints). Auth served `/health` 200 continuously, including the +21:52:05 probe. Cause: the probe requires `/health` AND `/ready` on the first attempt; auth has no `/ready` +(404 by design), so every auth sample takes the forced 11 s retry, and on the third sample the retry fetch threw +at the network layer on the runner (no request reached Cloud Run) and `check()` recorded the exception as +health=false. Neither freeze was fleet health. Fix delegated (relay-ops: a thrown fetch is not a reading; auth +does not require `/ready`). **Sequencing constraint for Roll 1:** monitor evidence must be < 5 min old at +canary dispatch, so the owner's go must precede the dry-run, and a green dry-run must be followed by the +canary dispatch immediately. + +stablyai/orca PR #18719 (3.2 + 4.3, desktop): the replay engine was not the refresh function but +`RelayAuthCoordinator.scheduleRetry`, since `shouldRetryRelayConnectionError` treats any non-HTTP error +(including a refresh `TimeoutError`) as retryable and re-reads the same stored token on backoff. Fix: refresh +gets one 60 s attempt; an ambiguous failure (no status line) records the token and blocks re-sending it for +30 s (bounded, not permanent); definitive 5xx gets exactly one retry after re-reading the store; a 401 on an +ambiguously-attempted token logs `orca_cloud_refresh_possible_replay`. Lease renewal gets ±10 % full jitter +(base shrunk so the latest sample stays ≥ 90 s before expiry); server resets the full 55-min TTL on any rebind +(`host-session-registry.ts:736-743`) so early renewal is free. Verified the retry-path claim and both server +cites against main. + +2.1 private IP: orca-cloud PR #477 (foundation: servicenetworking API, /24 peering range 10.42.128.0, private +network on the instance, `prevent_destroy`; real production plan 3 add / 1 in-place change, staging unchanged) +and stablyai/orca PR #18720 (relay: `relay_cloud_sql_private_ip` variable, conditional `--private-ip` in the +cell startup template; default false renders byte-identical to main). Findings that change the plan: Google +states the private-IP change **restarts the instance** with no in-place path, and it is a one-way door (cannot +disable private IP or remove the network link). The director uses the Cloud Run built-in connector, not the +relay VPC NAT, so it never consumed the exhausted ports and is out of scope. Disabling public IP later breaks +the local proxy workflow and the director. #18720 merges (inert); #477 held for owner decision. + +4.1 + 2.3 relay: stablyai/orca PR #18722. Premise correction: #18521 and #18606 had already bounded and +narrowed most of the fleet-wide lock before today; what remained were the sticky-refresh retry (all 23 rows → +the one pinned row), reservation reconciliation (23 → the 2 involved rows), a dead pool-default fallback, and +an absolute counter write (→ delta with capacity guard). Placement (`assignOnce`) deliberately keeps the +ordered inventory lock: least-loaded selection is fleet-wide and dynamic target-only locking previously caused +cross-cell cycles; converting it to optimistic snapshot + conditional delta is the remaining 55P03 floor and a +follow-up. Pool `statement_timeout` was already 5 s but hardcoded; now env-configurable, `57014` added to the +retryable set (it was terminal before), schema DDL on an untimed max:1 pool. Independently re-ran the new and +adjacent suites here against 55440: 66/66. Harness note: 55440 is not idempotent across full runs (2 +pre-existing failures on a second run); reset the schema between runs. Rollout: director first, watch +`orca_relay_postgres_transaction_exhausted` and `cellInventoryHoldMsP95` before cells. + +#18719 first CI run failed only on `windows-host-job.win32.test.ts` (EPERM on temp-dir cleanup), a Windows +PTY test the PR does not touch and which no other recent run failed on; rerun dispatched rather than waved. + +3.1 grace window: orca-cloud PR #478 merged (not yet deployed; deploy is an owner gate because the startup +schema apply adds a nullable column to `refresh_tokens` with a brief ACCESS EXCLUSIVE). Semantics: within +`ORCA_CLOUD_REFRESH_ROTATION_GRACE_MS` (60 s default, 300 s cap, 0 = off) a re-presented rotated token gets the +SAME successor refresh token + a fresh access token, no revoke, no audit, provided the successor is still the +live head. Third presentation / outside window / revoked family: unchanged (revoke + audit). Successor plaintext +is stored sealed (AES-256-GCM, key = HKDF of the predecessor token; the DB never holds the key). Cost stated +plainly: a stolen token replayed inside 60 s is served once instead of tripping detection; DB-read + stolen +predecessor recovers the successor offline until pruned. Rotation now runs in one transaction (proved by a +forced-INSERT-failure rollback test; the 8-way race alone did not kill the non-transactional mutant). Verified +locally 27/27 incl. the Postgres suite against 55440, and CI ran it on PG 16 and 17 (4/4 each, not skipped). +Deploy wiring: env is set by BOTH Terraform and the deploy workflow, with a test pinning all three sources to +one value. **Pre-existing bug surfaced:** the deploy script strips every env var it does not own, so the +Terraform-set `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` (from #476) silently reverts to the compiled default on each +release. Latent only because both defaults are 30. Follow-up: add it to `authEnvironment` + the workflow env. + +Monitor probe fix: stablyai/orca PR #18723. A thrown fetch (DNS/TCP/TLS/8 s abort) is now "no reading" and is +re-asked once after 1 s; only a second throw is `false`. A non-ok HTTP answer is still `false` with no extra +retry. `latencyMs` is the slowest answering round trip, never a sleep. `requiresReady` is per endpoint: auth +(no `/ready` by design) is judged on `/health` + latency; director and cells unchanged. No threshold or rule +touched; `auth.ready` had no consumer. 81/81 relay-ops tests and 9/9 evidence-script tests locally. The monitor +runs at `main` head, so once merged the next dry-run uses it. + +Applying #18717 (22:10Z): the cell-exit log metric `orca_relay_cell_process_exit` is created; the alert policy +raced descriptor propagation (404) and is being retried. **Not applied, deliberately:** the dashboard. Its +targeted plan drags in `google_logging_metric.relay_snapshot[*]`, and that plan is `32 to add, 21 to destroy`: +the Terraform source adds a `region` label to every runtime metric (`EXTRACT(jsonPayload.region)`) which the +live metrics do not have, and a label change on a log metric is a delete+create. Replacing 21 live metrics +resets their history and would blank the 14 existing relay alert policies during the swap. That is +pre-existing drift in the relay root (unapplied since the region work), not something #18717 introduced. It +needs its own reviewed apply in a quiet window, ideally with the runtime-metric replacement acknowledged as +intentional. Dashboard apply waits on that. + +**Wave 1 closed 22:20Z.** Merged: orca-cloud #478 (grace window); stablyai/orca #18717 (crash alert + +dashboard TF), #18719 (desktop no-replay + jitter), #18720 (private-IP flag, off), #18722 (relay per-cell +locks + pool timeout), #18723 (monitor probe fix). Applied to production: cell-exit log metric + alert policy. +Held for owner: orca-cloud #477 private IP (restart, one-way); the dashboard apply (behind the runtime-metric +label drift); the auth deploy carrying #478; Roll 1. Every wave-1 code change now sits on main un-deployed: +the next relay image build carries #18722 + #18723's monitor runs at main head already; the next auth deploy +carries #478. + +**Landing (2026-09-04 20:50Z–21:02Z, owner: "if you are confident the cloud changes are valid, you can land them"):** + +- Merged: orca-cloud #474, #475, #476; stablyai/orca #18693, #18694, #18698. Neither repo has branch + protection or environment reviewers; `verify` / `cloud-verify` green on main after each. +- Applied to production by targeted saved plans (each plan asserted create-only / exact-attribute before + apply, via `terraform show -json`): 4 relay resources (WAL-checkpoint log metric + 3 alert policies), 8 auth + resources (3 log metrics, propagation sleep, 4 alert policies), and the us-central1 NAT + (`enable_dynamic_port_allocation` false→true, ports 64..4096). Google's docs: switching to dynamic does not + break existing connections when max ≥ 1024 and max ≥ old min; only lowering max or reverting to static is + disruptive. asia-east2 NAT deliberately left for after a US soak. +- Not applied: the untargeted apps-root plan also carries 4 unrelated drifts (`ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` + env on the auth service from #476, a skill log exclusion filter change, skill pressure threshold 16→8, an + artifacts bucket lifecycle rule) and fails on the 1Password Cloudflare data source locally. The foundation + root plans clean (disk 250 / max_wal_size already match). Those drifts belong to whoever runs the next full + apps apply in CI. +- `deploy-auth-production` on main 8034955 (run 33919143723) **succeeded 21:04Z**: serving revision + `orca-cloud-auth-00031-tox` at 100%, previous `00018-4jc`, cap 20, smoke passed on both URLs. First 15 min on + the new revision: 31×200 / 1×401 on `/refresh`, max latency 56 ms, no 5xx. The new + `refresh_token_prune_cursor` table exists, so the new schema applied. +- US NAT soak (21:01–21:06Z): 0 drops, 0 proxy dial errors, 0 cell exits, port_usage 11, sqlMax ~1.07 s. + Asia NAT then applied 21:05:28Z from the pre-verified saved plan (same three attributes). The deploy script strips env vars it does not own, so the Terraform + TTL var will not be on the new revision until the full apps apply lands; the auth code defaults to 30 d. +- Terraform locally needs `GOOGLE_OAUTH_ACCESS_TOKEN="$(gcloud auth print-access-token)"`; ADC is stale. + +**Alerting + NAT follow-ups (19:58Z, superseded by the landing block above):** + +- stablyai/orca PR #18693 (`relay-nat-ports-and-sql-alerts`): both relay NATs switch to dynamic port + allocation (64–4096 per VM); new relay-channel alerts for the Cloud SQL WAL checkpoint loop (log metric on + `checkpoint starting: wal`, > 3 per 5 min), Cloud SQL disk > 70%, and NAT `OUT_OF_RESOURCES` drops. No + existing workflow applies these resources; the PR body carries the targeted plan. +- orca-cloud PR #475 (`auth-observability-alerts`): log metrics + policies for auth refresh 401 (> 100 per 5 + min; Sep 3 baseline 20–80 per hour), 429 (> 20 per 5 min; baseline 0), 5xx (> 10 per 5 min), and Cloud Run + p99 latency > 10 s. Production routes to the relay Slack channel. +- Desktop stale auth-status fix: stablyai/orca PR #18694 (`desktop-cloud-session-revoked-status`). Main pushes + an auth-status-changed IPC when a 401 clears the session; panes re-fetch on mount; the pairing notice says + "Your Orca account session expired. Sign in again to use Orca Relay" and hides Retry. StrictMode regression + test verified red on the old guard. Does not help desktops already revoked today (session cleared before + this code); it fixes every future revocation. +- orca-cloud PR #476 (`auth-refresh-token-pruning`): batched `refresh_tokens` pruner as a scheduled Cloud Run + job (revoked rows kept 30 d, rotated rows 60 d against a 30 d TTL, 5k-row batches, 200 ms pauses, persisted + cursor, per-run budget) plus a 10 s `statement_timeout` on the auth request pool with schema DDL on an + untimed connection. Merges cleanly onto #474 and does not need its index (walks the primary key; + EXPLAIN-asserted no seq scan). CI ran the Postgres integration tests for real on PG 16 and 17. Ships + `auth_token_pruner_enabled = false` in both environments: enabling needs an image digest from a build that + contains the new entrypoint. Operating rules once enabled: monitor the run summary's `stopReason` and + `deletedRows`, not the exit code (a run that only ever times out exits 0); ~48 M rows drain in ~10 days at + 200k/hour; deleting them leaves dead tuples, so the 16 GB is not reclaimed without a separate VACUUM FULL or + pg_repack pass, which is its own change. +- Phone-side copy when the desktop is signed out: stablyai/orca PR #18698 (`phone-desktop-signed-out-reason`). + Real path traced: the director resolves the phone to the host's last cell (durable assignment row), and the + cell's `acceptClient` rejects with 4404. The only additive slot every shipped peer tolerates is the WebSocket + close *reason* (relay-hello and resolve schemas are zod strict; a new close code drops old phones off the + host-offline cadence). Desktop closes its control with reason `signed-out` only when the cloud session is gone + (null context after a 401, or explicit sign-out); quit and relaunch stay reasonless. Cell remembers it per + host for the dormant-assignment TTL, forgets on re-auth, and echoes it as the 4404 close reason; phone + renders "Desktop signed out — sign in to Orca on your desktop to reconnect" with the same retry cadence. + Old×new matrix in the PR body; nothing changes for any old peer. Merges cleanly with #18694. + +## What actually blocks the roll now (12:58Z summary for the owner) + +0. **Cloud NAT ports** (Finding 11, found 12:55Z): every us-central1 cell reaches Cloud SQL's public IP + through a NAT with the default 64 ports/VM; port_usage pinned at 64 and 1,514 dropped SYNs to + Cloud SQL:3307 in one 4-min window. This is the 2 s connect stall that kills old-image cells and is + still active after the disk loop broke. Fix: `min_ports_per_vm = 1024` (or dynamic allocation) on + `google_compute_router_nat.relay_gce` in `cloud/infra/terraform/relay-gce-foundation.tf`, targeted + apply; durable fix is a private IP on the Cloud SQL instance. Online, no VM restart. +1. **Cloud SQL disk** (Finding 10): 49 GB PD-SSD saturated since 11:58Z, checkpoint loop, fleet-wide + 4–6 s stalls every ~45 s. Fix: bigger disk and/or `max_wal_size`. Owner: `stablyai/orca-cloud` + `infra/terraform-foundation/database.tf` `google_sql_database_instance.auth` (no `disk_size`, + `disk_autoresize`, or `database_flags` set today, so Terraform is at defaults: 10 GB initial, autoresize + grew it to 49 GB). Add `disk_size = 200` (+ `disk_autoresize = true`) and optionally + `database_flags { name = "max_wal_size" value = "4096" }`; production tfvars are + `infra/terraform-foundation/environments/production.tfvars`; applied by `deploy-production.yml` in + that repo. Online, no restart for disk; `max_wal_size` is also a non-restart flag. Note Terraform + `disk_size` below the live 49 GB would be a destructive shrink, so 200 is safe and 49 is the floor. **This is now the first thing to do**; nothing else can pass a + 15-min gate while it persists, and it is also what is killing the old-image cells several times an hour. +2. **Old cell image** (Finding 6): dies on every stall. Fixed by rolling 519f4914 (canary inputs ready). +3. **Gate policy**: `directorErrors: 0` and per-cell health probes freeze on any single stall. Recalibrate + after 1 and 2, or bypass by hand for the canary. + +## Plan agreed with the owner (2026-09-04 ~06:45Z), in execution order + +Owner: "feel free to improve operations to make things more effective ... continue driving everything e2e +until this process is complete." Owner has had multi-day experiences with cell rolls and does not want a +9-hour sequential roll. + +1. **Lock-removal PR** (root cause). *Status 08:55Z: pushed as branch `relay-single-row-reservation` + (2 commits). Opus adversarial review found one real defect: `acquireActivity` moving a client-chosen + activity id across cells locked the old cell's row before the new one, cycling with placement's + ascending inventory lock (reviewer reproduced it as paired 55P03s on real Postgres; no 40P01 because + lock_timeout == deadlock_timeout == 1 s). Fixed with `lockCellRows` (ordered, 500 ms bound); census now + fails on any inline `relay_cells FOR UPDATE` outside the named helpers. Three-cell Postgres test moves + an activity high->low while the target row is held; 5/5 revert-mutants fail it. 480 SQLite tests + + tsc green. Also fixed a pre-existing test leak (`relay_cell_connection_snapshots`) that made + `assignment-control-supersession-postgres` fail on reruns. Reviewer re-verified 65569be3de: cycle + repro completes in 7 ms (was 1022 ms + paired 55P03); no remaining out-of-order pair in the store; + flagged two evasions in the new census guard, closed in the third commit (whole-statement scan, + covers query() too, mutation-checked with both evasions). Headroom Postgres test's one failure is + pre-existing on main (verified by swapping in main's store).* Make `activateControl` superseded-control cleanup, `acquireActivity` + existing-lease branch, and `changeActivity` use the existing single-row + `adjustCellReservationAtomically` instead of the 23-row `lockCellInventory`. Keep the global lock only + for placement (`resolve`/assignment) and sweeps. Real-Postgres contention test on port 55440. +2. **Faster same-cap rollout workflow.** (a) paced drain instead of `graceMs: 0` so a cell's ~800 hosts + re-dial over minutes, not one second (director cap is 5 x 80 = 400 in-flight); (b) cells in a batch run + in parallel once drains are paced; (c) post-canary batches use a short freshness check instead of a new + 15-min dry-run, since the in-job safety recheck already runs before each drain; (d) job timeout > 75 min. + Target: 22 cells in ~6 batches x ~25 min. +3. **Build image** with (1) merged, then one roll of the fleet with (2). Asia cells c27/c28/c29 first. +4. Re-tighten the monitor retries bar; recalibrate the Terraform exhausted alert. +5. Consider deleting the 55-min control lease rebind entirely (no recorded reason; liveness is the 75 s + watchdog + 90 s activity lease). Separate PR after (1) so its effect is measurable. + +## Faster same-cap rollout: design (step 2 of the plan), from reading the real limits + +What actually bounds parallelism today (measured on the c7 canary, run 33843071283): + +| step | c7 duration | bound by | +|---|---|---| +| prechecks (recheck, backend init, resolve, verify) | 43 s | none | +| isolate + drain + transition wait | 7 min | drain is `graceMs: 0`; `verify-relay-capacity-transition --activity restart-safe` polls until leases drain | +| Terraform template + MIG recreate + wait-until stable | 8 min | GCE recreate; per cell, independent | +| verify new incarnation + trust proof + restore | 1.5 min | none | + +Real constraints: (1) the director is 5 x 80 = 400 in-flight `/v1/assign`; a `graceMs: 0` drain of ~800 +hosts pins it at cap for ~2 min (observed 79.75/84.75 p99). (2) `production-cloud-sql-rollout` lease and +workflow concurrency group serialise the whole run, by design, and the per-cell job shares it via +`holder-key`. Nothing else forbids parallel cells. + +Changes, smallest first: +1. **Paced drain.** `HostSessionRegistry.drain(graceMs)` already sends `drain {graceMs}` and closes each + session after `graceMs`, but the desktop's `handleDrain` re-dials immediately regardless of graceMs + (`relay-origin-pool.ts:150-162`), so graceMs only delays the *close*, not the stampede. Fix on the + cell: stagger the drain *send* across sessions over a window (e.g. 800 sessions over 120 s = ~7/s), + which needs no desktop change and works for every desktop version in the field. New admin body field + `spreadMs` (optional, default 0 keeps today's behaviour); canary script passes `spreadMs: 120000`. + Requires the cell to be on an image with the change, so it applies to batches after the first + post-lock-fix roll, not to this one. +2. **Parallel cells in a batch.** In `cloud-deploy-relay-production-same-cap.yml` make `cell_2..cell_4` + `needs: [gate]` instead of chaining, gated on the same evidence (drop the `+75 min x wave-index` + allowance, it exists only because of chaining). Each job already takes the rollout lease with the + run's `holder-key`, so they re-enter it rather than fail. With paced drains, 4 cells x ~800 hosts + over 120 s is ~27 dials/s, well under the director cap. Raise `timeout-minutes` to 90. +3. **Post-canary batches skip the 15-min dry-run.** The in-job "Recheck aggregate SQL, pool, + reconnect, migration, and selector safety" step (`pnpm incident:relay-preflight`) already runs a + live one-shot check before each drain. For `batch-apply` with a sealed `canary-run-id` from the + same commit, accept a dry-run of any age (the canary's) plus that live recheck; keep the 15-min + requirement for `canary-apply`. Change lands in `relay-monitor-evidence.mjs verify-authority` + + `relay-production-same-cap-wave.mjs` + their node:test suites. + +**Correction after reading the cell job (07:35Z):** (2) parallel cells is not a flag flip. Each cell job +asserts the exact selector generation `expected + 2 x wave-index` and exact memberships derived from +predecessors having completed (`ISOLATED_*`/`RESTORED_*` in the job, `applyExactAdmissionSelector` +compare-and-swap), and all cells share one Terraform state lock. Making that concurrent means a batch-level +isolate/restore in the gate and a rewrite of the 650-line job's expectations. That is the multi-day trap +the owner described. Deferred. + +What is cheap and removes most of the wall-clock: (3). The per-batch 15-min dry-run costs 15 min each +*and* fails ~50% of the time on old-image crashes, which is where hours go. Implement: `batch-apply` with a +verified canary authority accepts a passed dry-run up to 6 h old and may re-use one already consumed +(the consumed-marker check exists to stop replaying stale evidence; the canary binding plus the in-job +live preflight at drain time replace it). Files: `relay-monitor-evidence.mjs` (`--after-canary`), +`incident-live-preflight-cli.ts` (same flag), the same-cap workflow + job, and both test suites. +Revised expectation: 22 cells = 6 sequential batches x ~70 min = ~7 h wall-clock but *unattended-safe* +and with one dry-run total, versus today's 6 dry-runs at ~50% each. (1) paced drain rides the lock-fix +image. + +## Recommended next steps (superseded by the plan above; kept for history) + +1. Resolve the gate decision above, then: monitor dry-run -> c7 `canary-apply` only -> verify -> stop. + Each rolled cell leaves the Finding 6 crash class. +2. Merge #18565; publish; a later same-cap roll carries it to cells. +3. Remove the global inventory lock from per-connection paths (`acquireActivity` existing-lease + branch, `activateControl` superseded-control cleanup, `changeActivity`) by using the existing + `adjustCellReservationAtomically` single-row update. Own PR, after the roll. +4. Recalibrate the Terraform alert `relay_postgres_retry_exhausted` to 300/300 s (observability root). +5. Whether to raise `relayPostgresRetries` is a human call; the data is in Finding 5. + +## Canary blast radius (read before dispatching c7) + +- What `canary-apply` does to c7, in order: isolate (selector -> migration-only, no new + assignments), `/v1/admin/drain graceMs:0` (every control on c7 re-dials the director and is + reassigned), Terraform template + MIG update to the target image, wait stable, verify new + incarnation + exact digest + protocol, prove per-host trust, restore c7 to general admission. + On any failure c7 is left isolated (migration-only) with rehome disabled; nothing else is touched. +- c7 at 05:20Z: 788 controls, 5 splices, 800 connections. So ~790 desktops re-dial once. The fleet + already absorbs this exact event 201 times / 48 h uncontrolled (Finding 6); the controlled version + isolates first, so no new assignment lands on c7 mid-roll. Expect a director concurrency blip, not + a freeze-class one (six cells at once gave 85; one cell should stay well under 64). +- Precedent: the identical workflow (pre-move, in orca-cloud) ran 9 successful `apply` canaries and + batches on 2026-08-27 (last: c20 -> 5aedbca5). Its failures that day all stopped at the read-only + "Recheck aggregate SQL..." or "Require durable rehome disabled" step, before `MUTATION_STARTED`. + The moved copy in this repo has one run: the read-only `verify` of c7 (passed, including WIF auth). +- c7 side note: MIG autoheal recreated the c7 instance four times on 2026-09-01 08:02-08:42 PDT + at ~13 min spacing. Same crash class as Finding 6 (health check failing during restart loops). + +### Canary observed effect (c7 drain, 2026-09-04 06:10Z) + +- c7 807 controls -> 0 between 06:08:52Z and 06:10:52Z. Director `/v1/assign`: 200s 32 (06:09) -> 2628 (06:10) + -> 340 (06:11); 5xx 1969 (06:10) -> 31 (06:11). Director max-concurrency p99 7.9 -> 79.75 (06:10) -> 84.75 + (06:11), i.e. at the Cloud Run cap of 80 for ~2 min. My pre-dispatch estimate ("well under 64") was wrong. +- Confounder: c10 (us-central1, instance 2803000337345335589) crashed 06:09:56Z on the old-image class + (Node.js banner + container die), so ~1,600 hosts re-dialed in the same minute, not ~800. Coincidental; + the fleet has one of these every ~15 min. +- Recovery: 06:13 903 / 06:14 1471 assign 200s from 640 distinct desktop IPs; 503s 78 -> 183 -> 29/min. + No cell crash 06:12–06:16Z. Drain step passed ~06:16Z; template/MIG apply started. +- 06:16:03–06:17:08Z, during c7's template apply (not its drain): c27 (x4) and c29 (x3) crash-looped on the + old-image pg-pool connect timeout in `beginProof`, both MIGs autoheal-recreated (c27's second recreate in + 40 min). Fleet 23 -> 21 reporting cells, controls 13286 -> 12462, assign 503s 1000/min at 06:17, director + concurrency p99 74.8. Cloud SQL CPU 0.70 max, backends 174 max (bar 250). Same multi-cell pattern occurred + at 01:31Z (4 cells) and 04:47Z (5 cells) with nothing rolling; the c7 drain's SQL load 6 min earlier may + have nudged the pool timeouts but the class is pre-existing. c7 MIG RECREATING onto new template + `…20260904061618…` = the expected image swap. +- 06:20Z: 849 assign 503s. Closes 06:19:30–06:21: 162x1006 age<5min (hosts bouncing off the recreating + c27/c29), 73x4408 + 53x1006 in the 50-min age bin (Finding 3 rotation cohort). Not roll-caused. + c7 MIG `recreating=1` on the new template since 06:16:18Z; c27 and c29 MIGs also RECREATING (autoheal). +- 06:23:16Z c7 instance restarted in place (MIG RECREATE keeps name/id relay-c7-bwjc / 4545742188814054238), + pulled `relay@sha256:85bf6799…` 06:23:37Z, listening + readiness true 06:23:42Z. Apply step passed 06:24Z; + verify step running. Isolate -> ready on new image took ~14 min end to end. +- Post-restore c7 on new image (06:25:42–06:26:42Z): controls 143 -> 273 -> 377 refilling, sqlQueries + ~1,500/30 s, `sqlLatencyMsMax` 518 -> 1003 -> 1155 ms, still 55P03 `cell-inventory` retries. So the new + image alone does not remove lock waits; the request-path 500 ms cap from #18521 applies to the director's + paths, and cell-side `acquireActivity`/`activateControl` still ride the global lock (step 3 in next steps). + Watch: does c7's sqlLatencyMsMax settle below the old 1.0–1.2 s pin once refill finishes, and does c7 stop + appearing in `container die` (the real win: guardSessionTask). +- 08:25Z (2 h after restore): c7 817 controls, 0 crashes since 06:25Z. Fleet crashes last 2 h: c27 x6, + c28 x5, all old-image Asia cells. The new image stops the crash class as predicted; it does not move + lock latency (c7 sqlLatencyMsMax 1005 ms), which is #18606's job. +- Implication for the batch phase: every drain will push director concurrency past the monitor's 64 bar + for ~1-2 min. The batch job rechecks safety *before* it drains (read-only step), so that is fine per wave, + but never run a monitor dry-run concurrently with a wave, and prefer batches of 2 over 4 until the fleet + is on the new image and the crash class is gone. + +## Post-merge dispatch plan for #18606 (image -> director -> cells) + +1. `gh workflow run cloud-publish-relay-production.yml --ref main -f mode=publish` (after the squash lands + on main). Resolve the digest by tag, never by parsing the log (it mixes relay and fence-broker digests): + `gcloud artifacts docker images describe us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay:sha- --format='value(image_summary.digest)'`. +2. Director: `gh workflow run cloud-deploy-relay-production-director.yml --ref main -f image-digest= + -f regional-placement-mode=preserve -f prune-incompatible-revisions=false -f expected-rehome-generation=12 + -f bootstrap-runtime-identity=false -f predecessor-image-digest=` + (no monitor evidence needed; requires rehome disabled at gen 12, which it is). Last run 33826514754 used + the same shape. Watch director `orca_relay_postgres_transaction_retry` per minute before/after. +3. Cells: same-cap `verify` c7 with target=, rollback=85bf6799; fresh dry-run; `canary-apply` c7; + then batches (3 per batch, Asia c27/c29/c28 first). Each batch: new dry-run unless the batch-reuse + change (design section above) has shipped. + +## Finding 8 (2026-09-04 08:40Z): ten-cell crash cascade during the director deploy, not caused by it + +Timeline: candidate revision 00570-siv created 08:38:39Z, first log 08:39:20Z; traffic still 100% on +00565-fes through 08:43 (assign logs by revision). Cell crashes: c28 (5031087219978409220) looped 08:37:55– +08:40:07 (9x), then at 08:40:20–08:40:45Z **ten** instances died within 25 s (c10 2803…, 5110…, 532…, 5464…, +7536…, 7726…, 8671…, 8928…, 8966…). All old-image `beginProof` pg-pool timeouts. Fleet controls 13,423 -> +6,157 by 08:43; assign 503s 3,912 (08:42) and 4,624 (08:43) per minute, director concurrency 85 (cap 80), +Cloud Run autoscaled 5 -> 10 instances, Cloud SQL CPU 0.55 -> 0.99. Deploy finished cleanly at 08:45Z with +the new director taking the tail of the storm; by 08:46 503s were ~30/15 s, controls 7,913 and rising, +director lock retries 29/min (vs 105–157/min pre-deploy) and exhausted 2/min (vs 65/min at 08:36). +Same class as 01:31Z (4 cells) and 04:47Z (5 cells) today; this was the biggest. c7, on the new image +since 06:25Z, did not crash. What triggered the pool timeouts fleet-wide at 08:40 is not established; Cloud +SQL CPU was 0.78–0.88 in the minutes before, the highest of the day, so the cells' 2 s connect timeout is +the plausible tipping point under a busy database. Every cell still on 5aedbca5 remains exposed to this. + +## Finding 9 (2026-09-04 08:56Z): #18606 on the director cut lock retries ~10x + +`orca_relay_postgres_retries` per 5 min, director only: 08:21–08:41 windows 419–689 (old image, incl. the +crash storm); 08:46/08:51/08:56 (new image 519f4914, refilling ~7k hosts): **61 / 69 / 54**. Exhausted: +104–178 -> **11 / 14 / 12**. Inventory hold p95 ~200 ms, max 255 ms, ~366 holds/min. Cells (still old +image) 17–44 -> 0–3, because the director no longer holds the 23-row lock on their behalf. This is the +first direct measurement of the root-cause fix under real load. Cloud SQL CPU peaked 0.99 during the +cascade and is decaying (0.86 at 08:55); the monitor freezes above 0.80, so no dry-run until it clears. + +Fourth cascade 09:00:12–09:00:18Z: c23, c8, c16, c26, c22 (five cells, 11 container-die events in 6 s, +all `5aedbca5`, exitCode 1, Node banner, pg-pool `client closed the connection` burst right before). Cloud +SQL CPU 0.84 -> 0.78 in the preceding minutes, director concurrency 18–22 (idle), so this one fired +*without* a database or director spike. Fleet had just recovered to 13,015. Cadence today: 01:31 (4), +04:47 (5), 08:40 (10), 09:00 (5), 09:31 (c13, c23), 09:34 (c23 again, c14, c20, c9; c14/c20 crash-looping), +09:39 (c21, c24), 09:55 (c16, c8), 09:59 (c20), 10:05 (c8, c20), 10:19 (c16 stalled, no crash), then a 58-min +lull, 11:04 (c9; c28 died 13x in 4 min, autoheal recreate 11:09Z, its 3rd recreate today), 11:17 (c10, c28 +again, c22, c23, c14 x9 looping; 23 dies in ~90 s; fleet 13.3k -> 10.8k), 11:31 (c14, c23, c25, c15, c24, c19), 11:34 (c20, c26, c29 x4, c14, c27 x3, c25; fleet 13.1k -> 10.3k). +Three cascades in 17 min. 11:38–11:45 c27 crash-looped 17x and c28 4x (Asia cells), c29 recreating. +11:59 (c21, c9, c10, c23), 12:02 (c19; 4,109 assign 503s that minute, mostly hosts bouncing off the +recreating cells, code 1006 age<5min x217), 12:09–12:12 (c19, c27 x6, c28 x5, c13, c22, c15, c26, c14; +8 cells, c27/c28 recreating again). Cloud SQL CPU 0.62–0.85 through it. 12:20 (six more cells). Cascade +cadence since 11:00 is now ~every 8 min; the waiter has held correctly the whole time and there has been +no dispatchable window. Loop continues unattended; findings stop logging each cascade from here unless the +class changes. Every cell that has died today is on 5aedbca5; c7 (85bf6799, +5.5 h) has not. Cell dies per hour today: +01Z 5, 02Z 7, 03Z 4, 04Z 7, 05Z 4, 06Z 9, 07Z 11, 08Z 30, 09Z 26, 10Z 2, 11Z 68+ (to 11:42). +Director concurrency pinned at 85 for 09:32–09:33; 503s 4,141 and 4,396 per minute. 09:39: c21, c24 +(2,870 503s). Crashes per instance 08:10–09:40Z: c28 x14, c27 x5, c23 x5, c22 x4, c14 x4, c13/c20 x3, +then c26/c9/c24/c16/c8 x2. Mean gap between cascades since 08:40: ~12 min. Every 15-min gate attempt +now has well under even odds; the c7-style canary that ends this needs a gate it can pass. The old image is now cascading roughly hourly regardless of load; the +only cell on a fixed image (c7) has 0 crashes in 2.5 h across all four. + +Director 500s: 4 in the 09:00 window, all 2.0 s latency on `/v1/assign` or `/v1/resolve` = pg-pool connect +timeout surfacing as a 500. Pre-existing (Sep 3: 03h/08h/16h one each, same 2.0 s shape; 06:09Z today on +the old image during the c7 drain). The monitor's `directorErrors: 0` bar freezes on any of these, so a +dry-run needs a 15-min window with none; at ~1 per cascade that is a real but modest constraint. + +**Gate observation (09:26Z):** `directorErrors: 0` counts every non-503 5xx on the director, including +the monitor's own admin calls. The director on 519f4914 still sees an occasional 2.0 s pg-pool connect +timeout (~1 per 20 min under today's Cloud SQL load), which surfaces as a 500 on whichever request drew +it. Two consecutive dry-runs (#7, #8) froze on exactly this: one, isolated, 2 s 500. That bar was set for +"unexpected director 5xx"; a single connect timeout that the client retries is not an incident. Candidate +recalibration (own PR, not done): `directorErrors` 0 -> 2 per 5 min, or exclude the monitor's own +user-agent. Not changing it unasked; noting that at ~3 per hour the 15-min gate passes ~1 in 2 attempts. + +**Did the director deploy make cells crash more? (checked 09:45Z)** Cell `container die` per 30 min: +06:00 9, 07:00 2, 07:30 9, **08:30 30** (director candidate 08:38, traffic 08:43–08:45; the 10-cell burst +was 08:40:20, before the move), 09:00 11, 09:30 12. Per hour today 05:4 06:9 07:11 08:30 09:23 vs Sep 3 +same hours 2/7/8. So today is 2–3x worse than yesterday and was rising before the deploy; after the deploy +it is ~11–12 per 30 min, in line with 06:00–07:30. Cloud SQL backends (~230 max) and new connections +(~5k/30 min) are flat across the deploy. Latest crash (c21 09:39:11) is `Connection terminated due to +connection timeout` with cause `Connection terminated unexpectedly` in `verifyCellAssignment` <- +`beginProof`, the same unhandled path. Conclusion: no evidence the deploy worsened it; the old image's +crash rate simply climbed all day. Director lock retries stayed ~10x lower after the deploy. + +**Checkpoint-phase check (10:00Z, negative result):** Postgres checkpoints complete every 5 min at ~:07. +Cell crashes bucketed by phase within that 5-min cycle show a mild :00–:29 s cluster today (22 of 103) +that is absent on Sep 3 (7 of 114), so checkpoints are not the trigger. Disk write bytes in cascade +minutes are at or below the median except 09:00. Cloud SQL memory 0.47, transaction rate flat. The +09:55 stall (11 director + 4 cell pg-connect timeouts in the same 4 s) came with `could not obtain lock +on row in relation "relay_cells"` from a NOWAIT sweep at 09:55:36, i.e. someone was holding the full +inventory at that moment. On the new director that can only be placement or a sweep; on the old cells it +is still every rebind. What stalls *connections* (not locks) for 2 s fleet-wide remains unexplained; +Cloud SQL is `db-custom-4-15360` REGIONAL PD_SSD 49 GB at 0.5–0.75 CPU when it happens. + +**Stall census (10:01Z):** 33 pg-connect-timeout stall events today (clusters of timeouts < 20 s apart). +Before 08:35 they were 1–9 timeouts each and 10–60 min apart; from 08:35 the big ones are 16, 22, 21, +17 timeouts and 5–30 min apart. No second-of-minute phase (start seconds spread across all buckets), so +not a fixed timer. Cloud SQL backends by state at 09:55: active peaked 42 at 09:52, idle-in-transaction +≤ 10, nothing near the 400 ceiling; memory 0.47; disk normal. Each stall is a few seconds where *new* +connections to Cloud SQL (via the auth proxy socket) time out at the 2 s `connectionTimeoutMillis`, +hitting every process that happens to need a fresh pool connection in that window. Old-image cells die +on it (unhandled), new-image director logs a 2 s 500 and continues. Root cause of the stall itself is +outside the relay code (Cloud SQL proxy or instance); not chased further here. + +## Finding 10 (2026-09-04 12:40Z): Cloud SQL disk write saturation since 11:58Z is driving the stalls + +`orca-cloud-auth-db` is `db-custom-4-15360` on a **49 GB PD-SSD** (81% used). PD-SSD performance scales +with size: 49 GB gives roughly 1,470 write IOPS and ~23 MB/s write throughput. Measured: + +| | before 11:58Z | 11:59Z onward | +|---|---|---| +| disk write MB/s | 4–6 | **30–50** (over the ~23 MB/s cap) | +| disk write IOPS | 500–800 | 800–1,475 (at the ~1,470 cap in 11:59, 12:15, 12:24, 12:34) | +| checkpoint `sync=` | 0.07–0.2 s (Sep 3 max 0.65 s, 290 checkpoints) | 2–20 s; 27 of 39 checkpoints in 12Z were >= 2 s | +| checkpoints per hour | 12 (timed, every 5 min) | 39 (WAL-triggered, every ~45 s; `write=` fell from 270 s to 30 s) | +| Cloud SQL CPU / memory | 0.5–0.8 / 0.47 | same (not the bottleneck) | + +Every 4 s+ fleet-wide SQL stall since 11:04 (11:04, 11:17, 11:31, 11:34, 12:09, 12:10, 12:18, 12:20, +12:30) sits inside a slow checkpoint `sync` window; the 12:30:49 checkpoint synced 5.88 s (longest file +5.47 s), matching the 12:30:02–41 stall. During fsync the WAL writer stalls and every session waits, which +is why the stall hit all 23 cells and the director at once regardless of the relay lock changes. The +old-image cells then die on the pool timeout; the new image survives. What raised write volume ~8x at +11:58Z is not established (autovacuum ran on every relay table 11:55–11:57 and checkpoints are being +forced by WAL volume, so a write amplifier inside Postgres is the leading candidate; relay transaction +rate and Cloud SQL network bytes were flat). This is the first cause found today that is *upstream* of +the relay code and it explains the afternoon acceleration (11Z 68 dies, 12Z 47 by 12:34). + +Corrections after digging (12:45Z): relay query volume, renewals, reconnects, and assignments per 5 min +were **flat** across 11:58 (sqlQ ~330k, renewals ~115k), so the relay did not start writing more. WAL +recycling per checkpoint went 7 -> 10–11 files (16 MB each) at 45 s intervals, i.e. WAL output rose from +~0.4 MB/s to ~4 MB/s while data-file writes rose to 30–50 MB/s; checkpoints switched from `time` to `wal` +triggered at 11:58:24. No Postgres slow-statement or "checkpoints too frequently" lines. This is +write amplification inside Postgres (full-page writes after each of the now-frequent checkpoints on +hot pages, plus autovacuum on every relay table each minute) on a disk too small for its IOPS ceiling, +not new relay load. Instance label `managed_by=terraform`, created 2026-07-09; the instance resource is +**not** in `cloud/infra/terraform` (only the database, user, and secret are, via +`local.relay_database_instance_name`), so it lives in the other Terraform root (orca-cloud, per +[[orca-cloud-terraform-split-findings]]). `storageAutoResize=true` with limit 0, so Cloud SQL will grow +the disk only when it fills, not when IOPS saturate; disk is 81% full. + +Onset precisely: the 11:55:37 `time` checkpoint wrote 67,258 buffers (10.5% of shared_buffers, the +day's largest) over 163 s and completed 11:58:24. Every checkpoint since has been `wal`-triggered at +~45 s spacing (`max_wal_size` reached), each writing 13–20k buffers with 9–11 WAL files recycled. This is a +self-sustaining loop: a checkpoint completes -> every subsequent write to a hot page emits a full-page +image into WAL -> WAL fills `max_wal_size` in ~45 s -> next checkpoint -> repeat. The relay's hot rows +(`relay_cells`, `relay_assignments`, activity leases, cell runtime) are updated tens of thousands of +times a minute, so full-page-write amplification is large. Before 11:58 the 5-min timed checkpoints kept +WAL well under the limit; a one-off larger checkpoint tipped it over and the disk's write ceiling keeps +it there. Query Insights: io_time +30% in the 12:00 bucket, lock_time flat. + +**Owning workflow / mitigation (not applied):** raise the Cloud SQL data disk (PD-SSD IOPS and MB/s scale +linearly with GB; 49 -> 200 GB roughly quadruples the ceiling, online, no restart) in the Terraform root +that owns `google_sql_database_instance` for `orca-cloud-auth-db`, applied through that root's workflow. +A second, flag-level lever is raising `max_wal_size` (default 1 GB) so timed checkpoints resume; that is +also a Cloud SQL instance setting in the owning Terraform root. Per the standing rule, not applied from +this session. Until then the fleet-wide 4–6 s stalls recur on +every slow checkpoint sync, the old-image cells die on each one, and no 15-min gate window will exist. + +## Finding 11 (2026-09-04 12:55Z): **Cloud NAT port exhaustion** on the us-central1 cells is the second stall class + +`google_compute_router_nat.relay_gce` (us-central1, `AUTO_ONLY` IPs, no `min_ports_per_vm`, no dynamic +port allocation, i.e. the default **64 ports per VM**). `router.googleapis.com/nat/port_usage` per VM +hit **64 = the cap** in exactly the minutes the cells' Cloud SQL proxies logged `dial tcp +35.188.82.89:3307: i/o timeout` (12:20–12:22, 12:41–12:43, 12:51–12:53), and +`nat/dropped_sent_packets_count` went 0 -> 56/552/590, 82/272/133, 395/1565/1842 in those same minutes. +Hourly: port_usage max was 25–50 all of Sep 3 and until 10Z today, 64 in 11Z and 12Z; dropped packets 0 +until 11Z (219), then 5,491 in 12Z. Open NAT connections rose 400–600 -> 815–874. Every cell's Cloud SQL +traffic egresses through this NAT to the instance's public IP (the instance has no private IP: +`ipv4Enabled=true`, `privateNetwork` unset). When a VM's 64 ports fill, new TCP SYNs to 3307 are dropped, +the proxy's dial times out, and the relay pool's 2 s `connectionTimeoutMillis` fires: that is the exact +2 s stall the old image dies on and the new director surfaces as a 500. The dial timeouts hit c7 and c8 +hardest because they carry the most controls and open the most DB connections. + +What raised port demand today: each old-image crash re-opens a full pool through fresh NAT ports, the +autoheal recreates do the same, and the 55P03 retry storms keep more connections mid-transaction, so +crashes and NAT exhaustion feed each other. This is why the afternoon accelerated even after the disk +loop broke at 12:39. + +**Owning change (not applied):** `cloud/infra/terraform/relay-gce-foundation.tf` +`google_compute_router_nat.relay_gce` (this repo): set `min_ports_per_vm = 1024` (or enable +`enable_dynamic_port_allocation = true` with `max_ports_per_vm = 4096`) and, if needed, add manual NAT IPs +(each IP supplies 64,512 ports across VMs). Online change, no VM restart. The durable fix is giving the +Cloud SQL instance a **private IP** and pointing the proxy at `--private-ip`, which takes DB traffic off +NAT entirely; that is a Cloud SQL instance change in the orca-cloud foundation root plus a startup-script +flag here. Per the standing rule, not applied from this session. + +Direct proof: `resource.type="nat_gateway" AND jsonPayload.allocation_status="DROPPED"` shows **1,514 +dropped allocations to 35.188.82.89:3307** in 12:50–12:54 alone, every one of them the Cloud SQL public +IP. The NAT has zero manual IPs (AUTO_ONLY) and no port settings in Terraform, so it is at Google's +default 64 ports/VM. No workflow in this repo applies `relay-gce-foundation.tf` broadly (the roll +workflows apply cell templates with `-target`), so the NAT change needs a targeted apply of +`google_compute_router_nat.relay_gce`, which is an owner-run Terraform step. + +Original write-up of the symptom before the NAT correlation follows. + +The 12:50:30–12:50:50 stall (every cell 3.7–3.9 s SQL max, six old-image cells died) happened with +checkpoints healthy (85 ms) and disk at 6 MB/s, so it is not Finding 10. The cells' Cloud SQL Auth Proxy +logged `failed to connect to instance: dial error: dial tcp 35.188.82.89:3307: i/o timeout`. Count of +those per hour today: 08Z 1, 11Z 15, **12Z 416**; all of Sep 3: 4. Cloud SQL `up`/backends/connections +did not blip. So new TCP connections to the instance's public IP on 3307 are timing out from the cells' +proxies in bursts, which is exactly the "2 s connect timeout" the old image dies on. Query Insights for +12:49–12:54 attributes 1,380 s of lock wait to the placement CTE (`WITH assignment_state AS +MATERIALIZED …`) and 469 s to the single-row reservation UPDATE: the lock queue is the *consequence* of +connections stalling mid-transaction, not the cause. Not chased further; candidates are the proxy's +connection churn under the crash loops (each recreated cell opens a fresh pool) and the instance's +public-IP path. Relay code cannot fix this; it is Cloud SQL / network. Dial timeouts by minute today: 12:20 24, 12:21 +66, 12:41 22, 12:42 6, 12:51 160, 12:52 137, i.e. bursts of 20–160 s each, and they hit c7 (new image, +89 today) and c8 (93) hardest, so it is not the old image's connection churn either. Cloud SQL `up`=1 +throughout. The proxy dials the instance's public IP `35.188.82.89:3307`; a burst of i/o timeouts to a +healthy instance points at the path (public-IP egress / NAT / proxy connection limits), not at Postgres. +That is the same 2 s that the old image dies on and that the new director surfaces as a 500. + +## Roll inputs (verified by the read-only `verify` run) + +**Image census from instance templates, 2026-09-04 21:45Z (authoritative, read from `gcloud compute +instance-templates`):** 20 serving cells on `5aedbca5` (c8, c9, c10, c13–c16, c19–c29) — the image that exits +the process on a Postgres connect timeout (Finding 6); c7 on `85bf6799`; c4, c5, c17, c18 (draining / +migration-only) on `0e83408b` / `36a56b10`; c1, c2, c3, c6, c11, c12 (existing-only) on Jul/Aug images. Target +for Roll 1 is `519f4914` (director already on it). Monitor dry-run dispatched 21:45Z as the roll gate; waves +require owner go. + + +- target-image-digest `sha256:519f4914217f08cabcdcd34825965db8473ec37c6591553a3af0d65dcdeeb183` (lock fix; supersedes 85bf6799 as target) +- previous target `sha256:85bf67993869a769642995d0863f4c2b6b569c3850c2d8390ec2ca5f2b179e28` (c7 is on this; use as c7's rollback) +- rollback-image-digest `sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563` +- target/rollback rehome protocol 1 / 1; expected-rehome-generation 12; selector generation **112** (110 before the c7 canary) +- existing-only c1,c11,c12,c2,c3,c4,c5,c6; migration-only c17,c18; general c10,c13–c16,c19–c29,c7,c8,c9 +- confirmation for canary: `ROLL_RELAY_SAME_CAP production-gce-c7` +- monitor evidence is single-use and must be < 5 min old at dispatch (plus 75 min per predecessor wave) +- monitor dry-run dispatch (read-only, runs at `main` head so a merged bar change applies immediately): + `gh workflow run cloud-monitor-relay-production.yml --ref main -f mode=dry-run -f expected-selector-generation=110 + -f expected-existing-only-cells= -f expected-migration-only-cells=production-gce-c17,production-gce-c18 + -f expected-general-cells= -f migration-policy=strict -f recovery-source-cell-id=none -f capacity-cell-id=none` + +## Queries that worked (copy-paste) + +- Cell metrics: `resource.type="gce_instance" AND jsonPayload.event="orca_relay_runtime_metrics"` +- Container crashes: `resource.type="gce_instance" AND jsonPayload.MESSAGE:"container die" AND jsonPayload.MESSAGE:"relay@sha256"` +- Crash banner: `resource.type="gce_instance" AND jsonPayload.message:"Node.js v24"` +- Retries: `jsonPayload.event="orca_relay_postgres_transaction_retry"` (no resource filter to get both) +- Director lines are `textPayload`; cell lines are `jsonPayload.message` +- Cloud Run concurrency: Monitoring API `run.googleapis.com/container/max_request_concurrencies` +- Dry-run final state: download artifact `relay-monitor-dry-run--`, read `*.state.json` (the log's `schemaVersion` lines are only checkpoints, not the final verdict) + +## 2026-09-04 22:50Z onward: owner go received; driving the gates + +Owner: "sure, feel free to drive these." Sequence chosen: Roll 1 first (highest uplift), auth deploy with +#478 second, pruner enable third, label drift resolved by matching Terraform to live state, #477 still held. + +| Step | Result | +| --- | --- | +| Monitor dry-run #19 (gen 112, strict) | **Passed** 23:07:53Z, run 33927238469 attempt 1. First green since the probe fix (#18723). 16 samples, no freeze. Dispatched 22:51:33Z after confirming: 0 `container die` in 3 h, director 5xx in the last 4 h were all 503s (excluded by the `director.errors` filter). | +| c8 `canary-apply` onto 519f4914 (rollback 5aedbca5) | **Failed at 23:09:07Z before any mutation**: `relay monitor evidence provenance does not match` in `verify-authority`. Run 33928330631. Gate job passed, `cell_1 / rollout` failed on the manifest check, `seal_canary` skipped, lease released. Cause: the manifest binds `commitSha`; the dry-run ran at main `264c9ed8d2`, the canary dispatched at `--ref main` resolved to `4fab8e2f15` because unrelated PRs merged to main during the 15-minute gate. Verified no side effects: c8 MIG still on template `…c8-20260827…` (5aedbca5), stable, 25 controls; no `/v1/admin/drain` or isolate calls in the director log. | +| Constraint learned | Both workflows must run at the **same main commit**. The production environment's deployment branch policy allows only `main`, and the job gates on `github.ref == 'refs/heads/main'`, so a pinned tag/branch is not an option. Any merge to stablyai/orca main during the 15-minute dry-run invalidates the evidence. Mitigation for the retry: dispatch the canary within seconds of the green, and do not merge anything to stablyai/orca main myself during the window. A durable fix (accept evidence whose commit is an ancestor with identical workflow/script content) is a follow-up, not a same-day change to a safety check. | +| Label drift (5.x) | Resolved by dropping the `region` label from Terraform to match the 21 live metrics (stablyai/orca #18734, merged). Targeted plan asserted `27 no-op, 9 create, 0 destroy`; applied 23:11Z: 8 `orca_relay_control_*` renewal metrics that had never been applied, plus `google_monitoring_dashboard.relay_incident`. `orca_relay_controls` createTime unchanged (2026-07-13), label extractors unchanged. | +| Pruner enable (1.2) | orca-cloud #479 merged: `auth_token_pruner_enabled = true`, image digest of `00031-tox`, `max_rows_per_run = 20000`. Targeted plan asserted 9 create / 0 change / 0 destroy (job, scheduler at `41 * * * *` UTC, two service accounts, five IAM grants). **Not yet applied**: waiting until the roll canary has landed so the first hourly run does not overlap a drain. | +| Auth deploy with #478 (3.1) | Dispatched 23:13Z from orca-cloud main `f0fa4b5` (run 33928663526). Candidate startup adds nullable `successor_material` under a brief ACCESS EXCLUSIVE lock. | +| Auth deploy result | **Succeeded** 23:15:37Z: `orca-cloud-auth-00035-gos` serving 100 %, cap 20 preserved, 0 5xx. `refresh_tokens.successor_material` present (nullable text); 298 sealed successors written in the first 15 min against 924 rotations; `session-refresh-reuse-detected` at baseline (5 / 15 min). Grace window is live. | +| Monitor dry-run #20 | Froze 23:35:38Z on `runtime_power_unknown cell.production-gce-c11.powered`. Two window restarts earlier (23:24, 23:25) on `signal_stale auth.errors` (Cloud Monitoring publish lag 181–255 s vs 180 s bar). Cause: one transient rejection of the per-cell MIG GET in `readResourceInventory` yields `targetSize: null` → `runtimeKnown=false` → hard freeze. c11 is a parked existing-only cell (MIG size 0, stable) and was fine. Not fleet health. Fix delegated: stablyai/orca #18740 (retry the MIG read once, mirroring #18723). Run 33928912676. | +| Monitor dry-run #21 | **Green** 23:54Z at main `8064d1f991`, but main had moved to `0a821e5bc8` during the window; the chain re-gated instead of dispatching (the canary would have failed provenance again). Run 33930229711. | +| Monitor dry-run #22 | **Green** 00:10Z at `0a821e5bc8`; main moved to `2e80972450`. Re-gated. Run 33931177390. | +| Monitor dry-run #23 | Froze 00:18:31Z on `cell.production-gce-c29.latency_ms` 2635 > 2000, the probe's own round-trip from a US runner to asia-east2; c29 controls 17→19 and `sqlLatencyMsMax` flat ~1050 through the minute, no crash, no checkpoint stall. c29 probe max was 0 in the three previous gates, so a one-off. Run 33932092775. | +| Blocking constraint | Main receives unrelated merges every 5–10 min (23:08, 23:15, 23:17, 23:40, 23:42, …). A 15-min gate bound to an exact commit cannot be consumed under that traffic. Delegated a durable fix: `verify-authority` accepts evidence whose commit is an ancestor of the canary commit **and** has no diff on the monitor/deployer trusted paths; fails closed on shallow clones or unknown commits. Chain re-armed on dry-run #24 (run 33932679796) meanwhile. | +| Monitor dry-run #24 | Froze 00:28:00Z on `director.instances` 4 < 5. Cloud Run active-instance count read 4 for exactly one minute (00:27), 5 in every other minute for 3 h; min/max scale is pinned at 5; no new revision. A routine single-instance recycle. Not fleet health. Bar `directorInstancesMin: 5` with `latest-sum` cannot tolerate that; recalibrate to 4 or use a 3-min window minimum (follow-up, not same-day). Run 33932679796. Chain dispatched #25 (run 33933193511) at `86cd327749`. | +| Monitor dry-run #25 | **Green** 00:46Z at `86cd327749`; main moved to `8096cb2803`. Fourth green gate lost to unrelated main traffic (#19, #21, #22, #25). Run 33933193511. Chain's re-gate #26 (run 33934079533) cancelled by me. | +| Fixes merged 00:55Z | stablyai/orca #18740 (MIG inventory read retried once before `runtime_power_unknown`; 2 tests) and #18754 (`verify-authority` and the batch canary authority accept evidence sealed at an **ancestor** commit when every trusted monitor/deployer path is byte-identical; fails closed on shallow clones and unknown commits; deploy/rehome jobs now check out with `fetch-depth: 0`; 5 new tests, 18/18 pass). Reviewed both diffs; trusted-path set verified to exist on main. | +| Monitor dry-run #27 | Dispatched 00:56Z at `74ad08ec66` (first gate whose evidence the new rule can consume). Run 33934541092. Chain re-armed with the same ancestor + identical-trusted-code rule so an unrelated merge no longer forces a re-gate. | +| Monitor dry-run #27 | **Green** 01:11:35Z at `74ad08ec66`; main had moved to `38bde20121` with identical trusted code, so the new rule (#18754) let the chain dispatch. Run 33934541092. | +| c8 `canary-apply` #2 (run 33935407461) | Provenance check **passed** (first consumption of ancestor evidence). Isolate → gen 113, drain, template+MIG applied 01:14–01:22, new c8 came up on `519f4914` and `relay_capacity_transition_verified` (migration-only, image exact, heartbeat fresh) at 01:23:50. Then the step's next call, `curl --fail-with-body` to c8 `/v1/admin/runtime-status`, got a **503 with a 27-byte body** at 01:23:51 and the step exited 22. Director `cell-status` at 01:23:50.8 returned 200; c8's own logs show nothing at that second; c8 health/ready both 200 seconds later; backend HEALTHY (the health check had just flipped TIMEOUT→HEALTHY at 01:22:16 and UNKNOWN→HEALTHY at 01:23:47 as the new instance warmed). Read: a single 503 at the load-balancer/warm-up edge on a curl with no retry, on a cell that was already verified healthy one line earlier. Failsafe ran: c8 kept **migration-only**, rehome control disabled, selector gen 113. c8 is serving (40 controls at 01:39, sqlLatencyMsMax ~30 ms) on the target image, just not admitted for general traffic. Nothing to roll back. | +| Recovery | The job has an explicit resume path: `mode=rollback` with `rollback-image-digest` = the image the cell already runs skips isolate/apply, verifies, and restores general admission (`ROLLBACK_RESUME=true`). Dispatched gate #28 (run 33936966508) at gen 113 with c8 in migration-only; on green the chain dispatches that resume for c8 with rollback digest `519f4914` and target `5aedbca5` (the validator only requires them to differ). | +| Follow-up | The verify step's bare `curl --fail-with-body` needs the same "no reading is not a verdict" retry the monitor got (#18723/#18740); a 503 immediately after `verify-relay-capacity-transition` passed is not evidence of a bad cell. | +| Monitor dry-run #28 | **Green** 01:58:59Z at gen 113 with c8 in migration-only. Run 33936966508. | +| c8 recovery (run 33937756402, `mode=rollback`, rollback digest = 519f4914) | **Succeeded** 02:02Z. `ROLLBACK_RESUME=true` path: isolate/apply skipped, converged-Terraform check passed, verify passed (`relay_capacity_transition_verified` general, image `519f4914`, heartbeat fresh), activate → **gen 114**, c8 general. No restart, no drain. c8 at 43 controls, sqlLatencyMsMax 36 ms. **c8 is the second cell on 519f4914** (with c7 on 85bf6799). Because the recovery ran as `rollback`, `seal_canary` was skipped, so no canary authority exists for a `batch-apply`; the next cell runs as another `canary-apply`. | +| Merged 02:05Z | stablyai/orca #18769: bounded retries on every admin-endpoint curl/fetch in the same-cap job and the rehome/canary/verify scripts (`--retry 3 --retry-delay 2 --retry-connrefused`, per-attempt bodies to a file; script helper 2 attempts on network error or 500/502/503/504 only; 4xx never retried; 650/650 tests). Trusted-path change, so the next gate runs at a commit containing it. | +| Pruner enabled (1.2) | Terraform applied 02:06Z (8 creates, then the deploy-identity job IAM grant after a propagation 404, 9/9). Job `orca-cloud-auth-token-pruner`, image `343a0915…`, scheduler `41 * * * *` UTC, budget 20 000 rows/run. First run by hand (exec `sf5ct`): cold start 3m20s, then `stopReason: time-budget` at 480 s: 73 batches, 365 000 scanned, **1 040 deleted** (1 021 revoked, 19 expired, 0 rotated), ~6.4 s/batch of 5 000, `completedFullPass: false`. No errors, no lock-wait or checkpoint alert. Scan-bound, not budget-bound: at this pace a full pass over the table takes many hourly runs, and the row budget is never the limiter. Leave the budget alone; watch hourly runs for `stopReason` and a rising `deletedRows` as the cursor reaches the rotated backlog. | +| Monitor dry-run #29 | **Green** 02:20:58Z at gen 114, main `e2b70a5eba` (contains #18740, #18754, #18769). Run 33938052374. | +| c9 `canary-apply` (run 33938818286) | **Succeeded end to end** 02:21–02:34Z: isolate → gen 115, drain, template+MIG to `519f4914`, verify passed on the first try (retry-hardened step), trust proof, activate → **gen 116**, general. `seal_canary` **succeeded**: batch authority now exists. c9 at 38 controls, sqlLatencyMsMax 33 ms. No `container die` in 30 min. Three cells on new images (c7 `85bf6799`, c8 and c9 `519f4914`); 17 serving cells still on `5aedbca5`. | +| Monitor dry-run #30 | Dispatched 02:36Z at gen 116 (run 33939533990). On green the chain dispatches **batch 1**: `batch-apply` c10,c13,c14,c15 bound to canary run 33938818286 (sealed at gen 116, same commit `e2b70a5eba`). Preflight: all four on `5aedbca5`, MIGs stable, no crash in 20 min. Sequential cells inside the job (wave-index 0..3), each with its own isolate/drain/apply/verify/restore, so ~12 min per cell, ~50 min total. | +| Monitor dry-run #30 verdict | **Green** 02:52:15Z at gen 116, `e2b70a5eba`. | +| Batch 1 (run 33940290163) | Dispatched 02:52:27Z: `batch-apply` c10,c13,c14,c15, canary authority run 33938818286, same commit. | +| Batch 1 attempt 1 (run 33940290163) | **Failed at 02:54:39Z in the live preflight, before any mutation**: `relay live preflight failed: cloud-monitoring/signal_stale`. The step's `--retry-freshness` (5 attempts, 15 s apart, freshness-only codes) is passed only for `WAVE_INDEX != 0`; the first cell takes a single sample, so one Cloud Monitoring publish lag > 180 s at that instant fails the batch. Every candidate series was current again by the time I checked. c10 untouched (template `…c10-20260827…`, 47 controls), no selector write, gen still 116, failsafe no-op. Gate #31 dispatched 02:57Z (run 33940508865); chain re-dispatches the same batch (canary authority 33938818286 still valid: same gen 116, same commit). Fix delegated: wave 0 gets the same freshness retry. | +| Monitor dry-run #31 | **Green** 03:13:26Z at gen 116; main at `cb7f7dd11a` with identical trusted code. Run 33940508865. | +| Batch 1 attempt 2 (run 33941253533) | Dispatched 03:13:38Z: c10,c13,c14,c15, canary authority 33938818286. Runs at `cb7f7dd11a` (batch authority is accepted across the ancestor since trusted paths are unchanged). | +| Merged 03:14Z | stablyai/orca #18778: `--retry-freshness` on every same-cap wave including the first, and the retry loop now stops before the next wait would push evidence past the wave's age bound (it was checked only at entry before). Twin carve-out in the capacity job filed as a follow-up. | +| Batch 1 cell 1 (c10) | **Succeeded** 03:14–03:27Z (preflight, drain, apply, verify, restore). c13 started 03:27Z. | +| Batch 1 cell 2 (c13) | **Succeeded** 03:27–03:38Z. c14 started 03:38Z. | +| Batch 1 cell 3 (c14) | **Succeeded** 03:38–03:50Z. c15 started 03:50Z. | +| Batch 1 complete (run 33941253533) | **All four succeeded** 03:13–04:00Z: c10, c13, c14, c15 on `519f4914`, selector **gen 124**. Fleet at 936 controls, 23 cells. Two `container die` at 03:35:41/44 were **c13's new container** exiting during boot (`applyPostgresSchema` → `Connection terminated due to connection timeout`, exit 1, 2 s runtime each) because the `cloud-sql-proxy` sidecar had not finished starting; the third start at 03:35:45 succeeded and c13 has been serving since (57 controls). A boot-order race in the container spec, not a serving-cell crash. Follow-up: schema pool should wait for the proxy socket, or the container should depend on the proxy's readiness. **8 cells on new images** (c7 85bf6799; c8, c9, c10, c13, c14, c15 519f4914), 12 on `5aedbca5`: c16, c19–c26 (US), c27–c29 (Asia). | +| Monitor dry-run #32 | **Green** 04:19:50Z at gen 124, main `436ef827dd` (contains #18778). Run 33943539025. | +| c16 `canary-apply` (run 33944255902) | Dispatched 04:20:02Z. On success it seals the authority for batch 2 (c19,c20,c21,c22). | +| c16 canary (run 33944255902) | **Succeeded** 04:20–04:32Z, activate → gen 126, batch authority sealed. 9 cells on new images. | +| Monitor dry-run #33 | Failed 04:58:56Z on `continuity_deadline_exceeded` (1 500 004 ms > 1 500 000 ms). One `signal_stale cloud_sql.lock_waits` at 04:46 (189 s vs 180 s bar, Cloud Monitoring publish lag) restarted the 15-min window at sample 12; the restart could not complete inside the 25-min continuity cap. No health failure at any sample; no `container die` since c16's own boot race at 04:30. Run 33944873727. Chain re-gates. Note for recalibration: `cloudDataMaxAgeMs: 180000` vs observed Cloud Monitoring publish lag of 181–255 s has now cost three gates (#20 twice, #33). | +| Freshness recalibration | stablyai/orca #18798 (open, merge after batch 2 dispatch): `cloudDataMaxAgeMs` 180 s → 330 s, derived from Google's documented visibility delays (Cloud Run 60+120 s, Cloud SQL 60+165 s) and the 5-min window-sum query (a label series that stops emitting reads as up to 300 s old while its sum is complete, which is the 255 s `auth.errors` case) plus ~30 s collect latency. Director-admin and the lock-wait carry keep their own 180 s pins. A freshness-only failure may miss 2 consecutive samples without restarting the window; the sample still counts and is still threshold-checked; a 3rd miss, collector failure, runner gap, or any breach restarts/freezes as before. 92/92 tests. | +| Monitor dry-run #34 | **Green** 05:17:31Z at gen 126, `436ef827dd`. Run 33946093029. | +| Batch 2 (run 33946819345) | Dispatched 05:17:43Z: c19,c20,c21,c22, canary authority 33944255902 (c16). | +| Merged 05:19Z | stablyai/orca #18798 (freshness bar 330 s + two-sample tolerance). Next gate runs at a commit containing it. | +| Batch 2 cell 1 (c19) | **Succeeded** 05:19–05:32Z. c20 started. | +| Batch 2 cell 2 (c20) | **Succeeded** 05:32–05:43Z. c21 started. | +| Batch 2 cell 3 (c21) | **Succeeded** 05:43–05:59Z. c22 started. | +| Batch 2 complete (run 33946819345) | **All four succeeded** 05:17–06:12Z: c19, c20, c21, c22 on `519f4914`, selector **gen 134**. Fleet at 1 090 controls, 23 cells, refresh 401s at baseline (1–4 per 3 min). One `container die` at 06:08:45 was **c22's new container** exiting during boot (exit 1, 2 s runtime; started 06:08:43, restarted 06:08:46 and serving since), the same proxy-sidecar boot race seen on c13 and c16. No serving-cell crash. **Census: 15 of 23 serving cells on new images** (c7 `85bf6799`; c8–c10, c13–c16, c19–c22 `519f4914`), 7 on `5aedbca5`: c23–c26 (US), c27–c29 (Asia). Next: gate at gen 134 → canary c23 → batch c24,c25,c26; then canary c27 → batch c28,c29. | +| Monitor dry-run #35 | **Green** 06:32:01Z at gen 134, `b33d1972bc` (contains #18798, first gate at the 330 s freshness bar). Run 33949334606. | +| c23 `canary-apply` (run 33950075843) | Dispatched 06:32:13Z at main `b0c67eaf88` (ancestor gate SHA, identical trusted code). On success it seals the authority for batch 3 (c24,c25,c26). | +| c23 canary (run 33950075843) | **Succeeded** 06:32–06:46Z, activate → gen 136, batch authority sealed. No `container die` during boot. 16 of 23 serving cells on new images; 6 on `5aedbca5` (c24–c26 US, c27–c29 Asia). | +| Monitor dry-run #36 | Dispatched 06:46Z at gen 136, run 33950746574 (`58553bfe1c`). On green the chain dispatches batch 3 (c24,c25,c26) under canary authority 33950075843. | +| Monitor dry-run #36 result | **Green** 07:02:49Z at gen 136, `58553bfe1c`. | +| Batch 3 (run 33951468008) | Dispatched 07:03Z: c24,c25,c26, canary authority 33950075843 (c23). | +| Batch 3 cell 1 (c24) | **Succeeded** 07:04–07:18Z. c25 started. | +| Batch 3 cell 2 (c25) | **Succeeded** 07:18–07:31Z. c26 started. | +| Batch 3 complete (run 33951468008) | **All three succeeded** 07:03–07:44Z: c24, c25, c26 on `519f4914`, selector **gen 142**. Fleet at ~1 230 controls, 23 cells, refresh 401s at baseline. **Zero `container die`** during the batch (no boot race on c24–c26). **All 20 US serving cells now on new images** (c7 `85bf6799`; c8–c10, c13–c16, c19–c26 `519f4914`). Remaining on `5aedbca5`: c27, c28, c29 (asia-east2, probe hard cap 3000 ms). | +| Monitor dry-run #37 | Dispatched 07:48Z at gen 142, run 33953555224 (`4c5077d57a`). On green the chain dispatches the c27 canary (first Asia cell). | +| Monitor dry-run #37 result | **Green** 08:04:24Z at gen 142, `4c5077d57a`. | +| c27 `canary-apply` (run 33954264945) | Dispatched 08:04Z, first Asia cell (asia-east2-a). On success it seals the authority for batch 4 (c28,c29). | +| c27 canary (run 33954264945) | **Failed closed before any mutation** 08:07:25Z at "Verify exact current generation, digest, cap, and rollback point": `runtime predecessor mismatch fields=regionalRehomeProtocol`. **Operator input error, not a cell fault**: the chain script hardcoded `target-rehome-protocol=1 / rollback-rehome-protocol=1` for every cell, but `relay_region_rehome_source_cell_ids` lists only the 16 US cells (c7–c10, c13–c16, c19–c26), so the Asia startup template omits `ORCA_RELAY_REHOME_*` and c27–c29 report protocol 0 by design. `MUTATION_STARTED` never set, failsafe no-op, selector stays gen 142, c27 still serving on `5aedbca5`, no `container die`. Gate #37 evidence consumed. Fix: chain script now takes `PROTO`; Asia round dispatches with protocol 0 (the per-host trust proof step is protocol-gated and skips, as designed for non-source cells). Follow-up: the job already reads `relay_region_rehome_source_cell_ids`; it could derive the expected protocol from membership instead of trusting the operator input. | +| Monitor dry-run #38 | Dispatched 08:12Z at gen 142, run 33954621425 (`e95d247be1`). On green the chain dispatches the c27 canary with protocol 0. | +| Monitor dry-run #38 result | **Green** 08:28:36Z at gen 142, `e95d247be1`. | +| c27 `canary-apply` #2 (run 33955359385) | Dispatched 08:28Z with `target/rollback-rehome-protocol=0`. | +| c27 canary #2 (run 33955359385) | **Failed closed, no mutation** 08:31:19Z. Predecessor check passed with protocol 0; the isolate step then died at argument parsing: `production capacity target is not approved`. The same-cap job shells out to `prepare-relay-production-capacity-canary.mjs` for isolate/drain/activate, whose `PRODUCTION_CAPACITY_CELL_IDS` allowlist is the 16 US capacity cells (c7–c26), while the same-cap wave validator (`SAME_CAP_CELLS`) approves all 19 serving cells including c27–c29. The Asia cells have never been through this job (their Aug 14 rollout used the asia-topology workflow). Both the isolate step and the failsafe threw before any HTTP call, so `MUTATION_STARTED=true` was written but nothing was isolated: selector stays gen 142, c27 general and serving on `5aedbca5`, no `container die`. Gate #38 evidence consumed. Fix: stablyai/orca #18811 (`--approved-cells same-cap` on all four invocations, default unchanged for the US capacity job, census test over every `SAME_CAP_CELLS` member × isolate/drain/activate + the job's cell-shape bash block; 525/525 script tests). Sweep of the other job scripts found no further Asia blocker; gate #39 (run 33955668701) dispatched at gen 142 to prove the selector is unchanged before the next attempt. | +| Monitor dry-run #39 | **Green** 08:51:28Z at gen 142: independent proof the selector was untouched by both failed c27 attempts. Not used for dispatch (its commit predates #18811). | +| Merged 08:51Z | stablyai/orca #18811 → main `12e05203a4`. | +| Monitor dry-run #40 | Dispatched 08:51Z at gen 142 on main `12e05203a4` (contains #18811), run 33956408337. On green the chain dispatches the c27 canary, protocol 0, third attempt. | +| Monitor dry-run #40 result | **Green** 09:08:03Z at gen 142, `12e05203a4`. | +| c27 `canary-apply` #3 (run 33957151726) | Dispatched 09:08Z, protocol 0, on main containing #18811. | +| c27 canary #3 (run 33957151726) | **Failed after isolate; failsafe held** 09:17:21Z. Live check 09:26Z: c27 at 0 controls (drained), template still `…20260814235757`, c28/c29 absorbed the hosts (37 each), fleet 1 404 controls / 23 cells, refresh 401s baseline, no `container die` in 60 m. Predecessor check and allowlist passed; isolate → **gen 143** (c27 migration-only), drain sent (graceMs 0, hosts reconnected via director to c28/c29/US). Terraform plan built correctly (template replace + MIG update to `519f4914`), then `validate-relay-capacity-plan.mjs --mode same-cap-cell` rejected it: `cell plan does not contain the reviewed image and capacity`. Its same-cap rule demands exactly one `ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT` and one `ORCA_RELAY_REHOME_AUDIENCE` printf in the startup script; Asia templates omit both because c27–c29 are not rehome sources (same root as attempt 1, third US-only assumption in the job). **No apply ran**: c27 template unchanged, still `5aedbca5`, isolated and draining (drain is one-way in-process; only a restart clears it). Failsafe re-asserted migration-only at gen 143 and rehome disabled. Recovery plan: fix validator (protocol-0 path: require the rehome lines *absent*), merge, gate at gen 143, then `mode=rollback` with rollback-image=`519f4914` (the failed-canary re-entry path; accepts draining + migration-only) to restart c27 onto the target image and restore it; then single-cell canaries for c28 and c29 (batch needs ≥2 cells). | +| Plan-validator fix | stablyai/orca #18818 (merged 09:41Z → main `9f2a9a248e`): `validate-relay-capacity-plan.mjs --regional-rehome-protocol 0|1` in same-cap-cell mode; protocol 0 requires the rehome lines *absent*, protocol 1 unchanged; both plan-validation calls in the job pass `DESIRED_REHOME_PROTOCOL`; census test now validates a correct plan for every `SAME_CAP_CELLS` member at its tfvars-derived protocol. 529/529. Residual: the operator-supplied protocol is still unbound for Asia cells (no `SOURCE_CELLS` cross-check outside us-central1), so a wrong value fails late at plan validation rather than early; deriving it from membership is the checklist follow-up. | +| Monitor dry-run #41 | Dispatched 09:42Z at gen 143 (c27 expected migration-only) on main `9f2a9a248e` (contains #18811 + #18818), run 33958728141. On green: c27 recovery via `mode=rollback`, rollback-image `519f4914`, protocol 0, confirmation `ROLL_BACK_RELAY_SAME_CAP`. | +| Monitor dry-run #41 result | **Green** 09:58:51Z at gen 143, `9f2a9a248e`. | +| c27 recovery #1 (run 33959789773, `mode=rollback`) | **Failed closed, no mutation** 10:09:21Z at `Verify monitor evidence provenance`: `relay monitor dry-run authority is incomplete or stale`. The dry-run authority is valid for 5 min after `completedAt` at wave 0 (`EVIDENCE_MAX_AGE_MS`); the gate completed 09:58:51Z but the operator poller (20 s `gh run view` loop) only observed completion at 10:07:09Z during a local network outage, so the dispatch landed at 10:07:11Z, 8 m 20 s after completion. Failed before the rollout lease, isolate, or any Terraform step; c27 unchanged (migration-only, drained, `5aedbca5`, gen 143). Every prior canary dispatched ≤15 s after gate green, so this is a dispatch-latency miss, not a job defect; the freshness bound behaved as designed. | +| Monitor dry-run #42 | Dispatched 18:39Z at gen 143 on main `af82126058` (trusted paths byte-identical to `9f2a9a248e`), run 33984753269. Recovery script re-armed behind it (same `mode=rollback` onto `519f4914`, protocol 0). | +| Monitor dry-run #42 result | **Green** 18:55:47Z at gen 143, `af82126058`. | +| c27 recovery #2 (run 33985902062, `mode=rollback`) | **Failed closed, no mutation** 19:05:02Z, same `authority is incomplete or stale`. Dispatch landed 19:02:39Z, 6 m 52 s after the gate completed. Root cause of both misses is the operator laptop sleeping during the 15 min gate wait (`pmset -g log`: asleep 18:52:28Z → 19:02:17Z; the morning miss coincided with a sleep/dark-wake cycle too), so the 20 s poller never ran inside the 5 min window. Not a job or evidence defect: the freshness bound did its job. Operator fix: poller now runs under `caffeinate -i`. | +| Monitor dry-run #43 | Dispatched 19:06Z at gen 143 on main `af82126058`, run 33986121849. Recovery armed behind it under `caffeinate`. | +| Monitor dry-run #43 result | **Green** 19:22:53Z at gen 143, `af82126058`. | +| c27 recovery #3 (run 33986948522, `mode=rollback`) | **Failed closed, no mutation** 19:25:48Z. Dispatched 13 s after gate green (authority accepted this time), then the live preflight recheck failed: `relay live preflight failed: active-probe/threshold_max`. That is the 2 000 ms `endpointLatencyMs` bar on one endpoint's slowest /health or /ready round trip from the runner (8 s fetch timeout, one retry). The error names no endpoint and the job log prints none; gate #43 had zero failures across 16 samples, so this was a transient probe slow-down in the ~3 min between gate and preflight. Live probe 19:32Z from the operator: director and auth ~130–190 ms, US cells ≤540 ms, Asia cells 690–1 315 ms (c28/c29 /health ~1.3 s, the closest to the bar; c27 ~0.9 s). Existing-only cells c1–c3, c6, c11, c12 return 503 on both paths as expected (unpowered). Failed before the rollout lease, isolate, or any Terraform step; c27 unchanged. Follow-up (checklist): preflight should print the failing signal and observed value. | +| Monitor dry-run #44 | Dispatched 19:33Z at gen 143 on main `062db77118`, run 33987646501. Recovery re-armed behind it. | +| Monitor dry-run #44 result | **Frozen red** 19:50:01Z after 13 samples: `active-probe/threshold_max cell.production-gce-c27.latency_ms observed=2568 threshold=2000`. No other failure, no continuity event, no `container die` fleet-wide in 60 m. `/health` is a static JSON reply (`app.ts`), so the slow round trip was `/ready` (the probe reports the max of the two) or the path to the cell. Cloud SQL logs for 19:49:38Z–19:51:58Z show six `could not obtain lock on row in relation "relay_cells"` errors and a time-triggered checkpoint completing at 19:50:36Z (write phase 270 s, the spread target, not a stall). c27 is drained with 0 controls, so its `/ready` dependency check was the only thing it was doing. Recovery script stopped as designed (no auto re-gate). Operator probe 19:53Z: c27 and c28 both bimodal, ~0.27 s or ~0.89 s per `/health` from the US, identical shape, nothing c27-specific. Attributing the one 2.6 s sample to the same shared-DB contention that produced the lock errors is the best available reading; the retry at gate #45 tests whether it recurs. | +| Monitor dry-run #45 | Dispatched 19:53Z at gen 143 on main `062db77118`, run 33988383401. Recovery re-armed behind it. | +| Monitor dry-run #45 result | **Frozen red** 19:54:47Z after 3 samples, same signal: `cell.production-gce-c27.latency_ms observed=2668 threshold=2000`. Two gates in a row now attribute a >2 s round trip to c27 while every other cell passes. | +| c27 `/ready` tail analysis | `/health` is static; `/ready` (`relay-readiness.ts`) fetches the auth JWKS (2 s timeout) then runs `SELECT 1`, cached 10 s. Operator probes 19:57Z–20:00Z, 15 each from the US: c27 and c28 have the **same** tail (0.27 s / 0.88 s modes, then 1.3 s, then 2.17–2.27 s at the top); US cells c8/c20 sit at 0.08–0.18 s. Auth JWKS latency over the last hour: 400 requests, max 20 ms, none over 1 s. So the tail is cell→Cloud SQL (US) round trips plus the runner→Asia hop, not auth and not c27-specific; c27 is drained (0 controls) so nothing local competes. Cloud SQL `could not obtain lock on row in relation "relay_cells"` runs at 17–78 per 10 min all day (NOWAIT inventory locks, expected under placement bursts) with no spike in the failing minutes. The bar (`endpointLatencyMs` 2 000 ms, one shot per minute, max of two paths) leaves Asia cells ~10% of samples from tripping; the gate got unlucky twice on c27 and lucky on c28/c29. Not a health finding. | +| Monitor dry-run #46 | Dispatched 20:01Z at gen 143 on main `062db77118`, run 33988810139. Recovery re-armed behind it. If this also freezes on an Asia probe, the next move is a per-region latency bar (or p50 over the window) in `incident-monitor.ts`, reviewed and merged before further Asia gates rather than retrying blindly. | +| Monitor dry-run #46 result | **Frozen red** 20:07:44Z after 7 samples, third time on `cell.production-gce-c27.latency_ms` (observed 2 685). Operator 40-sample `/ready` probe per Asia cell at 20:10Z: c27 p50 0.88 s / p90 2.15 s / max 2.26 s / 6 over 2 s; c28 p50 0.88 / p90 1.25 / max 2.25 / 1 over; c29 p50 0.88 / p90 0.89 / max 1.27 / 0 over. All 200. `/ready` (`relay-readiness.ts`) fetches the auth JWKS in us-central1 then `SELECT 1` on Cloud SQL in us-central1, so an Asia cell's readiness is two trans-Pacific hops plus the runner→Asia hop; the fleet-wide 2 000 ms bar was calibrated on US cells (0.08–0.5 s). c27 being drained and idle has no local load, so this is path latency, not health. **Stopped retrying gates.** Fix in flight: per-region `cell..latency_ms` bar (us-central1 stays 2 000, asia-east2 4 000; hard faults still caught by the health/ready equal-1 checks and the 8 s probe timeout) plus attributable preflight failure messages, via review + CI before the next Asia gate. | +| Merged 20:33Z | stablyai/orca #18877 → main `a3c1d32995`: per-region `cellEndpointLatencyMs` (us-central1 2 000, asia-east2 4 000; director/auth rules and the `endpointLatencyMs` key unchanged), region carried from tfvars onto every cell expectation, preflight failures now print `source/code signal observed= threshold=`. relay-ops 95/95, cloud suite 633 + 529 + 148 green. | +| Monitor dry-run #47 | Dispatched 20:34Z at gen 143 on main `a3c1d32995` (first gate with the per-region bar), run 33989896150. Recovery re-armed behind it. | +| Monitor dry-run #47 result | **Green** 20:38:09Z at gen 143 on `a3c1d32995`: first gate under the per-region bar, 16/16 samples, no Asia latency failure. | +| c27 recovery #4 (run 33990715317, `mode=rollback`) | **Success** 20:51Z. Dispatched 13 s after gate green. Isolate re-asserted migration-only at gen 143 (already isolated, no change), Terraform applied the same-cap template `…20260905204141` and the MIG replaced the instance, new incarnation on `519f4914`, protocol 0, transition verifier passed at migration-only (1 180 assignments carried, hard cap 3 000, heartbeat fresh), then activate → **gen 144**, c27 general, verifier passed again. No `container die` fleet-wide 19:55Z–20:52Z. c27 now runs the target image; c28/c29 remain on `5aedbca5` (template `…20260814235757`). | +| Monitor dry-run #48 | Dispatched 20:53Z at gen 144 (c27 back in general, MIG = c17,c18) on main `61ebffa86e` (trusted paths identical to `a3c1d32995`), run 33991385880. On green the chain dispatches the c28 `canary-apply`, protocol 0. | +| Monitor dry-run #48 result | **Green** 21:08Z at gen 144, 16/16 samples, no Asia latency failure. Main had moved to `5cec2c2dfc`; the chain verified the trusted paths were identical to the gate commit and dispatched 12 s after green. | +| c28 canary (run 33992169289, `canary-apply`) | **Success** 21:27Z. Isolate → migration-only at **gen 145**, drain already clear, verifier passed on the old image (1 220 assignments carried, hard cap 3 000, heartbeat fresh), Terraform applied same-cap template `…20260905211352`, new incarnation on `519f4914` at protocol 0, verifier passed again at migration-only, activate → **gen 146**, c28 general, verifier passed (1 219 assignments). Seal step recorded the canary. No `container die` fleet-wide 21:08Z–21:30Z. Only c29 remains on `5aedbca5`. | +| Monitor dry-run #49 | Dispatched 21:33Z at gen 146 (c28 back in general, MIG = c17,c18) on main `dce5ebd83d` (trusted paths identical to `a3c1d32995`), run 33993075948. On green the chain dispatches the c29 `canary-apply`, protocol 0, the last Roll 1 cell. | +| Monitor dry-run #49 result | **Frozen red** 21:52:24Z, `active-probe/continuity_deadline_exceeded observed=1500005 threshold=1500000`. One continuity event at 21:41:27Z, `cloud-monitoring/collector_failed` (a Cloud Monitoring read failed, not tolerated), which reset the continuous window at sample 14; the restarted window reached 10 samples before the 25-minute lineage cap (`INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS`) expired. No health failure in any of the 25 samples, no Asia latency failure, no `container die`. Monitor-side transient, not a fleet finding. The chain re-gated automatically after its 2-minute back-off. | +| Monitor dry-run #50 | Dispatched 21:54Z at gen 146 on main `51eed5a1bc`, run 33994385666. **Green** 22:10Z, 16/16 samples. Main had moved to `d7767fb196`; trusted paths identical to `a3c1d32995`. Chain dispatched the c29 `canary-apply` (run 33995164002, protocol 0) 12 s after green. | +| c29 canary (run 33995164002, `canary-apply`) | **Success** 22:27Z. Isolate → migration-only at **gen 147**, verifier passed on the old image (1 199 assignments), Terraform applied same-cap template `…20260905221622`, new incarnation on `519f4914` at protocol 0, verifier passed at migration-only, activate → **gen 148**, c29 general, verifier passed (1 199 assignments carried). No `container die` fleet-wide 22:11Z–22:30Z. | +| **Roll 1 complete** | Image census 22:30Z from MIG templates: c8–c10, c13–c16, c19–c29 on `519f4914` (18 cells); c7 on `85bf6799` (the earlier rehearsal image, carries the same fix); existing-only c1–c6, c11, c12 and migration-only c17, c18 untouched by design. No serving cell remains on `5aedbca5`. Selector gen 148, membership unchanged from the start of the roll. Zero relay container exits fleet-wide across the roll (01:14Z–22:30Z). Gates used: #19–#50; freezes were all monitor-side (provenance, freshness, flat Asia latency bar, one Cloud Monitoring collector failure), none a fleet health finding. Roll 2 (fresh image with #18722 + #18720) is the next data-plane step and waits on the owner's private-IP window decision. | diff --git a/cloud/docs/relay-roll2-plan-2026-09.md b/cloud/docs/relay-roll2-plan-2026-09.md new file mode 100644 index 00000000000..f84de39163f --- /dev/null +++ b/cloud/docs/relay-roll2-plan-2026-09.md @@ -0,0 +1,154 @@ +# Relay Roll 2 and close-out plan (2026-09-05) + +Owner-approved scope 2026-09-05: finish the relay reliability work with one more cell image roll, +deferring the Cloud SQL private-IP move (2.1, orca-cloud #477) to a separate owner decision. Roll 1 +is complete (see `relay-reconnect-2026-09-findings.md`, "Roll 1 complete"); every serving cell runs +`519f4914` except c7 on `85bf6799`. + +Estimate: about two working days of effort over one week of calendar time. The cell roll itself is +6 to 7 hours of mostly unattended wall clock, run in the US night. + +## Phase 0. Land the code (half a day, no production change) + +### 0a. Split PR #18565 + +The branch mixes three relay/mobile/desktop fixes with the operator record. Split so the record +lands regardless of how the code review goes. + +- **Docs PR** (new branch off main): `relay-reconnect-2026-09-findings.md`, + `relay-improvement-checklist-2026-09.md`, `relay-improvement-roadmap-2026-09.md`, this file. + Docs only, merge on CI green. +- **Code PR** (rebase #18565 onto main, resolve two conflicts): + - `cloud/apps/relay/src/host-session-registry.ts`: conflict with #18698 (signed-out signal). + Keep both; the accept-abandonment and lease changes are orthogonal to the signed-out path. + - `src/main/runtime/relay/relay-origin-pool.ts`: **drop this branch's version**. #18719 already + merged the desktop early-window jitter (1 to 6 min). Also drop + `relay-session-broker.test.ts` additions that only exercise the dropped change. + - Keep: relay accept abandonment (`orca_relay_client_accept_abandoned` event), relay-side lease + jitter, mobile direct-probe fail-fast, and their tests. + +### 0b. Lengthen the control lease (same code PR) + +In `cloud/apps/relay/src/host-session-registry.ts`: + +``` +CONTROL_LEASE_MS = 6 * 60 * 60 * 1000 // was 55 min +CONTROL_LEASE_JITTER_MS = 30 * 60 * 1000 // was 5 min +``` + +Why 6 h: the lease bounds how long a host stays on a cell after a missed drain and is the only +passive rebalancing; 6 h keeps both and cuts control-activation traffic on the inventory lock by +about 6x. Nothing else depends on it: the relay JWT (5 min) is refreshed by the desktop on its own +schedule and liveness is the 75 s silence watchdog. Wire-safe: the relay sends `leaseExpiresAt` in +the hello ack and old desktops schedule from that value. + +Update the comment above the constants and the three assertions in +`host-session-client-accept.test.ts` that pin the lease arithmetic. Check that nothing in +`cloud/apps/relay-ops` or the monitor thresholds assumes a 55 min rotation period (grep +`55`, `CONTROL_LEASE`, `rotation`). + +### 0c. Review and merge + +Review rounds per the standing process (Opus review, then Codex pass). Merge order: docs PR first +(no dependency), then the code PR. Record the merge SHA of the code PR; that is the Roll 2 image +source. + +## Phase 1. Build and stage the image (half a day) + +Roll 2 image = code PR merge SHA. It carries, relative to `519f4914`: + +| Change | PR | Effect | +|---|---|---| +| Per-cell inventory locks, delta counters | #18722 | Removes the global `relay_cells FOR UPDATE` behind the phone accept hang | +| Relay pool `statement_timeout` 5 s | #18722 | A relay query can no longer hang a cell | +| Accept abandonment | #18565 | Cell stops finishing accepts for phones that already closed | +| Control lease 6 h ± 30 min | #18565 | Fewer, spread-out rebinds | +| `--private-ip` proxy flag support | #18720 | Code only; flag stays unset until 2.1 | + +Steps, in order (from the findings doc's post-merge dispatch plan): + +1. `gh workflow run cloud-publish-relay-production.yml --ref main -f mode=publish`. Resolve the + digest by tag, not from the log: + `gcloud artifacts docker images describe us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay:sha- --format='value(image_summary.digest)'`. +2. Staging: `cloud-deploy-relay-staging.yml` with the new digest; paired phone plus desktop smoke + (connect, background, reconnect). Confirm `orca_relay_client_accept_abandoned` appears only when + a client closes early, and that `sqlLatencyMsMax` no longer pins at the lock timeout. +3. Director: `cloud-deploy-relay-production-director.yml -f image-digest= + -f regional-placement-mode=preserve -f prune-incompatible-revisions=false + -f expected-rehome-generation=12 -f bootstrap-runtime-identity=false + -f predecessor-image-digest=`. Blue/green; prior revision stays as rollback. + Watch director `orca_relay_postgres_transaction_retry` per minute before and after. The director + goes first so the per-cell locks are live before any cell restart burst. +4. Same-cap `verify` mode against c7 with target=, rollback=`519f4914`. Read-only. + +Go/no-go for Phase 2: director serving the new image for at least 30 min, retries per minute at or +below the pre-deploy baseline, no `container die`, no auth 5xx. + +## Phase 2. Roll the cells (one US night, mostly unattended) + +Same machinery as Roll 1: `cloud-monitor-relay-production.yml` dry-run gate, then +`cloud-deploy-relay-production-same-cap.yml`. Cells roll one at a time by design (exact selector +assertions, single Terraform state, and one cell's ~1.2k-host reconnect burst per restart). Do not +add parallelism for this roll. + +Inputs: target=, rollback=`519f4914` (c7: rollback=`85bf6799`). Selector membership is +unchanged from the end of Roll 1 (gen 148; existing-only c1–c6, c11, c12; migration-only c17, c18). + +Order: + +1. **c7 canary** (`canary-apply`, protocol 1). c7 is the rehearsal cell and the only one not on + `519f4914`. +2. **c8 canary**, then **batch c9, c10, c13, c14**. +3. **c15 canary**, then **batch c16, c19, c20, c21**. +4. **c22 canary**, then **batch c23, c24, c25, c26**. +5. **Asia c27, c28, c29** as three single canaries at protocol 0 (`PROTO=0`). Batch mode cannot + take Asia cells yet and needs at least two cells. + +Each batch needs a same-commit canary authority; each wave needs a fresh 15 min gate. Use the +chain script pattern from Roll 1 (wait gate green, check trusted-path ancestry, dispatch within 5 min, +log `CANARY `) under `caffeinate -i`. Budget: 11 to 13 min per cell plus 15 min per gate, +about 6 to 7 h total. + +Per wave checks (same as Roll 1): transition verifier passes at migration-only and again at general +with assignments carried; no `container die` fleet-wide; selector generation advances by exactly 2 +per cell. After the Asia cells: image census from MIG templates; every general cell on the new digest. + +Failure handling: a failed canary re-enters through `mode=rollback` with rollback-digest = desired +image (Roll 1 c27 pattern). A gate freeze on an Asia latency probe despite the 4 000 ms bar is a +stop-and-investigate, not a retry. Monitor-side freezes (freshness, continuity deadline) re-gate +after a 2 min back-off; the chain does this on its own. + +Record every gate and wave in the findings doc as in Roll 1. + +## Phase 3. After the roll (spread over the following week) + +- **4.4 Recalibrate the retries bar.** After one week of `orca_relay_postgres_transaction_retry` + on the new image, re-derive the `postgres_retries` monitor threshold from the new baseline + (PR against `cloud/apps/relay-ops/src/incident-monitor.ts` thresholds). About 2 h. +- **1.2 Pruner budget.** Raise `auth_token_pruner_max_rows_per_run` to the default 200k after a + clean day; watch Cloud SQL write MB/s and the checkpoint alert. Then **1.5** log metric plus + policy on `stopReason != complete`. +- **1.3 Reclaim.** Once pruner runs delete ~0 rows: `pg_repack -t refresh_tokens` off-peak (check + `pg_available_extensions` first; not `VACUUM FULL`). Confirm table, index, and `disk/utilization` + dropped. +- **Monitor residuals** already in the checklist: `probeEndpointHealth` retry decision still uses the + flat 2 000 ms bar; operator protocol unbound for Asia; `probe-relay-rehome-trust` regex. +- Update the checklist status header; tick 2.3, 4.1, 4.3 relay-side as deployed. + +## Deferred, owner decision required + +- **2.1 Private IP** (orca-cloud #477). One-way door with a Cloud SQL restart. When chosen: apply the + foundation off-peak, then a template-only change that sets the `--private-ip` proxy flag. That is + another cell roll unless bundled with a future image. +- **5.2 Paging channel** for auth alerts: needs a destination. +- **Parallel cell rolls** (2 or 3 at a time): about 1.5 days (relax exact-selector assertions to + "exact except in-flight", single coordinator Terraform apply, parallel job shape, tests). Only + worth building if more image rolls are planned after Roll 2, and only once the per-cell locks are + live so a multi-cell reconnect burst is safe. +- **2.2 Database split**: deferred to ~2026-11-01. + +## Not in this plan + +Desktop and mobile changes already merged (#18719 desktop early-window jitter and no same-token +refresh retry; #18565 mobile fail-fast once merged) ship with the next desktop and mobile releases +on their own schedules. No relay action needed. From 6a3e446c69b46ee63e13304cb7402d5c893915fa Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 17:24:12 -0700 Subject: [PATCH 21/23] test: make SSH artifact regression fixtures reliable at narrow widths (#18947) --- tests/e2e/ssh-codex-display-artifacts-repro.spec.ts | 2 +- tests/e2e/ssh-codex-repro-remote-fixtures.ts | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts b/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts index e19a090df4d..f4c02d04c94 100644 --- a/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts +++ b/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts @@ -48,7 +48,7 @@ import { resetWebglAndCaptureGraySlabAnalysis } from './terminal-webgl-reset-cap const RUN_DOCKER_SSH = process.env.ORCA_E2E_SSH_DOCKER === '1' const RUN_REAL_REMOTE_CODEX = process.env.ORCA_E2E_REAL_REMOTE_CODEX === '1' -const EXPECT_NO_ARTIFACTS = process.env.ORCA_E2E_EXPECT_NO_CODEX_ARTIFACTS === '1' +const EXPECT_NO_ARTIFACTS = process.env.ORCA_E2E_EXPECT_NO_CODEX_ARTIFACTS !== '0' const CAPTURE_WHILE_REMOTE_TUI_RUNNING = process.env.ORCA_E2E_CAPTURE_WHILE_REMOTE_TUI_RUNNING === '1' const HIDE_UNTIL_REMOTE_TUI_DONE = process.env.ORCA_E2E_HIDE_UNTIL_REMOTE_TUI_DONE === '1' diff --git a/tests/e2e/ssh-codex-repro-remote-fixtures.ts b/tests/e2e/ssh-codex-repro-remote-fixtures.ts index 3ee48187e7c..14597bc084a 100644 --- a/tests/e2e/ssh-codex-repro-remote-fixtures.ts +++ b/tests/e2e/ssh-codex-repro-remote-fixtures.ts @@ -135,7 +135,7 @@ async function insertCodexHistory(frame) { const phase = String(frame).padStart(4, '0') + '.' + index await write('\\r\\n') await write(\`\\x1b[48;2;72;72;72m\\x1b[K\`) - await write(\`\\x1b[38;2;220;220;220;48;2;72;72;72m\${pad('gpt-5.5 high · ~/code/pr-12250-migration-compare-move-baseprice-claim · /ps to view · /stop to close ' + phase, width)}\\x1b[0m\`) + await write(\`\\x1b[38;2;220;220;220;48;2;72;72;72m\${pad('gpt-5.5 high · ' + phase + ' · ~/code/pr-12250-migration-compare-move-baseprice-claim · /ps to view · /stop to close', width)}\\x1b[0m\`) } await write('\\x1b[r') await write(\`\\x1b[\${viewportBottom};1H\`) @@ -176,7 +176,7 @@ for (let frame = 0; frame < ${REMOTE_CODEX_FIXTURE_FRAMES}; frame += 1) { await reverseIndexCodexHistory(frame) } if (frame % 9 === 0) { - await grayScrollLine(\`gpt-5.5 high · ~/code/pr-12250-migration-compare-move-baseprice-claim · /ps to view · /stop to close \${frame}\`) + await grayScrollLine(\`gpt-5.5 high · \${frame} · ~/code/pr-12250-migration-compare-move-baseprice-claim · /ps to view · /stop to close\`) } await sleep(${REMOTE_CODEX_FIXTURE_FRAME_DELAY_MS}) } From 2e2ecc5193313fcded1a8b340340d2075fde4db9 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 17:26:54 -0700 Subject: [PATCH 22/23] test: order restart fixture readiness around daemon recovery (#18949) --- ...minal-host-restart-background-sync.spec.ts | 21 +++++++++++++++++++ .../restart-restore-terminal-input.spec.ts | 3 ++- 2 files changed, 23 insertions(+), 1 deletion(-) diff --git a/tests/e2e/paired-remote-terminal-host-restart-background-sync.spec.ts b/tests/e2e/paired-remote-terminal-host-restart-background-sync.spec.ts index 9855d8d92b0..cf2f96e9c84 100644 --- a/tests/e2e/paired-remote-terminal-host-restart-background-sync.spec.ts +++ b/tests/e2e/paired-remote-terminal-host-restart-background-sync.spec.ts @@ -283,6 +283,24 @@ async function expectTerminalInteractive( } async function moveHostAwayFromWorktree(page: Page, targetWorktreeId: string): Promise { + await expect + .poll( + () => + page.evaluate(async (targetId) => { + const state = window.__store?.getState() + const target = state?.allWorktrees().find((worktree) => worktree.id === targetId) + if (!state || !target) { + return false + } + await state.fetchWorktrees(target.repoId) + return window + .__store!.getState() + .allWorktrees() + .some((worktree) => worktree.repoId === target.repoId && worktree.id !== targetId) + }, targetWorktreeId), + { message: 'Seeded alternate host worktree never loaded' } + ) + .toBe(true) const alternateWorktreeId = await page.evaluate((targetId) => { const state = window.__store?.getState() const alternate = state?.allWorktrees().find((worktree) => worktree.id !== targetId) @@ -423,6 +441,9 @@ test('foregrounds a preserved daemon PTY after the paired host relaunches', asyn expect(reconnectControl.ptyId).not.toBe(target.ptyId) await openClientTab(client.page, worktreeId, reconnectControl.webTabId) await waitForPaneConnected(client.page, reconnectControl.webTabId) + await expect + .poll(() => readPaneContent(client!.page, reconnectControl.webTabId), { timeout: 30_000 }) + .toContain('READY') await expectTerminalInteractive(client, reconnectControl, 'y') } finally { if (client) { diff --git a/tests/e2e/restart-restore-terminal-input.spec.ts b/tests/e2e/restart-restore-terminal-input.spec.ts index 1ceed4254c5..79528fabffe 100644 --- a/tests/e2e/restart-restore-terminal-input.spec.ts +++ b/tests/e2e/restart-restore-terminal-input.spec.ts @@ -239,7 +239,6 @@ test('restored pane recovers input after the daemon un-wedges', async (// oxlint const second = await session.launch() secondApp = second.app - await settleRestoredLaunch(second.page) // Field-fidelity check, not a hard gate: does the pane paint restored // content while its PTY attach cannot complete? That visible-but-dead @@ -258,6 +257,8 @@ test('restored pane recovers input after the daemon un-wedges', async (// oxlint } stoppedDaemonPid = null + // Session readiness requires a daemon response; resume it before waiting for restoration. + await settleRestoredLaunch(second.page) await expectRestoredPaneAcceptsInput( second.page, `daemon wedged during relaunch (painted while wedged: ${paintedWhileWedged}, ` + From 54a8afc91de07e53e3d1de3791dc1c5ffe709f9b Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:27:29 -0400 Subject: [PATCH 23/23] fix(orchestration): typed error codes for dispatch and worker-start refusals (#18902) * fix(orchestration): typed error codes for dispatch and worker-start refusals orchestration dispatch (and worker-start, which composes it) surfaced task not found, task not ready, and inject rejected as the same bare runtime_error, so an agent reading the receipt could not choose between creating the task, waiting on dependencies, or picking another terminal. Add task_not_found (data.taskId), task_not_ready (data.status, data.unmetDependencies), and inject_rejected (data.terminal, data.reason), each carrying data.nextSteps so every shipped CLI already prints the recovery. worker-start's not-ready refusal moves from task_not_startable to task_not_ready with the same detail. runtime_error stays for genuinely unexpected failures. Proven red-first from RpcDispatcher through the CLI's own failure formatting, plus an SSH bridge test that the host CLI's typed refusal relays unchanged. * test(orchestration): load CLI formatter at runtime in the dispatch-code test The composite node typecheck (config/tsconfig.node.json without --composite false, as CI runs it) rejects a static import of src/cli from a main test with TS6307. Load the formatter and error class dynamically behind narrow structural types, as the CLI/runtime boundary test does. * fix(orchestration): keep task_not_startable and split the CLI-format proof Review on #18902: - Drop task_not_ready. worker-start already published task_not_startable for a not-ready Task, so renaming it would change an existing receipt value under old clients. dispatch now emits task_not_startable too (it was a bare runtime_error before, so this is purely additive), with the new data.status / data.unmetDependencies / data.nextSteps. - Move the refusal receipts (code, message, data) into src/shared/orchestration-dispatch-refusal-contract.ts so the runtime emits them and the CLI test formats the identical envelope. The RPC test under src/main asserts toEqual against the contract; the new src/cli/orchestration-dispatch-refusal-format.test.ts feeds those same receipts to formatCliError / reportCliError. Neither tsconfig widens and the composite typecheck CI runs is clean. * fix(orchestration): keep published refusal messages and type the DB claim guards Codex review of #18902: - Every call site keeps the exact message it published on main ("Task not found: ", "only a ready Task can start.", "cannot retry from Dispatch"); the shared contract now takes the message per site and only owns the code and data. Baseline strings are pinned as literals. - createDispatchContext's own missing/non-ready guards, including the atomic-claim loser, now emit the same typed receipt instead of a bare Error, so a dispatch that races a status change no longer flattens to runtime_error. Covered by a dispatcher-level race test. - Invalid --retry-of keeps task_not_startable but now carries status, unmetDependencies, retryOf, and a retry-specific next step. - Dependency recovery text distinguishes waiting on running deps from retrying/unblocking failed ones. - CLI test adds an unknown-code case so the old-client claim rests on an assertion, not a comment; SSH test asserts exact stdout. - Guide table narrowed to the covered preflight cases; occupancy stays runtime_error and is named as such. --- skill-guides/orchestration.md | 9 + src/cli/bundled-skill-guides.ts | 2 +- ...hestration-dispatch-refusal-format.test.ts | 77 ++++++ .../dispatch-context-store.ts | 18 +- .../worker-dispatch/worker-dispatch-start.ts | 18 +- .../orchestration-worker-dispatch-db.test.ts | 7 +- .../orchestration/task-dispatch-refusal.ts | 61 +++++ src/main/runtime/rpc/errors.ts | 1 + ...orchestration-dispatch-error-codes.test.ts | 242 ++++++++++++++++++ .../methods/orchestration-dispatch-methods.ts | 24 +- ...estration-inject-rejection-message.test.ts | 31 --- .../orchestration-inject-rejection-message.ts | 16 -- .../orchestration-tasks-dispatch.test.ts | 2 +- .../rpc/methods/orchestration-workers.ts | 9 +- ...e-cli-dispatch-refusal-passthrough.test.ts | 64 +++++ ...stration-dispatch-refusal-contract.test.ts | 67 +++++ ...orchestration-dispatch-refusal-contract.ts | 102 ++++++++ 17 files changed, 676 insertions(+), 74 deletions(-) create mode 100644 src/cli/orchestration-dispatch-refusal-format.test.ts create mode 100644 src/main/runtime/orchestration/task-dispatch-refusal.ts create mode 100644 src/main/runtime/rpc/methods/orchestration-dispatch-error-codes.test.ts delete mode 100644 src/main/runtime/rpc/methods/orchestration-inject-rejection-message.test.ts delete mode 100644 src/main/runtime/rpc/methods/orchestration-inject-rejection-message.ts create mode 100644 src/main/ssh/ssh-remote-cli-dispatch-refusal-passthrough.test.ts create mode 100644 src/shared/orchestration-dispatch-refusal-contract.test.ts create mode 100644 src/shared/orchestration-dispatch-refusal-contract.ts diff --git a/skill-guides/orchestration.md b/skill-guides/orchestration.md index 0878532c447..eab866f13d0 100644 --- a/skill-guides/orchestration.md +++ b/skill-guides/orchestration.md @@ -180,6 +180,15 @@ Dispatch rules: - After 3 consecutive failures on one task, the dispatch context circuit-breaks and the task is marked failed. - Use `task-list --brief --json` for coordinator sweeps; it collapses whitespace and caps each echoed spec at 160 characters (`spec_truncated` marks shortened rows). Omit `--brief` when the full spec is required, or when an older CLI rejects it as an unknown flag. +`dispatch` and `worker-start` refuse the following preflight cases with a stable `error.code`; read it before choosing a recovery, and treat `error.data.nextSteps` as the exact recovery text. Older hosts may omit `data`, so treat every field as optional. + +| Code | Meaning | Recovery | +| -------------------- | --------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------ | +| `task_not_found` | No Task with that id, or not in the bound Run (`data.taskId`, `data.runId`) | Check `task-list --json`; create the Task with `task-create` if it does not exist | +| `task_not_startable` | Task cannot start now: not `ready`, or invalid `--retry-of` (`data.status`, `data.unmetDependencies`, `data.retryOf`) | Wait for running dependencies with `check --wait`; retry or unblock failed ones; inspect `dispatch-show` if already dispatched | +| `inject_rejected` | `--inject` refused because no recognized agent runs in the target (`data.terminal`, `data.reason`) | Start a recognized agent there or pick another terminal; or dispatch without `--inject` and use `terminal send` | +| `runtime_error` | Any other failure, including a target terminal that already owns an active Dispatch | Read the message, inspect state, and do not retry unchanged | + ## How deep workers can nest A dispatched worker normally cannot dispatch sub-workers. Attempting it fails with diff --git a/src/cli/bundled-skill-guides.ts b/src/cli/bundled-skill-guides.ts index 06aae7bdd7d..5e68efbe8da 100644 --- a/src/cli/bundled-skill-guides.ts +++ b/src/cli/bundled-skill-guides.ts @@ -30,7 +30,7 @@ const ORCA_LINEAR_MARKDOWN = "---\nname: orca-linear\ndescription: >-\n Use Orc const ORCA_PER_WORKSPACE_ENV_MARKDOWN = "---\nname: orca-per-workspace-env\ndescription: >-\n Set up, review, debug, or validate Orca per-workspace environment recipes —\n on-demand, disposable runtimes (cloud sandboxes, VMs, or local) created fresh\n for each workspace. Covers first-time setup (provider prerequisites, the\n reusable base snapshot, the coding-agent auth snapshot, credentials, and\n state), not just the per-workspace lifecycle scripts. Use to stand up\n per-workspace environments, fix an `environmentRecipes` entry in `orca.yaml`, scaffold\n provider lifecycle scripts, or resolve an `orca vm recipe doctor` failure.\n---\n\n# Per-Workspace Environments\n\nHelp a user stand up and maintain a repo-owned per-workspace environment recipe end to end. Each\nworkspace gets its own on-demand, disposable runtime (a cloud sandbox, a VM, or a local one),\ncreated fresh and torn down after.\n\nOrca is a **thin wrapper**: you guide, detect, and scaffold; you never own the user's cloud account,\nbilling, images, or credentials.\n\n- **You DO:** sequence the setup, detect what's detectable (provider CLI present/logged-in? recipe\n present? `doctor` passing?), scaffold provider-templated scripts the user fills in, drive the slow\n snapshot/auth phases with the user, and always show the next action.\n- **You DO NOT:** create accounts, choose plans/regions, invent org/project/scope ids, store or print\n secrets, or run anything that spends money without an explicit user OK.\n\nFirst-time setup has **four phases before the per-workspace recipe runs** — easy to miss, so walk\nthem in order:\n\n1. **Prerequisites** — cloud account, provider CLI, scope/project, plan limits, git token (§2).\n2. **Base snapshot** — reusable image: tools + repo + headless build, snapshotted once (§3).\n3. **Agent-auth snapshot** — boot the base, run interactive device-auth, re-snapshot (§4).\n4. **State** — thread snapshot id / scope / project / port between phases via a state file (§6).\n\nThen the **per-workspace contract** (create/suspend/resume/destroy) runs fast (§8).\n\n**The one branch that shapes everything — connection mode:** **Orca-server** (`create` runs `orca serve`\nin the env and emits a `pairingCode`; §7c/§7f) vs **SSH** (`create` runs no server and emits a\n`connection.type:\"ssh\"` block Orca dials into; §7g/§7h). Settle this first — it changes the `create`\noutput shape and half the templates.\n\nKeep Orca's checkout behavior unchanged by default: omit `checkoutMode`, emit schema version 1, and\nlet Orca create a linked worktree. Only use `checkoutMode: provisioned-root` when the user explicitly\nwants one ephemeral machine to clone the finished workspace itself. This niche mode currently requires\ndirect SSH, an ordinary non-bare/non-sparse primary checkout at `projectRoot`, and schema version 2.\n\n**Quick-start (happy path):** interview the user (connection mode Orca-server vs SSH, provider, agent CLI,\ngit auth — §1.2) + read the provider's CLI docs → scaffold `scripts/orca-vm/` from §7 → run the\nbase-snapshot script, then the auth script (you invoke these by hand; not via `orca.yaml`) → wire\n`environmentRecipes` in `orca.yaml` → `orca vm recipe doctor --json` (free) → then the `--provision`\nself-test loop (§9) until it passes.\n\n---\n\n## 1. Setup workflow\n\nDrive these with the user. **[CHECKPOINT]** steps need explicit confirmation — they spend money, take\na long time, or need the user at the keyboard. Never create an Orca workspace or commit unless asked.\n\n1. **Inspect the repo** for an existing `environmentRecipes` entry, `scripts/orca-vm/`, a state file, or setup\n notes. If a working recipe exists, jump to Doctor (§9) instead of rebuilding.\n2. **Interview the user up front** — gather these choices and confirm them back before scaffolding\n anything. Don't pick for them (§11); don't guess.\n - **Connection mode:** how Orca attaches to the environment — an **Orca server** (the VM runs\n `orca serve` and Orca pairs over its pairing URL; worked example §7f) or **SSH** (Orca connects to\n the host over SSH; §7g). This decides the recipe's connection shape, so settle it first.\n - **Checkout ownership:** do not ask by default. Only when the user requires the environment to\n create the exact final checkout, confirm `provisioned-root` and direct SSH; otherwise omit it.\n - **Provider:** Vercel Sandbox, Fly, Modal, an existing SSH host, … For non-obvious providers, also\n ask scope/project/region and plan limits (§2). Then **read that provider's CLI/SDK docs** (or\n ` --help`) before scaffolding — you need its exact create/exec/snapshot/remove verbs.\n If a provider advertises `ssh`, verify whether it exposes a real dialable SSH target\n (host/port/user/key or proxy command) or only a provider-mediated interactive shell; Orca SSH mode\n needs the former.\n - **Coding-agent CLI + account:** which agent runs in the VM (`codex`, `claude`, …) and that the user\n has an account for it — it gets logged in during the Phase-3 auth snapshot (§4).\n - **Git auth:** the token source for cloning a private repo (`GH_TOKEN`/`GITHUB_TOKEN` or `gh auth\ntoken`; §5).\n3. **Check prerequisites (§2)** — detect the provider CLI + auth and confirm the items above are in\n place before any paid step.\n4. **Scaffold scripts + state file** from §7 (worked Vercel example: §7f; SSH host: §7g; Docker SSH:\n §7h; Windows: §7i), filling in the provider's real commands. Make them executable.\n5. **[CHECKPOINT] Build the base snapshot (§3)** — paid, slow.\n6. **[CHECKPOINT] Authenticate the agent (§4)** — interactive; the user follows a URL/code. **You cannot\n drive this step** — you run commands non-interactively, so there's no TTY for `docker exec -it` /\n `ssh -t` to prompt against. The **user** runs the Phase-3 login in their own terminal (or via the\n Claude Code harness bang-prefix — `! `, with the required space after `!`); you scaffold and drive\n the non-interactive phases around it. After kicking it off, **ask the user to report back once the login\n finishes** — you can't observe it completing, and you need that confirmation before resuming the\n non-interactive steps (base/auth commit, doctor, provision).\n7. **Wire the recipe** so `orca.yaml` points create/suspend/resume/destroy at the scripts (§8). The\n workspace composer reads `environmentRecipes` from the project's primary checkout of `orca.yaml`, **not** from\n a feature branch or worktree. So a recipe added only on a branch won't appear as a \"Run on\" option\n until that `orca.yaml` change is committed and merged to the project's primary branch. Tell the user\n this up front: `doctor`/`--provision` validate the scripts from the working copy on any branch, but\n creating a workspace from the recipe in the picker needs it on primary.\n8. **Dry-run doctor** — `orca vm recipe doctor --repo-path --json` (free, static; §9).\n Fix every failure before going live.\n9. **[CHECKPOINT] Live self-test** — get the user's OK once, then run\n `orca vm recipe doctor --provision --json` as a loop: it runs create → validates →\n destroys, and on failure returns a full transcript. Read it, fix the scripts, and re-run yourself until\n it passes (§9). Spends cloud money; the one approval covers the loop.\n10. **[CHECKPOINT] Optional workspace test** — only if asked: create a workspace via the picker, then\n verify sleep/wake/delete.\n\n---\n\n## 2. Phase 1 — Prerequisites\n\nThe user's responsibility; verify what's verifiable, ask for the rest, invent nothing. State which\nitems you verified vs. which the user asserted.\n\n- **Connection mode** (Orca server vs SSH) confirmed with the user — see §1 step 2; it shapes the recipe.\n- **Cloud account + plan** that allows sandboxes/VMs. Ask.\n- **Provider CLI installed + authenticated** — detect (`command -v `), check auth (e.g.\n `vercel whoami`). If missing, point at the provider's docs; don't log them in.\n- **Scope / project / region** the sandboxes live under. Ask; flows into every script via state.\n- **Plan / timeout / RAM caps.** Record them — e.g. Vercel Hobby caps sandbox timeout at **45m**,\n which limits both the base build and per-workspace runtime (see §10).\n- **Git token for private repos** (`GH_TOKEN`/`GITHUB_TOKEN`, or the provider's git auth; can fall back\n to `gh auth token`). See §5.\n- **Coding-agent CLI choice** (`codex`, `claude`…) and that the user has an account — it gets\n authenticated into the VM in Phase 3.\n\n---\n\n## 3. Phase 2 — Base snapshot (the reusable image)\n\nBuild **once**, snapshot, and every workspace boots from it in seconds instead of rebuilding.\nProvisioning + building takes a while (often ~20–30 min), so it runs behind a checkpoint. The script\nshape is §7a; key points:\n\n- Build the **headless Electron main only** (not the renderer) so it fits in plan RAM.\n- Use the VM image's package manager (`apt`/`dnf`/`apk`, per the base distro — not the provider brand).\n- Clone with the git token via `GIT_ASKPASS` (§5).\n- **Trap errors and remove the half-built sandbox** so a crash doesn't leave a paid resource running.\n- **Never snapshot a machine on which the Orca runtime has already run.** The first `orca serve` creates\n the runtime's user-data dir, and everything in it gets baked into the image and shared by every VM\n booted from it: the pairing keypair and device-token registry (`orca-devices.json`,\n `orca-e2ee-keypair.json`), `agent-session-authority.key`, and the build box's logs, terminal history\n and orchestration db. Confirmed: two VMs from one such snapshot emitted **identical `deviceToken` and\n `pairedDeviceId`**. Snapshot **before** the runtime has ever run, or delete the resolved user-data\n directory first: `orca_user_data_path=\"${ORCA_USER_DATA_PATH:-${XDG_CONFIG_HOME:-$HOME/.config}/orca}\"; rm -rf -- \"$orca_user_data_path\"`.\n This matches Orca's Linux precedence for custom and default paths; deleting a named file list will\n drift as Orca adds state.\n- Snapshot the stopped sandbox, parse the snapshot id, and write it + scope/project/port/repo to state.\n\n---\n\n## 4. Phase 3 — Agent-auth snapshot (interactive)\n\nThe base snapshot has the agent CLI installed but **not logged in**, and per-workspace VMs are\nephemeral — so authenticate once and bake it into a second snapshot layer. Script shape is §7b:\n\n1. Boot a sandbox from the base `snapshotId` (from state).\n2. Run the agent's login **interactively** (`--interactive --tty`); the user completes the URL/code in\n their browser. On a **headless VM this must be the device-auth flow** (e.g. `codex login --device-auth`),\n **not** plain `codex login`: the default OAuth login starts a loopback callback server on a container\n port the host browser can't reach, so it hangs. Device-auth instead prints a URL + code the user opens\n on the **host**.\n3. Verify login; **refuse to snapshot an unauthenticated VM.** Prefer the status command's **exit code**\n (most agent CLIs exit non-zero when unauthenticated). If you grep instead, agent status often goes to\n **stderr** (e.g. `codex login status` prints \"Logged in using ChatGPT\" there), so **fold stderr first**\n (`... 2>&1 | grep …`) and match the agent's **exact success line** — never `grep -qi 'logged in'`, which\n also matches \"**not** logged in\" and would commit an unauthenticated image.\n4. Re-snapshot, parse the new id, and overwrite `snapshotId` in state to the authenticated image\n (recording `authSourceSnapshotId`). Remove the auth sandbox.\n\n**You can't drive step 2 yourself** (you run commands non-interactively — no TTY). The **user** runs it in\ntheir own terminal, or via the Claude Code harness bang-prefix (`! `, with the required space after\n`!`). You scaffold/boot the sandbox and run steps 3–4, but **you cannot observe the interactive login\nfinishing** — so **ask the user to tell you when it's done** before you verify and re-snapshot.\n\nThis layer inherits §3's rule: if you started `orca serve` on the base or auth sandbox to smoke-test it,\ndelete the runtime's user-data dir (`~/.config/orca` on Linux) before re-snapshotting, or every workspace\nbooted from this image shares one pairing identity and one `agent-session-authority.key`.\n\nIf the agent's credentials are short-lived, warn that the snapshot may need periodic re-auth (§10).\n\nFor disposable runtimes, do **not** treat a host agent config directory (for example `~/.codex`) as the\nauth snapshot by bind-mounting or copying it wholesale. Agent homes often contain sqlite state, hook\napproval state, caches, logs, and host-specific env/config. Instead, authenticate/configure the agent\ninside the disposable runtime and snapshot/commit that runtime layer.\n\n---\n\n## 5. Credentials\n\n- **Never** commit secrets or put them in `userData`, recipe JSON, comments, docs, or the state file.\n- **Git token:** read from env (`GH_TOKEN`/`GITHUB_TOKEN`), falling back to `gh auth token`. Pass to the\n VM only via the provider's ephemeral `--env`. Inside the VM, use a `GIT_ASKPASS` helper with\n `x-access-token` (not the token in the clone URL) and `GIT_TERMINAL_PROMPT=0` so a missing token fails\n fast instead of hanging. When you write the helper from inside `bash -lc` under `set -u`, escape the\n positional arg and the token (`\\$1`, `\\$GH_TOKEN`) so they land **literally** and resolve at git-runtime\n — an unescaped `$1` aborts with \"unbound variable\", and a literal `$GH_TOKEN` keeps the real token out of\n the written file. `rm -f` the helper after the clone/fetch.\n- **Provider auth:** rely on the provider CLI's logged-in session, not checked-in keys.\n- **Agent auth:** lives in the authenticated snapshot (Phase 3) — never a file you write or commit.\n- State holds only **non-secret** wiring (snapshot ids, scope, project, port, repo url/ref).\n\n---\n\n## 6. State file\n\nA repo-local JSON file (e.g. `scripts/orca-vm/-state.json`) threads non-secret values between\nphases. Each script resolves values as **env var → state → built-in fallback**, and merges its outputs\nback. Phase 2 writes the base `snapshotId`; Phase 3 overwrites it with the authenticated snapshot;\nper-workspace `create` boots from `snapshotId`.\n\n```json\n{\n \"baseName\": \"orca-base\",\n \"snapshotId\": \"snap_authenticated_image_id\",\n \"authSourceSnapshotId\": \"snap_base_image_id\",\n \"scope\": \"\",\n \"project\": \"\",\n \"port\": 7331,\n \"repoUrl\": \"https://host/org/repo.git\",\n \"repoRef\": \"main\",\n \"projectRoot\": \"/abs/path/on/remote/repo\"\n}\n```\n\n---\n\n## 7. Script templates (provider-agnostic shapes)\n\nScaffold under `scripts/orca-vm/`. These are **shapes** — fill in the provider's real commands. All\nreserve stdout for the final JSON and log progress to stderr. Include a shared `json_value ` /\n`env_value ` reader (env → state → fallback) in each.\n\n**Where each script runs:**\n\n- **Local-side** (`create`/`suspend`/`resume`/`destroy` + the base-snapshot/auth scripts the user\n invokes) runs **on the user's desktop**, so it must run on their OS. macOS/Linux: `#!/usr/bin/env\nbash`, `set -euo pipefail`, quoted paths. **Windows:** a bare `.sh` won't run — scaffold `.ps1`/`.cmd`\n or require WSL/Git-Bash and point `orca.yaml` at the right launcher.\n- **Remote-side** (commands you `exec` _inside_ the Linux VM) always runs in the VM's Linux shell, so\n bash is fine there regardless of the user's OS.\n\n### 7a. Base-snapshot (`-base-snapshot.sh`) — Phase 2\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve base_name/repo_url/repo_ref/project_root/port/scope/project/timeout (env→state→fallback)\n# resolve gh token: GH_TOKEN | GITHUB_TOKEN | `gh auth token`\n# 1. provision a sandbox (timeout/vcpus/published port/snapshot retention); trap: remove on error\n# 2. remote exec (long timeout): install pkgs + gh + corepack/pnpm + agent CLI;\n# clone with GIT_ASKPASS(token); write headless main-only build config;\n# dev setup; pnpm install; build CLI; build headless electron main; smoke-check tools\n# 3. snapshot stopped sandbox; parse snapshot id (fail if unparseable)\n# 4. merge { baseName, snapshotId, projectRoot, repoUrl, repoRef, port, scope, project } into state\n# print only the state JSON to stdout\n```\n\nWorked Vercel commands for this phase are in §7f. You run this script by hand (not via `orca.yaml`),\nafter exporting the first-run inputs the state file doesn't have yet — e.g. provider scope/project, the\nrepo URL/ref, and a git token (`GH_TOKEN`); later runs read them back from state.\n\n### 7b. Auth (`-base-auth.sh`) — Phase 3\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# read source snapshot from state.snapshotId (fail if absent); auth_name=\"${base_name}-auth\"\n# 1. boot sandbox from source snapshot; trap: remove on error\n# 2. INTERACTIVE/TTY remote exec: agent login — user completes URL/code. Headless VM: MUST use the\n# device-auth flow (e.g. `codex login --device-auth`) — plain OAuth login binds a loopback callback\n# port the host can't reach and hangs. User runs this themselves (you have no interactive TTY); ask\n# them to report back when it's done before continuing.\n# 3. verify login, then refuse to snapshot if not logged in. Prefer the status command's EXIT CODE (most\n# agent CLIs exit non-zero when unauthenticated) over string-matching. If you must grep, fold stderr\n# first (`status 2>&1 | grep …` — many agents print the success line there) and match the agent's exact\n# success line; never `grep -qi 'logged in'`, which also matches \"not logged in\". Codex example: §7f.\n# 4. snapshot; parse new id\n# 5. merge { snapshotId:, authSourceSnapshotId: } into state; remove auth sandbox\n# print only the state JSON to stdout\n```\n\n### 7c. Create (`-create.sh`) — per workspace\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# read authenticated snapshotId/scope/project/port/repo*/project_root (env→state→fallback)\n# fail clearly if snapshotId is missing (point back to Phases 2–3)\n# name = orca-${ORCA_RECIPE_ID}-${ORCA_VM_INSTANCE_ID} (sanitized, length-capped)\n# 1. boot sandbox from snapshotId with a published port; capture the public URL → pairing address\n# (an externally reachable wss:// URL); trap: remove sandbox on error\n# 2. remote exec: ensure repo at desired commit; rebuild only if commit changed (cache marker)\n# 3. remote exec: start orca serve in the background and read the recipe JSON it writes (see below)\n# 4. print serve's JSON to stdout, optionally enriched with userData:\n# { schemaVersion:1, pairingCode, projectRoot, userData:{ provider, resourceId:name, snapshotId } }\n```\n\n**The exact `orca serve` invocation and its output (verified — do not improvise the flags).** Inside the\nVM, run:\n\n```bash\norca serve \\\n --port \"$PORT\" \\\n --project-root \"$ABS_REPO_PATH_ON_REMOTE\" \\\n --pairing-address \"$EXTERNAL_WSS_URL\" \\\n --recipe-json\n```\n\n**Binary name:** in a VM built from source (the Phase-2 flow), run it as `pnpm exec orca-dev serve …`\nfrom the repo root — `orca-dev` is the in-repo entrypoint and is what the §7f example uses. Plain\n`orca serve …` is the same command when the built CLI is installed on the VM's PATH. The flags/output\nare identical either way.\n\nThere is **no `--host` flag**. `--project-root` must be an absolute directory on the remote. With\n`--recipe-json` the server **stays running** and prints exactly this single object to **stdout**, then\nkeeps serving:\n\n```json\n{\n \"schemaVersion\": 1,\n \"pairingCode\": \"\",\n \"projectRoot\": \"\"\n}\n```\n\n`pairingCode` is the pairing URL, already pointing at whatever you passed as `--pairing-address` — so set\n`--pairing-address` to the externally reachable address and **pass `pairingCode` through unchanged; never\nhand-rewrite it**. Because serve runs in the foreground and doesn't exit, redirect its stdout to a file\nand poll until that file parses as JSON (and bail if the process dies — dump its stderr log). Your\n`create` script then prints that JSON (optionally merging `userData`). Concrete pattern: §7f.\n\n### 7d. Suspend / resume / destroy — per workspace\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\npayload=\"$(cat)\" # Orca passes lifecycle JSON on stdin\nresource_id=\"$(node -e 'const d=JSON.parse(process.argv[1]); process.stdout.write(d.recipeResult?.userData?.resourceId ?? \"\")' \"$payload\")\"\n[ -n \"$resource_id\" ] || { echo \"No resource id in lifecycle payload\" >&2; exit 1; }\n# suspend: provider suspend \"$resource_id\"\n# resume: provider resume \"$resource_id\"; then RE-EMIT fresh recipe JSON (pairing may change)\n# destroy: provider remove \"$resource_id\" (or set destroy: none in orca.yaml)\n```\n\n### 7e. State file — scaffold with scope/project/repo filled in and snapshot ids empty (§6).\n\n### 7f. Worked example — Vercel Sandbox (all three phases)\n\nA real, working shape (the Vercel surface is a CLI: `vercel sandbox create|exec|snapshot|remove`). Adapt\nnames; verify flags against `vercel sandbox --help` for the user's CLI version before relying on them.\nThese ground §7a (base snapshot) and §7b (auth), which are otherwise generic skeletons.\n\n**Phase 2 — base snapshot (§7a):** provision → install tools + clone + headless build → snapshot.\n\n```bash\n# provision a fresh build sandbox (retain a couple of snapshots); trap-remove on error\nvercel sandbox create --name \"$base\" --runtime node24 --timeout 30m --vcpus 4 --publish-port \"$port\" \\\n --snapshot-expiration 30d --keep-last-snapshots 2 \"${vercel_args[@]}\" >&2\n# remote build (long timeout): install pkgs+gh+pnpm+agent CLI, clone with GIT_ASKPASS (write the helper\n# with LITERAL \\$1/\\$GH_TOKEN so they resolve at git-runtime, not write-time — see §5/§7f create — then\n# `rm -f /tmp/askpass.sh`), write the headless main-only build config (drop the renderer), dev setup,\n# build CLI + headless main, smoke-check\nvercel sandbox exec \"$base\" \"${vercel_args[@]}\" --timeout 25m --env \"GH_TOKEN=$gh_token\" … -- bash -lc '…build…' >&2\n# snapshot the STOPPED sandbox and parse the id from CLI output (fail if unparseable)\nout=\"$(vercel sandbox snapshot \"$base\" --stop --expiration 30d \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$out\" >&2\nsnapshot_id=\"$(printf '%s\\n' \"$out\" | sed -nE 's/.*(snap_[A-Za-z0-9]+).*/\\1/p' | tail -1)\"\n# merge { baseName, snapshotId, scope, project, port, repoUrl, repoRef, projectRoot } into state; print state JSON\n```\n\n**Phase 3 — agent-auth snapshot (§7b):** boot the base, log the agent in interactively, re-snapshot.\n(`codex` below is an example — substitute the user's chosen agent's login/status verbs, e.g. `claude`.)\n\n```bash\nvercel sandbox create --name \"$auth\" --snapshot \"$snapshot_id\" --timeout 30m --publish-port \"$port\" \"${vercel_args[@]}\" >&2\n# INTERACTIVE — the USER runs this in their own terminal (you have no interactive TTY) and completes the\n# URL/code on the HOST. --device-auth is MANDATORY on a headless VM: plain `codex login` binds a loopback\n# callback port the host browser can't reach and hangs. Ask the user to report back when login finishes.\nvercel sandbox exec --interactive --tty \"$auth\" \"${vercel_args[@]}\" -- bash -lc 'codex login --device-auth'\n# refuse to snapshot an unauthenticated VM — fold stderr, match codex's exact success line (§4)\nvercel sandbox exec \"$auth\" \"${vercel_args[@]}\" --timeout 30s -- bash -lc 'codex login status 2>&1' | grep -Eqi 'Logged in using ChatGPT|Logged in via device' \\\n || { echo \"agent not logged in; not snapshotting\" >&2; exit 1; }\nout=\"$(vercel sandbox snapshot \"$auth\" --stop --expiration 30d \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$out\" >&2\nnew_id=\"$(printf '%s\\n' \"$out\" | sed -nE 's/.*(snap_[A-Za-z0-9]+).*/\\1/p' | tail -1)\"\n# overwrite state.snapshotId = new_id, record authSourceSnapshotId = snapshot_id; remove the auth sandbox\n```\n\n**Per-workspace `create`** (the fast path):\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve from env→state→fallback: snapshot_id, scope, project, port, repo_url, repo_ref, project_root\nvercel_args=(); [ -n \"$scope\" ] && vercel_args+=(--scope \"$scope\"); [ -n \"$project\" ] && vercel_args+=(--project \"$project\")\n[ -n \"$snapshot_id\" ] || { echo \"snapshotId missing — run Phases 2–3 first\" >&2; exit 1; }\ngh_token=\"${GH_TOKEN:-${GITHUB_TOKEN:-$(command -v gh >/dev/null 2>&1 && gh auth token 2>/dev/null || true)}}\"\nrecipe_id=\"${ORCA_RECIPE_ID:-vercel-sandbox}\"\nrecipe_id=\"${recipe_id//./-}\" # Vercel names forbid dots.\ninstance_id=\"${ORCA_VM_INSTANCE_ID:-$(date +%s)}\"\nmax_recipe_id_length=$((128 - ${#instance_id} - 6)) # Preserve the unique instance suffix.\n[ \"$max_recipe_id_length\" -gt 0 ] || { echo \"ORCA_VM_INSTANCE_ID is too long for a Vercel sandbox name\" >&2; exit 1; }\nname=\"orca-${recipe_id:0:max_recipe_id_length}-${instance_id}\"\n\n# Arm cleanup BEFORE create so a failing create can't leak a half-built paid sandbox.\ncleanup_on_error() { [ \"$?\" -ne 0 ] && vercel sandbox remove \"$name\" \"${vercel_args[@]}\" >/dev/null 2>&1 || true; }\ntrap cleanup_on_error EXIT\n\n# 1. boot from the authenticated snapshot, publish the serve port\ncreate_output=\"$(vercel sandbox create --name \"$name\" --snapshot \"$snapshot_id\" \\\n --timeout 30m --publish-port \"$port\" \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$create_output\" >&2\n# Vercel prints the published https URL; derive the external wss:// pairing address from it\npublic_url=\"$(printf '%s\\n' \"$create_output\" | sed -nE 's#.*(https://[^[:space:]]+\\.vercel\\.run).*#\\1#p' | head -1)\"\n[ -n \"$public_url\" ] || { echo \"no published URL in create output\" >&2; exit 1; }\npairing_ws=\"${public_url/https:\\/\\//wss://}\"\n\n# 2. (remote) ensure the repo is at the right commit; rebuild only if the commit changed (cache marker)\nvercel sandbox exec \"$name\" \"${vercel_args[@]}\" --timeout 20m \\\n --env \"GH_TOKEN=$gh_token\" --env \"ORCA_PROJECT_ROOT=$project_root\" \\\n --env \"ORCA_REPO_URL=$repo_url\" --env \"ORCA_REPO_REF=$repo_ref\" \\\n -- bash -lc 'set -euo pipefail; cd \"$ORCA_PROJECT_ROOT\"; \\\n # Re-establish git auth for the private-repo fetch (why + full rationale: §5); else it hangs on a prompt.\n # Load-bearing escaping: \\$1 and \\$GH_TOKEN must land LITERALLY and resolve at git-runtime. Test after\n # any edit here — reformatting the nested printf/node quoting silently breaks the fetch or leaks the token.\n if [ -n \"${GH_TOKEN:-}\" ]; then \\\n printf \"%s\\n\" \"#!/usr/bin/env bash\" \"case \\\"\\$1\\\" in *Username*) echo x-access-token;; *Password*) echo \\\"\\$GH_TOKEN\\\";; esac\" > /tmp/askpass.sh; \\\n chmod 700 /tmp/askpass.sh; export GIT_ASKPASS=/tmp/askpass.sh GIT_TERMINAL_PROMPT=0; fi; \\\n git fetch origin \"$ORCA_REPO_REF\"; \\\n git checkout -B \"$ORCA_REPO_REF\" FETCH_HEAD; \\\n rm -f /tmp/askpass.sh; \\\n c=\"$(git rev-parse HEAD)\"; [ -f .orca-built ] && [ \"$(cat .orca-built)\" = \"$c\" ] || { \\\n pnpm install --prefer-offline && pnpm run build:cli && \\\n node config/scripts/run-electron-vite-build.mjs --config config/electron-vite.vm-serve.config.ts && \\\n printf \"%s\" \"$c\" > .orca-built; }' >&2\n\n# 3. (remote) start orca serve in the background, writing recipe JSON to a file; poll until it parses\nrecipe_json=\"$(vercel sandbox exec \"$name\" \"${vercel_args[@]}\" --timeout 60s \\\n --env \"ORCA_PORT=$port\" --env \"ORCA_PROJECT_ROOT=$project_root\" --env \"ORCA_PAIRING_ADDRESS=$pairing_ws\" \\\n -- bash -lc 'set -euo pipefail; cd \"$ORCA_PROJECT_ROOT\"; rm -f /tmp/orca-recipe.json /tmp/orca-serve.log; \\\n nohup pnpm exec orca-dev serve --port \"$ORCA_PORT\" --project-root \"$ORCA_PROJECT_ROOT\" \\\n --pairing-address \"$ORCA_PAIRING_ADDRESS\" --recipe-json >/tmp/orca-recipe.json 2>/tmp/orca-serve.log /dev/null 2>&1 && { cat /tmp/orca-recipe.json; exit 0; }; \\\n kill -0 \"$pid\" 2>/dev/null || { cat /tmp/orca-serve.log >&2; exit 1; }; sleep 0.25; \\\n done; cat /tmp/orca-serve.log >&2; echo \"serve recipe JSON timed out\" >&2; exit 1')\"\n\n# 4. print serve's JSON enriched with userData (single object on stdout)\nnode -e 'const p=JSON.parse(process.argv[1]); console.log(JSON.stringify({...p, schemaVersion:1,\n userData:{...p.userData, provider:\"vercel-sandbox\", resourceId:process.argv[2], snapshotId:process.argv[3]}}))' \\\n \"$recipe_json\" \"$name\" \"$snapshot_id\"\ntrap - EXIT\n```\n\n`suspend`/`resume`/`destroy` use `vercel sandbox stop|...|remove \"$resource_id\"` reading\n`userData.resourceId` from stdin (§7d). This is the **Orca-server** connection mode (the recipe emits a\npairing URL). If the user chose **SSH** in the §1 interview, use §7g instead.\n\n### 7g. Worked example — existing SSH host (SSH connection mode)\n\nSSH mode is **fundamentally different from §7c/§7f**, not a relabeling of them:\n\n- **`create` does NOT run `orca serve` and does NOT emit a `pairingCode`.** Orca itself connects to the\n host over its SSH relay, brings up the git + filesystem providers, and imports the repo. The script's\n only job is to make the host ready and **print SSH connection details** Orca will dial.\n- The result uses a `connection` block with `type: \"ssh\"` and a `target`, **not** the flat\n `pairingCode`/`projectRoot` shape. Exact shape (Orca rejects anything else):\n\n```json\n{\n \"schemaVersion\": 1,\n \"connection\": {\n \"type\": \"ssh\",\n \"projectRoot\": \"/abs/path/to/repo/on/host\",\n \"target\": {\n \"label\": \"my-box\",\n \"host\": \"192.0.2.10\",\n \"port\": 22,\n \"username\": \"ubuntu\",\n \"identityFile\": \"~/.ssh/id_ed25519\",\n \"jumpHost\": \"bastion.example.com\",\n \"proxyCommand\": \"cloudflared access ssh --hostname %h\",\n \"relayGracePeriodSeconds\": 0,\n \"portForwards\": []\n }\n }\n}\n```\n\n`label`, `host`, `port`, `username` are required; the rest are optional — omit any you don't need.\n\nFor an explicitly requested one-VM-per-workspace checkout, the create script must read\n`ORCA_RECIPE_RESULT_SCHEMA_VERSION`, `ORCA_REPO_URL`, `ORCA_REPO_REF`, `ORCA_REPO_REF_HEAD`, and\n`ORCA_REPO_BRANCH`. Use `ORCA_REPO_REF` to fetch the selected source, but create\n`ORCA_REPO_BRANCH` at the exact `ORCA_REPO_REF_HEAD` commit; resolving the symbolic ref again can race\nwith an upstream update. `ORCA_REPO_URL` and `ORCA_REPO_REF` are a matched fetch pair, including when\nthe desktop source uses multiple remotes. Return that primary checkout at `projectRoot` and emit the\nsame SSH result with:\n\n```bash\n[ -n \"${ORCA_REPO_REF_HEAD:-}\" ] || { echo \"missing pinned source commit\" >&2; exit 1; }\ngit fetch origin \"$ORCA_REPO_REF\"\ngit cat-file -e \"${ORCA_REPO_REF_HEAD}^{commit}\"\ngit checkout -B \"$ORCA_REPO_BRANCH\" \"$ORCA_REPO_REF_HEAD\"\n```\n\n```json\n{\n \"schemaVersion\": 2,\n \"checkoutMode\": \"provisioned-root\",\n \"connection\": {\n \"type\": \"ssh\",\n \"projectRoot\": \"/abs/repo\",\n \"target\": { \"label\": \"my-box\", \"host\": \"192.0.2.10\", \"port\": 22, \"username\": \"ubuntu\" }\n }\n}\n```\n\nFail if the requested schema is not `2`; do not silently fall back to the ordinary recipe shape.\n\n**Networking → which `target` fields to set** (how _your desktop_ reaches the box — there is no\n`orca serve` URL in SSH mode):\n\n- Public IP / DNS, or a Tailscale/VPN address → `host`; SSH port → `port` (usually 22).\n- Key auth → `identityFile` (add `identitiesOnly: true` if the agent has many keys).\n- Through a bastion → `jumpHost` (a `user@host` ProxyJump) **or** a full `proxyCommand` (e.g. an access\n proxy). Use one, not both.\n- A service port the workspace needs → add entries to `portForwards`.\n- `relayGracePeriodSeconds` (optional): how long Orca keeps the SSH relay alive after the workspace\n detaches before tearing it down; `0` = tear down immediately. Leave it off unless the user wants a\n reconnect grace window.\n\n**Toolchain & agent auth on a persistent (no-snapshot) host — do this ONCE, by hand, before wiring the\nrecipe** (there's no base image to bake; the host _is_ the base). Run the §7f Phase-2 install steps and\nthe §7f Phase-3 ` login --device-auth` **directly over SSH on the host** (interactive, e.g.\n`ssh -t user@host ' login --device-auth'`). After that the host stays ready across workspaces.\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve from env→state→fallback (default unset optionals to \"\"): ssh_username, host,\n# ssh_port (default 22), identity_file, jump_host, proxy_command, project_root, repo_url, repo_ref\n: \"${identity_file:=}\"; : \"${jump_host:=}\"; : \"${proxy_command:=}\" # avoid set -u aborts on optionals\ngh_token=\"${GH_TOKEN:-${GITHUB_TOKEN:-$(command -v gh >/dev/null 2>&1 && gh auth token 2>/dev/null || true)}}\"\nssh_target=\"${ssh_username}@${host}\"\nssh_opts=(-p \"$ssh_port\"); [ -n \"$identity_file\" ] && ssh_opts+=(-i \"$identity_file\")\n# Why: a fresh host's key isn't in known_hosts; a StrictHostKeyChecking prompt would HANG a\n# non-interactive create. Pre-add the key (or set the option) so it can't block.\nssh-keyscan -p \"$ssh_port\" \"$host\" >> \"$HOME/.ssh/known_hosts\" 2>/dev/null || true\n\n# 1. ensure the repo is present and at the right commit on the host (NO orca serve here)\nssh \"${ssh_opts[@]}\" \"$ssh_target\" \\\n \"GH_TOKEN='$gh_token' GIT_TERMINAL_PROMPT=0 bash -lc '\n set -euo pipefail\n [ -d \\\"$project_root/.git\\\" ] || git clone \\\"$repo_url\\\" \\\"$project_root\\\"\n cd \\\"$project_root\\\" && git fetch origin \\\"$repo_ref\\\" && git checkout -B \\\"$repo_ref\\\" FETCH_HEAD\n '\" >&2\n\n# 2. print the SSH connection block (NO pairingCode, NO orca serve). host/port/username tell Orca's\n# relay how to dial in; identityFile/jumpHost/proxyCommand/portForwards are emitted when set.\nnode -e 'const [host,port,user,idf,jh,pc,root]=process.argv.slice(1);\n const target={ label:\"per-workspace-host\", host, port:Number(port), username:user };\n if(idf) target.identityFile=idf; if(jh) target.jumpHost=jh; if(pc) target.proxyCommand=pc;\n // add target.portForwards=[...] here if the workspace needs forwarded service ports\n console.log(JSON.stringify({ schemaVersion:1, connection:{ type:\"ssh\", projectRoot:root, target } }))' \\\n \"$host\" \"$ssh_port\" \"$ssh_username\" \"$identity_file\" \"$jump_host\" \"$proxy_command\" \"$project_root\"\n```\n\n`suspend`/`resume`/`destroy`: on a persistent host there's usually nothing to tear down — set\n`destroy: none` and omit suspend/resume. (Orca still disconnects/reconnects its own SSH relay on\nsleep/wake/delete — that's separate from these scripts.)\n\nIf the SSH host is instead an **ephemeral/snapshot-capable VM** (your hypervisor, or a cloud VM with\nimage support), keep the §7f Phase-2/3 base-image model for provisioning, but still emit the\n`connection.type:\"ssh\"` block above instead of starting `orca serve`.\n\n### 7h. Worked example — local Docker SSH (SSH connection mode)\n\nLocal Docker can model an ephemeral SSH VM without cloud cost: build a base image with `sshd`, tools,\nrepo prerequisites, and the agent CLI; run an **interactive auth container** once; then `docker commit`\nthat container as the authenticated image used by per-workspace `create`.\n\nKey points:\n\n- Publish container SSH to a random localhost port (`-p 127.0.0.1::22`) and emit\n `connection.type:\"ssh\"` with `host:\"127.0.0.1\"`, that port, `username`, `identityFile`, and\n `identitiesOnly:true`.\n- Generate a repo-local SSH key if needed, but gitignore the private/public key files.\n- **Bake SSH host keys into the base image** (`ssh-keygen -A` at **build** time; at runtime only generate\n if absent). Ephemeral containers all present the **same** host key, so `known_hosts` on `127.0.0.1`\n doesn't churn as the published port rotates across workspaces (otherwise every container's freshly\n generated key collides on `localhost` and trips host-key-changed warnings).\n- The auth image is the Docker equivalent of Phase 3: the **user** runs the agent login **inside** the\n container (you can't drive it — you have no interactive TTY), configures proxy env/config, approves\n hooks, and you commit once they report it's done. On a headless container use the **device-auth** flow\n (§4). Verify login before committing — exit code, or fold stderr and match the exact success line (§4).\n- Do not bind-mount or copy the host's full agent home into the image. Let each container have writable\n agent state; only the committed auth image should carry reusable authenticated state.\n- If committing from an interactive shell, force the runtime entrypoint back to `sshd`:\n `docker commit --change='ENTRYPOINT [\"/usr/local/bin/orca-docker-ssh-entrypoint\"]' …`.\n- `destroy` should read `recipeResult.userData.resourceId` and run `docker rm -f \"$resource_id\"`.\n\nValidation before wiring/live use:\n\n```bash\ndocker image inspect \"$auth_image\" --format '{{json .Config.Entrypoint}}'\ndocker run -d --name \"$name\" -p 127.0.0.1::22 -e \"ORCA_SSH_PUBLIC_KEY=$pubkey\" \"$auth_image\"\ndocker ps -a --filter \"name=$name\"\ndocker logs \"$name\"\nssh -i \"$key\" -p \"$port\" -o IdentitiesOnly=yes user@127.0.0.1 'codex --version'\n```\n\nIf the container exits immediately, inspect logs before the cleanup trap removes it; a committed\ninteractive image with `ENTRYPOINT [\"bash\"]` is a common cause.\n\nAlso confirm the **host key is stable** across containers: the SSH `ssh -i … 127.0.0.1` dial should not\ntrigger a host-key-changed warning when a second container reuses the port. If it does, the host keys\nweren't baked into the base image (see the `ssh-keygen -A` point above).\n\n### 7i. Windows local-side scripts\n\nThe local-side scripts run on the user's desktop. On **Windows**, a bare `.sh` won't execute. Either\nrequire WSL/Git-Bash (and point `orca.yaml` at e.g. `bash ./scripts/orca-vm/.sh` via a `.cmd`\nlauncher), or scaffold PowerShell equivalents. Minimal PowerShell shape:\n\n```powershell\n#requires -Version 5\n$ErrorActionPreference = 'Stop'\n# resolve env→state→fallback; run the provider CLI / ssh the same way;\n# capture provider output; build the result object for the chosen mode and write ONE line of JSON to stdout.\n# Orca-server mode: @{ schemaVersion=1; pairingCode=$pairingCode; projectRoot=$projectRoot; userData=@{...} }\n# SSH mode: @{ schemaVersion=1; connection=@{ type=\"ssh\"; projectRoot=$projectRoot;\n# target=@{ label=$label; host=$host; port=$port; username=$user } } } (see §7g/§7h)\n($result | ConvertTo-Json -Compress -Depth 6)\n# progress/errors → Write-Error / the error stream, never stdout.\n```\n\nThe remote-side commands you run _inside_ the Linux VM stay bash regardless of the desktop OS.\n\n---\n\n## 8. Per-workspace recipe contract (the fast path)\n\nOnce the authenticated snapshot exists, this runs on every workspace create. Define recipes in\n`orca.yaml`:\n\n```yaml\nenvironmentRecipes:\n - id: cloud-sandbox\n name: Cloud Sandbox\n create: ./scripts/orca-vm/cloud-sandbox-create.sh\n suspend: ./scripts/orca-vm/cloud-sandbox-suspend.sh\n resume: ./scripts/orca-vm/cloud-sandbox-resume.sh\n destroy: ./scripts/orca-vm/cloud-sandbox-destroy.sh\n```\n\n`create` runs **locally from the repo root** and prints **one** JSON object to stdout. Its shape depends\non the connection mode chosen in §1:\n\n**Orca-server mode** — boot the env, start `orca serve` in it, and print serve's result:\n\n```json\n{\n \"schemaVersion\": 1,\n \"pairingCode\": \"orca-pairing-code-or-url\",\n \"projectRoot\": \"/absolute/path/to/repo/on/remote\",\n \"userData\": { \"provider\": \"example\", \"resourceId\": \"provider-resource-id\" }\n}\n```\n\nHere `pairingCode` (from `orca serve --recipe-json`) and `projectRoot` are required; `schemaVersion` (`1`)\nand `userData` are optional.\n\n**SSH mode** — do **not** run `orca serve`; print the `connection.type:\"ssh\"` block instead (full shape +\nworked script in §7g). `pairingCode` is **not** used in SSH mode.\n\n**Optional provisioned root** — only for direct SSH and only when explicitly requested. Add\n`checkoutMode: provisioned-root` to the recipe, require `ORCA_RECIPE_RESULT_SCHEMA_VERSION=2`, create\nthe requested `ORCA_REPO_BRANCH` at the pinned `ORCA_REPO_REF_HEAD` commit (use `ORCA_REPO_REF` only\nto fetch that commit) at the returned `projectRoot`, and emit schema version 2 with\n`checkoutMode: \"provisioned-root\"`. All recipes without this field retain the schema-v1 behavior above.\n\nLifecycle hooks (all run locally):\n\n- `create`: required. Prints recipe result JSON.\n- `suspend`: optional. Sleep; reads lifecycle payload on stdin.\n- `resume`: optional. Wake; reads payload on stdin and **prints fresh recipe JSON** (pairing may change).\n- `destroy`: optional unless `destroy: none`. Delete/cleanup; reads payload on stdin.\n\nStart Orca remotely with `orca serve --port \"$PORT\" --project-root \"$ABS_ROOT\" --pairing-address\n\"$EXTERNAL_WSS_URL\" --recipe-json` (exact flags + output in §7c). Set `--pairing-address` to the\nexternally reachable address so the emitted `pairingCode` is reachable; tunneling/port mapping is the\nscript's job.\n\nBackward compatibility: `command`→`create`, `cleanup`→`destroy`, `cleanup: none`→`destroy: none`.\nPrefer the lifecycle names.\n\n---\n\n## 9. Doctor and validation\n\nValidate in two stages — the cheap dry run first, then the live self-test.\n\n### Dry run (free, non-destructive) — always do this first\n\n`orca vm recipe doctor --repo-path --json` validates **static wiring only** — it does\n**not** boot anything. It checks: local-host execution (v1), repo path, recipe id exists,\ncreate/destroy/suspend/resume command paths resolve, suspend/resume are paired, and each script is\nexecutable (POSIX exec bit; skipped on Windows). Fix every failure here before spending any cloud money.\n\n### Live self-test (`--provision`) — diagnose and iterate yourself\n\n`orca vm recipe doctor --repo-path --provision --json` actually runs the recipe end\nto end: it executes `create`, validates the returned recipe JSON, then runs `destroy` to **tear the\nenvironment back down** (so the test leaves nothing running, as long as `destroy` works). It spends real\ncloud money, so get the user's OK **once** before starting — that one approval covers the whole loop\nbelow; do not re-ask before each run.\n\nOn failure, the JSON result includes a `provisionTranscript` with the **complete** captured output of\neach stage so you can self-diagnose without asking the user to relay logs:\n\n```json\n{\n \"ok\": false,\n \"checks\": [{ \"id\": \"recipe.provision\", \"status\": \"fail\", \"message\": \"…\" }],\n \"provisionTranscript\": {\n \"provision\": { \"exitCode\": 0, \"signal\": null, \"stdout\": \"…\", \"stderr\": \"…\", \"parseError\": \"…\" },\n \"destroy\": { \"exitCode\": 0, \"signal\": null, \"stdout\": \"…\", \"stderr\": \"…\" }\n }\n}\n```\n\n**Run it as a loop:** read `provisionTranscript.provision.stderr` / `.stdout` / `.parseError` (and\n`destroy.*`), fix the script, and re-run `--provision` until `ok` is `true` — iterating on your own\nrather than waiting for the user to paste errors. Common reads: a non-empty `stderr` with `exitCode 0`\nplus a `parseError` means `create` ran but printed something other than the single recipe-result JSON on\nstdout (often a stray `echo` — route it to stderr, see §10); a non-zero `exitCode` is a provider/script\nfailure described in `stderr`. Each stream is redacted and capped (head+tail) — large logs keep both the\nsetup context and the failure.\n\nThe self-test cannot see provider-side truth beyond what the scripts print, so still confirm: state has a\npopulated **authenticated** `snapshotId` (Phases 2–3 done), and `destroy` is implemented/tested (or\nexplicitly `none` — in which case the self-test won't tear down, so clean up manually).\n\nFor SSH recipes, also smoke-test the exact emitted target before declaring success: dial the host/port\nwith the identity/proxy settings, run `pwd`, verify the repo path, check the agent binary, and confirm\n`destroy` removes the provider resource/container. For Docker, inspect the auth image entrypoint and do a\nstartup-only `docker run` before the full clone/install path.\n\n---\n\n## 10. Failure modes\n\n- **Build exceeds plan timeout (e.g. Hobby 45m).** Use enough vCPUs and a timeout covering the build;\n else split work or use a higher plan. The cap also limits per-workspace runtime — surface it.\n- **Build exceeds plan RAM.** Build the **headless main only** (drop the renderer) — the biggest fitter.\n- **Private-repo clone hangs/fails.** Wrong/missing token. Use `GIT_ASKPASS` + `GIT_TERMINAL_PROMPT=0`\n so it fails fast instead of prompting.\n- **`GIT_ASKPASS` helper aborts the clone with \"`$1: unbound variable`\".** The `printf`/heredoc that writes\n the helper inside `bash -lc` under `set -u` expanded `$1`/`$GH_TOKEN` at **write** time. Escape them\n (`\\$1`, `\\$GH_TOKEN`) so they land literally and resolve at git-runtime; this also keeps the real token\n out of the file. `rm -f` the helper afterward (§5, §7f).\n- **Agent verified as \"not logged in\" despite a good login.** `codex login status` (and similar) print\n \"Logged in …\" to **stderr**; an stdout-only `grep` misses it. Prefer the status **exit code**; if you\n grep, fold stderr first (`status 2>&1 | grep …`) and match the exact success line — not `grep -qi\n'logged in'`, which also matches \"not logged in\".\n- **Headless agent login hangs.** Plain OAuth `login` starts a loopback callback server on a VM/container\n port the host browser can't reach. Use the **device-auth** flow (`login --device-auth`) — it prints a\n URL + code the user opens on the host.\n- **`known_hosts` host-key churn on local Docker.** Each ephemeral container regenerating its SSH host key\n collides on `127.0.0.1` as the published port rotates. Bake host keys into the base image at build time\n (`ssh-keygen -A`; runtime generates only if absent) so all containers share one stable key (§7h).\n- **Snapshot expired/evicted.** If `create` hits an unknown snapshot id, rerun Phases 2–3 and update\n `snapshotId`.\n- **Agent auth didn't persist.** Confirm `snapshotId` points at the **authenticated** snapshot; re-run\n Phase 3. Warn that short-lived tokens may need periodic re-auth.\n- **Agent auth copied from the host breaks.** Do not bind-mount/copy a full host agent home; sqlite\n files can be unwritable or host-specific, hooks may need approval again, and config may reference\n local-only env vars. Authenticate inside the runtime and snapshot/commit that layer.\n- **Docker auth image exits immediately.** Inspect `docker image inspect … .Config.Entrypoint` and\n `docker logs`. If the image was committed from an interactive shell, reset the entrypoint to the SSH\n entrypoint during `docker commit`.\n- **Leaked paid resource.** Every long script must trap errors and remove the sandbox it created.\n- **`create` emits non-JSON on stdout.** A stray `echo` corrupts the result — stdout is for the final\n JSON only; everything else to stderr. The `--provision` self-test surfaces this as `exitCode 0` + a\n `parseError` with the offending stdout in `provisionTranscript` (§9).\n\n---\n\n## 11. Boundaries\n\n- Don't create accounts, choose plans/regions, or invent scope/project/org/image/billing ids.\n- Don't invent or store credentials; no secrets in `userData`, state, comments, docs, or commits.\n- Don't run paid/long phases (base snapshot, auth, live test) without an explicit OK.\n- Don't hide provider errors behind generic messages — preserve actionable stderr.\n- Don't make Orca own provider lifecycle beyond invoking the configured scripts.\n- Don't commit or create an Orca workspace unless asked.\n" // oxfmt-ignore -const ORCHESTRATION_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Use Orca orchestration for structured multi-agent coordination: threaded\n messages, blocking ask/reply flows, task dispatch, worker_done/escalation\n waits, task DAGs, decision gates, or coordinator loops. Use `orca-cli`\n instead for full ownership handoffs, including requests phrased as \"hand\n off\", \"handoff\", \"handover\", \"give this to another agent\", or \"another\n worktree\" when the user did not explicitly ask to supervise, monitor, wait\n for results, or coordinate a DAG. Use `orca-cli` for terminal control,\n lightweight terminal prompts, shell commands, Orca worktree management,\n reading or waiting on terminals, and the Orca embedded browser. Use Computer\n Use for external browser windows, webviews, Orca app UI, or desktop UI\n outside Orca's embedded browser only when the task requires OS/window-level\n control such as focus, menus, dialogs, coordinates, or screenshots. Use\n `orca-cli` for Orca's embedded pages and a page-automation tool such as\n Playwright or CDP for external pages.\n---\n\n# Orca Inter-Agent Orchestration\n\nOrchestration is Orca's structured coordination layer for agent messages, task ownership, dispatch state, and worker completion tracking.\n\nUse this skill when coordination state matters. For lightweight terminal prompts or basic worktree/terminal/built-in-browser control, use `orca-cli`.\n\n## Tool Boundary\n\nIf a task says to use Orca orchestration, the coordinator must create or bind a Run, create the Task with `orca orchestration task-create`, then attach the worker with either the preferred `orca orchestration worker-start` composition or the low-level `orca orchestration dispatch --inject` path.\n\nDo not substitute non-Orca subagent tools, generic agent-spawn APIs, or chat-only parallel worker features. Those may create useful workers, but they do not create Orca task/dispatch provenance, injected lifecycle preambles, `worker_done` authority, or decision gates.\n\nBefore claiming a worker was orchestrated, verify the task/dispatch exists:\n\n```bash\norca orchestration task-list --json\norca orchestration dispatch-show --task --json\n```\n\nIf the work was accidentally run outside Orca orchestration, say so plainly. To repair provenance, rerun or revalidate the needed work through a fresh Orca terminal plus injected dispatch; do not retroactively describe the external worker as orchestrated.\n\n## When To Use\n\n- Send/reply/ask between agent terminals with persistent messages.\n- Dispatch structured tasks to workers and wait for `worker_done` or `escalation`.\n- Track task DAGs with dependencies.\n- Run coordinator loops or decision gates.\n\nDo not use orchestration merely because the user says \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", or asks for another worktree/agent/model/effort. Those are full ownership transfers unless the user explicitly asks to supervise, monitor, wait for worker completion/results, coordinate a DAG, use decision gates, or keep a blocking ask/reply loop.\n\n## Preconditions\n\n- `orca status --json` should show a running runtime.\n- `orca` must be on PATH (`orca-ide` on Linux).\n- The orchestration experimental feature must be enabled in Settings > Experimental.\n- `orca orchestration` commands are RPC calls to the running Orca runtime.\n\n## Contract Migration\n\nOrca adopts a live pre-update orchestration assignment into an ordinary Run. Adoption preserves the existing agent process, PTY/session, terminal handle, tab/leaf/pane, worktree or folder workspace, Task, and Dispatch; it never restarts or replaces the worker. The retired scheduler is not revived, and a newly created attempt uses the current grammar.\n\nTreat the authority label on injected or formatted messages as definitive:\n\n- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported command printed with the message, using the same CLI executable and arguments that the original prompt supplied.\n- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded, at-least-once cutover replay. Process it idempotently and acknowledge it only through the exact displayed guidance.\n- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or lifecycle action.\n- An unlabeled current message uses the current guide and current grammar.\n\nAn explicitly selected current Run, attested current Run binding, current Dispatch, or federated attachment takes precedence over legacy fallback. A retained adoption record alone never turns a current command into a legacy call.\n\nDatabase provenance, an old-looking terminal, or a legacy Run ID does not prove mutation authority. If the runtime cannot prove liveness, principal ownership, capability, or the exact legacy contract, it degrades to read-only inspection and must not fall back to local execution. Exact recovery may restore the already-live PTY once in its original inactive background tab. It must not spawn, write, signal, stop, switch, focus, split, or inject a terminal. Loss of lifecycle authority does not invalidate the existing assignment, process, or filesystem work.\n\nCompatibility retries have narrow guarantees. A pending ask, a reply, a final Dispatch settlement, and a consuming check have durable recovery identities. A-era heartbeat and escalation calls remain at-least-once across a manual A-to-B retry because identical later signals may be intentional. If an A-era ask may already have been answered, run the exact non-consuming recovery check printed by the runtime first; after its answer is printed and acknowledged, a new invocation with the same question creates a new question. Never guess among multiple identical question threads.\n\nWhen a compatibility or recovery command returns structured next-step arguments, run those exact arguments with the same CLI executable. The arguments intentionally omit the executable name so the guidance works with `orca`, `orca-ide`, `orca-dev`, or another configured Orca CLI command. Do not translate the command from memory, broaden its recipient, or retry it as a current mutation unless the returned guidance explicitly says to.\n\nOn packaged Windows, a legacy ask uses a two-step commit/resume protocol. The initial command durably commits the question, prints its exact `ask --resume ` command, and exits with launcher status `75`; it does not wait for the answer. Run that exact resume command after the launcher or update boundary. Resume is idempotent and read-oriented: it waits for the already-committed question and does not create another one. For a WSL process that received compatibility proof at launch, use the printed executable `orca-ide` WSL resume command so the same distro and packaged launcher authority are preserved; do not substitute a PATH-resolved local CLI. Older WSL processes that never received the hidden launch token remain lifecycle read-only after the update, even while their terminal and filesystem work continue.\n\nLegacy inspection remains available without consuming mail:\n\n```bash\norca orchestration run-list --json\n# run_legacy_local is an empty audit tombstone after adoption.\norca orchestration run-show --id run_legacy_local --json\n# In run-list, find the ordinary Run whose objective is:\n# \"Recovered orchestration work from a contract update\"\norca orchestration run-show --id --json\norca orchestration task-list --run --json\norca orchestration inbox --full --json\norca orchestration check --terminal --peek --format --json\norca terminal read --terminal --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\n```\n\nIf the original coordinator is unavailable or cannot prove its retained authority, a current coordinator may explicitly take over the adopted Run from its own live agent terminal:\n\n```bash\norca orchestration run-use --id --takeover-legacy --json\norca orchestration check --run --json\n```\n\nTakeover fences only the old coordinator, binds the current one, and moves pending worker mail into current Run Delivery. It is bound to the authenticated invoking terminal; `--from` cannot name another coordinator. Live legacy workers keep their original Tasks, Dispatches, processes, filesystems, and old prompt commands; their later questions, escalations, and completion reports route to the current coordinator. Do not use takeover while the original coordinator is still actively coordinating, because its later lifecycle mutations are rejected.\n\nDo not launch a replacement editor merely because the desktop app or runtime was updated. If adoption cannot prove continuing authority, keep the original worker as the only editor until it reaches a stable handoff point, then use a new current Dispatch in a conflict-free placement for any remaining work.\n\n## Ownership\n\nNew orchestration messages and tasks belong to one explicitly bound Run. A Run is only a durable namespace and coordinator inbox; it never schedules or places workers. Lifecycle authority comes from the active Dispatch, and terminal handles remain routing metadata rather than durable identity. Send `worker_done` and `heartbeat` from the worker's own terminal; Orca routes them to that Dispatch's Run.\n\nClassify inherited context before sending lifecycle messages:\n\n- Coordinated subtask: a live coordinator owns the DAG and waits on this dispatch. Follow the preamble exactly, including `worker_done`, heartbeat/status, `ask`, and `escalation`.\n- Full handoff means ownership transfer, not supervised dispatch. The original actor is not monitoring a DAG, so do not create lifecycle obligations unless the user explicitly asks you to supervise.\n- Classify requests containing \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"another agent\", or \"another worktree\" as full handoffs by default, even when the user names a custom model or reasoning effort.\n- Use supervised orchestration only when the user explicitly asks you to \"supervise\", \"monitor\", \"wait\", \"track completion\", \"wait for worker_done\", return results, coordinate a DAG, use a decision gate, or manage ask/reply flow.\n- Do not use `orca orchestration dispatch --inject` for full handoffs. It injects a coordinator preamble that tells the worker to send `worker_done`, heartbeat, and `ask` messages, then end its turn under the original terminal's dispatch lifecycle.\n- Do not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. Do not peek at terminal output after prompt delivery to monitor progress.\n- A review-only `worker_done` reports findings; it does not authorize coordinator file edits. After a review-only completion, synthesize findings, ask a decision gate if ownership is unclear, and dispatch or hand off fixes unless the user explicitly asked the coordinator to own fixes.\n- If the user's plan names a next owner agent (for example, \"then use opencode to create a PR\"), post-review corrections and PR prep belong to that named owner. The coordinator routes, synthesizes, asks decision gates when needed, and supervises; the named owner edits files and creates the PR.\n\nIf unclear, inspect orchestration state before sending lifecycle messages:\n\n```bash\norca orchestration task-list --json\norca terminal list --json\n# If inherited context includes a task id:\norca orchestration dispatch-show --task --json\n```\n\n## Messaging\n\n```bash\norca orchestration send --subject [--to ] [--from ] [--body ] [--type ] [--priority ] [--thread-id ] [--payload ] [--json]\norca orchestration check [--terminal ] [--ack ] [--peek|--all] [--types ] [--format] [--wait] [--timeout-ms ] [--json]\norca orchestration reply --id --body [--from ] [--json]\norca orchestration ask (--question |--resume ) [--options ] [--timeout-ms ] [--from ] [--json]\norca orchestration inbox [--limit ] [--json]\n```\n\nRules:\n\n- Omit `--from` unless impersonating another terminal; Orca auto-resolves it from the current terminal.\n- A coordinator `check` returns the bound Run's oldest FIFO Delivery (up to 50 messages) and replays that exact batch until `--ack `. Process every message before acknowledging; `check --ack --wait` acknowledges, checks, and waits in one operation.\n- Use `--peek` and `--all` only for read-only history/debugging. Type filters decide when a waiter wakes; the returned actionable Delivery is still the oldest full batch.\n- Use `dispatch:` for coordinator guidance to one supervised worker. Orca routes that stable address locally or through the connected-server relay; do not substitute a remote terminal handle.\n- Terminal handles remain appropriate for low-level pre-Dispatch messaging. Prefer `agentTerminalHandle` from the create response, fall back to `startupTerminal.handle` for older runtimes, then re-resolve with `orca terminal list --worktree ... --json` if missing or stale. Continue with the replacement handle only; never dual-send to old and new handles.\n- `terminal list --json` omits `visualLayouts` because handle recovery does not need topology. Add `--include-visual-layouts` only for explicit tab and pane inspection.\n- `orca orchestration check --peek --format --json` returns locally formatted unread mail without consuming it; it never writes to terminal input or remotely wakes another terminal. Use `orchestration dispatch --inject` to deliver a tracked task, or `terminal send` when an existing agent needs a free-form prompt.\n- While supervising workers manually, use `check --wait --types worker_done,escalation,question --timeout-ms ` instead of sleep/poll loops. Process the whole Delivery, reply to `question` messages with `orca orchestration reply --id --body --json`, then acknowledge and keep waiting.\n- `check --json` prints exactly one JSON document on stdout. While `--wait` blocks it also prints keepalive lines (`{\"_keepalive\":true,...}`) to stderr so you can tell the process is alive; those are never on stdout. Do not merge the streams before a parser — `check --wait --json 2>&1 | ` fails with \"Extra data: line 2\". Pipe stdout only.\n- Treat a `check --wait` timeout or `{count:0}` as a checkpoint, not a worker failure. Long coding tasks routinely run 15-60 minutes; keep using rolling waits unless you receive `worker_done`/`escalation`, the terminal exits or disappears, or the user explicitly asks you to stop.\n- Heartbeats and visible terminal activity mean the worker is alive, not done. Do not stop, close, kill, or restart a worker just because it has not produced a completion message yet.\n- Use `ask` when a worker needs a blocking answer from the coordinator; it defaults to the active Dispatch's Run. Timeout or disconnect leaves the question pending, so resume by its original message ID instead of asking again.\n- `check --wait` returns one bounded Delivery, not every future completion. Process every message, acknowledge it, then keep waiting until every expected Dispatch settles.\n- Group addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`, `@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`.\n- Message types include `status`, `dispatch`, `worker_done`, `merge_ready`, `escalation`, `handoff`, `question`, `decision_gate` (legacy/gates), and `heartbeat`.\n- Use group addresses only for messages that are genuinely useful to many terminals, such as `status` broadcasts or intentional fan-out questions. Do not send dispatch lifecycle messages to groups.\n- `worker_done` belongs to the active Dispatch and defaults to its Run mailbox; never target a group.\n- A valid `worker_done` for the active `taskId` + `dispatchId` marks the task and dispatch completed automatically. Do not follow it with `task-update --status completed`; reserve manual updates for explicit recovery or overrides.\n- `heartbeat` is also Dispatch-scoped. Include both IDs and omit `--to` so Orca uses the owning Run; use `status` for broad progress updates.\n\n## Tasks And Dispatch\n\nA Run is the namespace/inbox, a Task is the work item, and a Dispatch assigns one Task attempt to a terminal. Create or bind a Run once before the common loop.\n\n```bash\norca orchestration run-create --objective --json\norca orchestration task-create --spec [--deps ] [--parent ] [--json]\norca orchestration task-list [--status ] [--ready] [--brief] [--json]\norca orchestration task-update --id --status [--result ] [--json]\norca orchestration dispatch --task --to [--from ] [--inject] [--json]\norca orchestration dispatch-show --task [--json]\n```\n\nTask statuses: `pending`, `ready`, `dispatched`, `completed`, `failed`, `blocked`.\n\nDispatch rules:\n\n- `--inject` sends the task spec plus preamble into a recognized agent CLI so it can report `worker_done`.\n- If the target is a bare shell, omit `--inject`, dispatch for tracking if needed, then send the prompt manually with `orca terminal send --terminal --text --enter --json`.\n- After 3 consecutive failures on one task, the dispatch context circuit-breaks and the task is marked failed.\n- Use `task-list --brief --json` for coordinator sweeps; it collapses whitespace and caps each echoed spec at 160 characters (`spec_truncated` marks shortened rows). Omit `--brief` when the full spec is required, or when an older CLI rejects it as an unknown flag.\n\n## How deep workers can nest\n\nA dispatched worker normally cannot dispatch sub-workers. Attempting it fails with\n`nested_worker_depth_exceeded` and a message telling the worker to complete the task\nitself. Do that — do not try to route around it.\n\nThe limit is a number, not an on/off switch. `Settings -> Orchestration -> Nested worker depth`\nsets how many generations are allowed:\n\n- `1` (default): a coordinator dispatches workers; those workers do not dispatch.\n- `2`: workers may dispatch one further generation.\n\nDepth is counted from the terminal that issues the command, not from the Run. Creating a\nnew Run does not reset it — a worker that runs `run-create` then `worker-start` is still a\nworker, and still counted. This is the part that changed: the old behaviour rejected\nsub-dispatch only because a worker's terminal was not bound to a Run, so creating a Run was\nenough to slip past it.\n\nTwo limits worth knowing:\n\n- **It is a guardrail, not a security boundary.** A caller that declares another terminal's\n handle while its own launch evidence is unverifiable (an ordinary restored terminal, for\n example) can be counted as that terminal instead. Orca does not treat workers as hostile.\n- **It applies while a Dispatch is active.** After `worker_done`, or after a coordinator\n settles the task, the terminal is no longer a worker and is counted as a root again. The\n process may still be alive; that is the documented boundary, not an accident.\n\n## Preferred Supervised Worker Loop\n\nUse `worker-start` for the normal supervised path. It composes the existing worktree, terminal, readiness, and dispatch primitives while returning exact created/reused effects. Agents still choose placement and concurrency; Orca does not schedule workers or infer conflicts.\n\nCreate the Run and every independent Task first, then start all independent workers before waiting:\n\n```bash\norca orchestration run-create --objective \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration worker-start --task --worktree current --agent codex --json\norca orchestration worker-start --task --worktree current --agent claude --json\n```\n\n`current` and exact existing worktrees create a fresh agent terminal and do not rerun setup. Reuse an existing agent only with `--terminal `.\n\nFor a per-invocation Claude, Codex, or Cursor launch, pass an opaque provider model id with `--model`; add `--effort` only when that agent/model supports the level. These options apply only to fresh agent terminals, override general agent default arguments, and are reported under `launch.requested` and `launch.effective` in the receipt:\n\n```bash\norca orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`, and neither option can combine with `--terminal`. A connected worker server must advertise launch-preference support before Orca forwards either option.\n\nFor a new worktree, setup runs by default and agent-first creation reuses the returned startup agent terminal:\n\n```bash\norca orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n# Independent/top-level:\norca orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nSetup normally starts alongside the agent. Only a repository explicitly configured with `wait-for-setup` delays agent launch until setup succeeds. Use `--setup skip` or `--setup inherit` only for a concrete reason.\n\nRead the returned receipt before continuing: `ready` plus setup `running` is normal for start-immediately, while wait-for-setup returns setup `succeeded` before accepting task input. A failed or unknown start exits nonzero; inspect its `stage`, `effects`, and `residualResources` instead of guessing or automatically retrying. A wait-for-setup timeout can honestly leave setup `running`, which is not proof of failure.\n\nTo run the worker on another connected Orca server, add `--on `. The Run and Tasks remain authoritative on the current server; later commands route by Dispatch ID, so never repeat `--on`:\n\n```bash\n# Mac Run home -> Windows worker (the reverse is identical from a Windows Run home)\norca orchestration worker-start --task --on windows --worktree new-top-level --repo --name --agent codex --setup run --json\norca orchestration worker-show --dispatch --json\norca orchestration worker-read --dispatch --limit 50 --json\norca orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nRemote `current` and `new-child` are intentionally invalid because those words are ambiguous across servers. Use an exact discovered remote worktree selector or `new-top-level` with an explicit remote repo selector.\n\nThe follow-up is structured inbox mail, not prompt injection. The worker's next\n`orchestration check` receives it even when the Dispatch is on another connected Orca server.\n\n`worker-read` defaults to `--source auto`: Orca returns the exact hook-reported Codex, Claude, OpenClaude, or Grok transcript when it can prove the worker session, otherwise it returns bounded terminal output with `source: \"terminal\"` and a typed `fallbackReason`. Continue with the returned top-level `cursor`; it stays pinned to that exact source. If Orca reports `source_changed`, start a fresh read without the old cursor. Never supply or guess a provider session ID or transcript path.\n\nWait until every expected Dispatch settles, not for a fixed number of batches:\n\n```bash\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n# Process every message. For each accepted worker_done that is not immediately reused:\norca orchestration worker-release --dispatch --json\n# Acknowledge only after every message and required release decision is handled:\norca orchestration check --ack --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\nAfter processing each accepted `worker_done`, choose the terminal's next owner before you acknowledge the Delivery or wait again. If the same exact agent has an immediate follow-up Task, read the `worker.agent_terminal_handle` field of `worker-show --dispatch --json`, then run `orca orchestration worker-start --task --terminal --json` so Orca transfers cleanup ownership to the new Dispatch. Otherwise run `orca orchestration worker-release --dispatch --json`.\n\nRun `worker-release` after both succeeded and failed `worker_done` reports unless the user explicitly asked to keep that worker live. Release is post-completion cleanup, not cancellation: Orca first preserves inspectable output, then closes only the exact agent terminal owned by that settled Dispatch. Reused or pre-existing terminals, setup terminals, coordinators, active workers, user-taken-over terminals, and identities Orca cannot prove are retained. If the user explicitly asks to keep the live terminal for debugging, record that exception with `orca orchestration worker-retain --dispatch --json` instead of silently skipping cleanup. When the user is finished, the same Dispatch can be passed to `worker-release`, which clears the requested retention and releases the terminal.\n\nDo not release a worker because of a timeout, TUI idle state, heartbeat, status, question, escalation, or rejected/stale `worker_done`. If release returns `release_pending` or `release_unknown`, do not substitute `terminal close`; follow the exact recovery action in the receipt. A replayed Delivery may repeat `worker-release` safely.\n\nWorkers report exactly once using the IDs and capability injected by Orca; they do not supply Run/server/terminal identity:\n\n```bash\norca orchestration send --type worker_done --subject \"\" --body \"\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a,path/b\" --json\n# On failure, use --outcome failed; never encode failure only in prose.\n```\n\nA worker question defaults to its owning Run. Timeout leaves it pending:\n\n```bash\norca orchestration ask --question \"\" --options \"yes,no\" --timeout-ms 600000 --json\norca orchestration ask --resume --timeout-ms 600000 --json\n# Coordinator:\norca orchestration reply --id --body \"\" --json\n```\n\nRecovery is conditional, never a fixed destructive sequence:\n\n- The response was lost and named no Dispatch: run `orca orchestration request-show --request --json` first. It is read-only. `completed` means the mutation already took effect. `pending` means the original mutation is still running or Orca restarted before recording its outcome. For either state, replaying the original command with `--retry-request ` reuses the same operation identity so Orca can replay, join, or safely recover it without starting a separate duplicate. `absent` means this runtime holds no receipt under your caller identity and is not proof that nothing happened; inspect the affected state before deciding whether to retry.\n- `worker-show --dispatch ` says `ready`: keep waiting or read bounded output.\n- It proves `failed` or `stopped`: start a replacement with `worker-start --task --retry-of ` plus an explicit `--on`/`--worktree` and `--agent`/`--terminal` choice. Retry does not silently inherit placement.\n- It remains `outcome_unknown`: either `worker-stop --dispatch ` and inspect again, or explicitly `worker-abandon --dispatch ` while accepting that resources may still be live. Abandon performs no remote, process, or filesystem action.\n- `worker-stop` closes only the exact supervised agent terminal. It never deletes the worktree, setup terminal, configured tabs, or unrelated processes.\n\nLow-level `worktree create`, `terminal create`, and `dispatch --inject` remain valid recipes for custom argv or topology that `worker-start` does not express.\n\n`dispatch --inject` deliberately keeps an operator-started terminal unsupervised: it never creates a `worker_dispatches` row and `worker-stop`/`worker-abandon` never close that process. The dispatch context is still authoritative, so `worker-show`, `worker-read`, and `worker-list` report it as `unsupervised`; settled `worker-retain` and `worker-release` report `retained` with `no_owned_resource` and take no process action. Use `worker-start --terminal ` when supervision and worker lifecycle state are required.\n\n## Gates And Legacy Inspection\n\n```bash\norca orchestration gate-create --task --question [--options ] [--json]\norca orchestration gate-resolve --id --resolution [--json]\norca orchestration gate-list [--task ] [--status ] [--json]\n```\n\nUse `ask` for worker-to-coordinator questions; it creates a `question` message that the coordinator answers with `reply`. Use `gate-create` only for coordinator-managed task DAG decisions, not for answering a worker's `ask`.\n\n`coordinator-start`, `coordinator-stop`, `run`, and `run-stop` are retired scheduler commands. They perform no effects and return the current-skill recovery action. They are not aliases for lightweight Run creation or binding.\n\nRecovery only: `orca orchestration reset --tasks|--messages|--all --json` clears the selected local orchestration database state. Do not run it during active coordination unless explicitly abandoning that state.\n\n## Full Handoffs\n\nFor full ownership transfer, use non-lifecycle terminal/worktree commands and then stop monitoring unless the user asks for supervision.\n\nTreat these as full handoff requests by default: \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"send this to another agent\", \"another agent\", \"another worktree\", or \"launch another agent to own this.\" Custom model or reasoning effort words such as `gpt-5.5`, `high`, or `xhigh` do not make the handoff supervised.\n\nSupervised orchestration remains available only when the user explicitly asks for supervision or coordination: \"supervise\", \"monitor\", \"wait for worker_done\", \"wait for results\", \"track completion\", \"DAG\", \"decision gate\", \"ask/reply\", or \"coordinate workers.\"\n\nDo not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. `task-create` is also forbidden because it records coordinator-owned tracking state; if a task row is needed, the user asked for supervised orchestration. Do not create a `taskId`/`dispatchId`, inject a lifecycle preamble, wait for completion, or read the worker terminal after prompt delivery except to avoid losing the initial prompt.\n\nNew top-level worktree handoff:\n\n```bash\norca worktree create --name --no-parent --agent codex --prompt \"\" --setup run --json\n```\n\nBefore creating a new worktree from an active feature branch, decide and state whether the desired Orca lineage is child or top-level. Use child worktree lineage only when the new work is conceptually stacked under or dependent on the active worktree. For independent repo-wide fixes, standalone feature work, or unrelated follow-up tasks, create a top-level worktree with `--no-parent`.\n\nExisting terminal handoff:\n\n```bash\norca terminal send --terminal --text \"\" --enter --json\n```\n\nCustom Codex model/effort handoff:\n\n`orca worktree create --agent codex --prompt ...` launches the known Codex agent but does not accept Codex-specific `--model` or `-c model_reasoning_effort=...` arguments. When the user asks for a specific Codex model or effort, create the independent worktree first, launch Codex with the requested command in that worktree, wait only for TUI readiness if prompt delivery would otherwise race startup, send the prompt, and stop.\n\nThe two-step custom-argv path cannot enforce a repository's explicit `wait-for-setup` startup policy because the later `terminal create` is not the startup owned by `worktree create`. Use it only when the repository starts agents immediately. If the repository requires `wait-for-setup`, use an agent-first configured launcher that can preserve sequencing, or stop and ask rather than silently bypassing the policy.\n\nNote: when no repo default-terminal configuration supplies a primary terminal, bare create opens a fallback shell before `terminal create` adds the agent. Configured default tabs are materialized instead and may run real commands. Prefer `--agent` whenever custom argv is not required. With the two-step path, target only the agent handle; close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell.\n\nUse the exact full `::` worktree id returned by `orca worktree create --json`; a bare repo id cannot target the new worktree.\n\n```bash\norca worktree create --name --no-parent --setup run --json\norca terminal create --worktree id: --title --command 'codex --model gpt-5.5 -c model_reasoning_effort=\"xhigh\"' --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca terminal send --terminal --text \"\" --enter --json\n```\n\nWait only for `tui-idle` when needed to avoid losing the prompt. Do not monitor task completion.\n\n`--no-parent` only controls Orca lineage; it does not choose the Git base. If the work should start from the repo default base, omit `--base-branch` so Orca uses that default, or explicitly pass the repo default base (`origin/main`, `origin/master`, or the `orca repo show --repo --json` value); never base it on the current feature branch unless the user explicitly asks for stacked work or \"branch from current\". Put current-branch context in the prompt instead.\n\n## Worker Terminals\n\nChoose the worker location before creating a terminal. `Fresh worker` means a fresh agent session, not a new git worktree. For parallel work, create one fresh agent terminal per worker in the same required worktree, falling back to the active worktree when none is named. If the task says current worktree only, depends on uncommitted files/artifacts, or must validate/PR the current branch, keep every worker in the active worktree:\n\n```bash\norca terminal create --worktree active --title --command \"codex\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nReuse an idle agent in the required worktree only if the prompt allows reuse; otherwise create a fresh terminal there. Create a new worktree only when the user explicitly requests one or a concrete checkout or filesystem conflict makes sharing unsafe or impossible; if the user did not request it, state that conflict before running `worktree create`. Independent tasks, parallel execution, convenience, or a preference for separate checkouts are not isolation requirements.\n\nWhen a new worktree is allowed, use child lineage for isolated work that is stacked under or dependent on the active worktree, and use `--no-parent` when it is not stacked. Decide the Git base separately: `--no-parent` makes the worktree top-level in Orca, while omitted `--base-branch` uses the repo default base.\n\nFor every new worktree, pass `--setup run` so any configured repository setup hook runs. This does not mean waiting for setup before agent launch: preserve the repository's startup policy, whose default starts setup and the agent side by side. Use `--setup skip` or `--setup inherit` only when there is a concrete task-specific reason, and state that reason before creating the worktree. This rule does not rerun setup for current or existing worktrees.\n\n```bash\norca worktree create --name --agent codex --setup run --json\n# or: --agent claude | omp | pi | grok | ...\n# Read from agentTerminalHandle, falling back to startupTerminal.handle.\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nFor new-worktree workers, read the id and `agentTerminalHandle` from `worktree create`, falling back to `startupTerminal.handle` for older runtimes. Use that as the sole worker handle when present; otherwise use `terminal list` to resolve the agent handle. Omit `--repo` only inside an Orca-managed worktree; otherwise pass `--repo `.\n\n**For an allowed new worktree, use agent-first:** `--agent` reveals the new worktree and launches the selected agent **in its first terminal**, without adding a separate fallback shell for that worker. Pass `--setup run`; repo setup and default-terminal settings may add intentional tabs or splits. Do **not** run bare `worktree create` and then `terminal create --command ` for the same worker when agent-first create is available: without configured default tabs, that two-step path leaves a fallback shell + agent pair. Only use it when custom agent argv is required (for example Codex model/effort flags) or when an older CLI rejects `--agent`; if you must, message only the agent handle. Configured default tabs are intentional surfaces, so close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell. Do not run `worktree create` when the task must stay in the current worktree.\n\nUse `orca worktree create --prompt ...` or `orca terminal send ...` for full handoffs or untracked/lightweight prompts. Those paths do not attach `taskId`/`dispatchId`; the worker should not send lifecycle messages unless the prompt supplies a live orchestration preamble.\n\nSidebar lineage and orchestration lifecycle are related but not identical. A same-worktree worker may appear as a peer under that worktree in the sidebar while remaining a child dispatch in orchestration state; only an actual child worktree creates visible parent/child worktree lineage.\n\nOther terminal commands coordinators often need:\n\n```bash\norca terminal list [--worktree ] [--include-visual-layouts] [--json]\norca terminal create [--worktree ] [--title ] [--command ] [--json]\norca terminal split --terminal [--direction horizontal|vertical] [--command ] [--json]\norca terminal wait --terminal --for tui-idle --timeout-ms --json\norca terminal read --terminal --json\norca terminal send --terminal --text --enter --json\n```\n\nIf an older CLI rejects `worktree create --agent`, create the worktree normally, then run `orca terminal create --worktree --command \"codex\" --json` or `--command \"claude\"`.\n\nWait for `tui-idle` before dispatching. Always pass `--timeout-ms`; real coding tasks can take 15-60 minutes. During supervision, use rolling `check --wait` windows. If a window returns no matching message, inspect `task-list`, `terminal read`, or `terminal wait --for tui-idle` as a liveness checkpoint; if the terminal is still working or producing activity, keep waiting instead of retrying the task.\n\n## Agent Guidance\n\n- Workers with a valid live preamble must send `worker_done` exactly once from their own terminal with an explicit `--outcome succeeded` or `--outcome failed`:\n `orca orchestration send --type worker_done --subject \"\" --body \"<3-sentence summary: what you did, what you found, what's left>\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a\" --report-path \"\" --json`\n- A failed outcome is still a terminal report, but Orca records both the Dispatch and Task as failed. Never encode failure only in the subject/body.\n- After sending `worker_done`, end that dispatched turn and idle at the agent prompt. Do not autonomously start more work, poll, or attempt to close the terminal yourself. A direct user instruction takes precedence and starts ordinary user-owned work: follow it without coordinator approval or a fresh Dispatch, never refuse it because of worker/coordinator roles, and do not reuse the settled Dispatch's lifecycle IDs. A coordinator-supervised follow-up still arrives with a fresh preamble + TASK block.\n- For long tasks, send heartbeat/status only when the preamble asks for it, including both IDs:\n `orca orchestration send --type heartbeat --subject \"alive\" --payload '{\"taskId\":\"\",\"dispatchId\":\"\",\"phase\":\"implementing\"}' --json`\n- If blocked before completion, use `ask`; use `escalation` only when ownership is valid and the coordinator must intervene.\n- Treat preambles inherited through terminal history or full handoffs as stale unless the current prompt explicitly keeps that coordinator in the loop.\n- Coordinators must account for every settled worker terminal before waiting again or ending the turn: immediately reuse the exact worker for a new Dispatch, explicitly retain it at the user's request with `worker-retain`, or run `worker-release`. Do not leave a completed worker live merely to inspect output; released workers remain readable through `worker-read`.\n- Coordinators should use `task-list --ready` as external memory, dispatch parallel waves, and avoid dependency chains deeper than 3-4 steps.\n\n## Example\n\n```bash\norca terminal create --worktree active --title login-css-worker --command \"claude\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration task-create --spec \"Fix the login button CSS\" --json\norca orchestration dispatch --task --to --inject --json\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\n## Next Action\n\nCoordinator: confirm `orca status --json`, create or bind a Run, inspect `task-list`/`dispatch-show` if inheriting state, then use the explicit supervised loop (`task-create` -> `worker-start` -> `check --wait`). Use low-level terminal creation plus `dispatch --inject` only when the composed start does not express the needed topology. After every accepted `worker_done`, either transfer the exact terminal to an immediate follow-up Dispatch or run `worker-release` before the next wait.\n\nWorker: if the current prompt contains a live dispatch preamble, do the task, use `ask` for blocking questions, and send `worker_done` once with the required payload. If the preamble is stale or absent, do not send lifecycle messages; inspect state or treat the prompt as an ordinary handoff.\n" +const ORCHESTRATION_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Use Orca orchestration for structured multi-agent coordination: threaded\n messages, blocking ask/reply flows, task dispatch, worker_done/escalation\n waits, task DAGs, decision gates, or coordinator loops. Use `orca-cli`\n instead for full ownership handoffs, including requests phrased as \"hand\n off\", \"handoff\", \"handover\", \"give this to another agent\", or \"another\n worktree\" when the user did not explicitly ask to supervise, monitor, wait\n for results, or coordinate a DAG. Use `orca-cli` for terminal control,\n lightweight terminal prompts, shell commands, Orca worktree management,\n reading or waiting on terminals, and the Orca embedded browser. Use Computer\n Use for external browser windows, webviews, Orca app UI, or desktop UI\n outside Orca's embedded browser only when the task requires OS/window-level\n control such as focus, menus, dialogs, coordinates, or screenshots. Use\n `orca-cli` for Orca's embedded pages and a page-automation tool such as\n Playwright or CDP for external pages.\n---\n\n# Orca Inter-Agent Orchestration\n\nOrchestration is Orca's structured coordination layer for agent messages, task ownership, dispatch state, and worker completion tracking.\n\nUse this skill when coordination state matters. For lightweight terminal prompts or basic worktree/terminal/built-in-browser control, use `orca-cli`.\n\n## Tool Boundary\n\nIf a task says to use Orca orchestration, the coordinator must create or bind a Run, create the Task with `orca orchestration task-create`, then attach the worker with either the preferred `orca orchestration worker-start` composition or the low-level `orca orchestration dispatch --inject` path.\n\nDo not substitute non-Orca subagent tools, generic agent-spawn APIs, or chat-only parallel worker features. Those may create useful workers, but they do not create Orca task/dispatch provenance, injected lifecycle preambles, `worker_done` authority, or decision gates.\n\nBefore claiming a worker was orchestrated, verify the task/dispatch exists:\n\n```bash\norca orchestration task-list --json\norca orchestration dispatch-show --task --json\n```\n\nIf the work was accidentally run outside Orca orchestration, say so plainly. To repair provenance, rerun or revalidate the needed work through a fresh Orca terminal plus injected dispatch; do not retroactively describe the external worker as orchestrated.\n\n## When To Use\n\n- Send/reply/ask between agent terminals with persistent messages.\n- Dispatch structured tasks to workers and wait for `worker_done` or `escalation`.\n- Track task DAGs with dependencies.\n- Run coordinator loops or decision gates.\n\nDo not use orchestration merely because the user says \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", or asks for another worktree/agent/model/effort. Those are full ownership transfers unless the user explicitly asks to supervise, monitor, wait for worker completion/results, coordinate a DAG, use decision gates, or keep a blocking ask/reply loop.\n\n## Preconditions\n\n- `orca status --json` should show a running runtime.\n- `orca` must be on PATH (`orca-ide` on Linux).\n- The orchestration experimental feature must be enabled in Settings > Experimental.\n- `orca orchestration` commands are RPC calls to the running Orca runtime.\n\n## Contract Migration\n\nOrca adopts a live pre-update orchestration assignment into an ordinary Run. Adoption preserves the existing agent process, PTY/session, terminal handle, tab/leaf/pane, worktree or folder workspace, Task, and Dispatch; it never restarts or replaces the worker. The retired scheduler is not revived, and a newly created attempt uses the current grammar.\n\nTreat the authority label on injected or formatted messages as definitive:\n\n- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported command printed with the message, using the same CLI executable and arguments that the original prompt supplied.\n- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded, at-least-once cutover replay. Process it idempotently and acknowledge it only through the exact displayed guidance.\n- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or lifecycle action.\n- An unlabeled current message uses the current guide and current grammar.\n\nAn explicitly selected current Run, attested current Run binding, current Dispatch, or federated attachment takes precedence over legacy fallback. A retained adoption record alone never turns a current command into a legacy call.\n\nDatabase provenance, an old-looking terminal, or a legacy Run ID does not prove mutation authority. If the runtime cannot prove liveness, principal ownership, capability, or the exact legacy contract, it degrades to read-only inspection and must not fall back to local execution. Exact recovery may restore the already-live PTY once in its original inactive background tab. It must not spawn, write, signal, stop, switch, focus, split, or inject a terminal. Loss of lifecycle authority does not invalidate the existing assignment, process, or filesystem work.\n\nCompatibility retries have narrow guarantees. A pending ask, a reply, a final Dispatch settlement, and a consuming check have durable recovery identities. A-era heartbeat and escalation calls remain at-least-once across a manual A-to-B retry because identical later signals may be intentional. If an A-era ask may already have been answered, run the exact non-consuming recovery check printed by the runtime first; after its answer is printed and acknowledged, a new invocation with the same question creates a new question. Never guess among multiple identical question threads.\n\nWhen a compatibility or recovery command returns structured next-step arguments, run those exact arguments with the same CLI executable. The arguments intentionally omit the executable name so the guidance works with `orca`, `orca-ide`, `orca-dev`, or another configured Orca CLI command. Do not translate the command from memory, broaden its recipient, or retry it as a current mutation unless the returned guidance explicitly says to.\n\nOn packaged Windows, a legacy ask uses a two-step commit/resume protocol. The initial command durably commits the question, prints its exact `ask --resume ` command, and exits with launcher status `75`; it does not wait for the answer. Run that exact resume command after the launcher or update boundary. Resume is idempotent and read-oriented: it waits for the already-committed question and does not create another one. For a WSL process that received compatibility proof at launch, use the printed executable `orca-ide` WSL resume command so the same distro and packaged launcher authority are preserved; do not substitute a PATH-resolved local CLI. Older WSL processes that never received the hidden launch token remain lifecycle read-only after the update, even while their terminal and filesystem work continue.\n\nLegacy inspection remains available without consuming mail:\n\n```bash\norca orchestration run-list --json\n# run_legacy_local is an empty audit tombstone after adoption.\norca orchestration run-show --id run_legacy_local --json\n# In run-list, find the ordinary Run whose objective is:\n# \"Recovered orchestration work from a contract update\"\norca orchestration run-show --id --json\norca orchestration task-list --run --json\norca orchestration inbox --full --json\norca orchestration check --terminal --peek --format --json\norca terminal read --terminal --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\n```\n\nIf the original coordinator is unavailable or cannot prove its retained authority, a current coordinator may explicitly take over the adopted Run from its own live agent terminal:\n\n```bash\norca orchestration run-use --id --takeover-legacy --json\norca orchestration check --run --json\n```\n\nTakeover fences only the old coordinator, binds the current one, and moves pending worker mail into current Run Delivery. It is bound to the authenticated invoking terminal; `--from` cannot name another coordinator. Live legacy workers keep their original Tasks, Dispatches, processes, filesystems, and old prompt commands; their later questions, escalations, and completion reports route to the current coordinator. Do not use takeover while the original coordinator is still actively coordinating, because its later lifecycle mutations are rejected.\n\nDo not launch a replacement editor merely because the desktop app or runtime was updated. If adoption cannot prove continuing authority, keep the original worker as the only editor until it reaches a stable handoff point, then use a new current Dispatch in a conflict-free placement for any remaining work.\n\n## Ownership\n\nNew orchestration messages and tasks belong to one explicitly bound Run. A Run is only a durable namespace and coordinator inbox; it never schedules or places workers. Lifecycle authority comes from the active Dispatch, and terminal handles remain routing metadata rather than durable identity. Send `worker_done` and `heartbeat` from the worker's own terminal; Orca routes them to that Dispatch's Run.\n\nClassify inherited context before sending lifecycle messages:\n\n- Coordinated subtask: a live coordinator owns the DAG and waits on this dispatch. Follow the preamble exactly, including `worker_done`, heartbeat/status, `ask`, and `escalation`.\n- Full handoff means ownership transfer, not supervised dispatch. The original actor is not monitoring a DAG, so do not create lifecycle obligations unless the user explicitly asks you to supervise.\n- Classify requests containing \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"another agent\", or \"another worktree\" as full handoffs by default, even when the user names a custom model or reasoning effort.\n- Use supervised orchestration only when the user explicitly asks you to \"supervise\", \"monitor\", \"wait\", \"track completion\", \"wait for worker_done\", return results, coordinate a DAG, use a decision gate, or manage ask/reply flow.\n- Do not use `orca orchestration dispatch --inject` for full handoffs. It injects a coordinator preamble that tells the worker to send `worker_done`, heartbeat, and `ask` messages, then end its turn under the original terminal's dispatch lifecycle.\n- Do not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. Do not peek at terminal output after prompt delivery to monitor progress.\n- A review-only `worker_done` reports findings; it does not authorize coordinator file edits. After a review-only completion, synthesize findings, ask a decision gate if ownership is unclear, and dispatch or hand off fixes unless the user explicitly asked the coordinator to own fixes.\n- If the user's plan names a next owner agent (for example, \"then use opencode to create a PR\"), post-review corrections and PR prep belong to that named owner. The coordinator routes, synthesizes, asks decision gates when needed, and supervises; the named owner edits files and creates the PR.\n\nIf unclear, inspect orchestration state before sending lifecycle messages:\n\n```bash\norca orchestration task-list --json\norca terminal list --json\n# If inherited context includes a task id:\norca orchestration dispatch-show --task --json\n```\n\n## Messaging\n\n```bash\norca orchestration send --subject [--to ] [--from ] [--body ] [--type ] [--priority ] [--thread-id ] [--payload ] [--json]\norca orchestration check [--terminal ] [--ack ] [--peek|--all] [--types ] [--format] [--wait] [--timeout-ms ] [--json]\norca orchestration reply --id --body [--from ] [--json]\norca orchestration ask (--question |--resume ) [--options ] [--timeout-ms ] [--from ] [--json]\norca orchestration inbox [--limit ] [--json]\n```\n\nRules:\n\n- Omit `--from` unless impersonating another terminal; Orca auto-resolves it from the current terminal.\n- A coordinator `check` returns the bound Run's oldest FIFO Delivery (up to 50 messages) and replays that exact batch until `--ack `. Process every message before acknowledging; `check --ack --wait` acknowledges, checks, and waits in one operation.\n- Use `--peek` and `--all` only for read-only history/debugging. Type filters decide when a waiter wakes; the returned actionable Delivery is still the oldest full batch.\n- Use `dispatch:` for coordinator guidance to one supervised worker. Orca routes that stable address locally or through the connected-server relay; do not substitute a remote terminal handle.\n- Terminal handles remain appropriate for low-level pre-Dispatch messaging. Prefer `agentTerminalHandle` from the create response, fall back to `startupTerminal.handle` for older runtimes, then re-resolve with `orca terminal list --worktree ... --json` if missing or stale. Continue with the replacement handle only; never dual-send to old and new handles.\n- `terminal list --json` omits `visualLayouts` because handle recovery does not need topology. Add `--include-visual-layouts` only for explicit tab and pane inspection.\n- `orca orchestration check --peek --format --json` returns locally formatted unread mail without consuming it; it never writes to terminal input or remotely wakes another terminal. Use `orchestration dispatch --inject` to deliver a tracked task, or `terminal send` when an existing agent needs a free-form prompt.\n- While supervising workers manually, use `check --wait --types worker_done,escalation,question --timeout-ms ` instead of sleep/poll loops. Process the whole Delivery, reply to `question` messages with `orca orchestration reply --id --body --json`, then acknowledge and keep waiting.\n- `check --json` prints exactly one JSON document on stdout. While `--wait` blocks it also prints keepalive lines (`{\"_keepalive\":true,...}`) to stderr so you can tell the process is alive; those are never on stdout. Do not merge the streams before a parser — `check --wait --json 2>&1 | ` fails with \"Extra data: line 2\". Pipe stdout only.\n- Treat a `check --wait` timeout or `{count:0}` as a checkpoint, not a worker failure. Long coding tasks routinely run 15-60 minutes; keep using rolling waits unless you receive `worker_done`/`escalation`, the terminal exits or disappears, or the user explicitly asks you to stop.\n- Heartbeats and visible terminal activity mean the worker is alive, not done. Do not stop, close, kill, or restart a worker just because it has not produced a completion message yet.\n- Use `ask` when a worker needs a blocking answer from the coordinator; it defaults to the active Dispatch's Run. Timeout or disconnect leaves the question pending, so resume by its original message ID instead of asking again.\n- `check --wait` returns one bounded Delivery, not every future completion. Process every message, acknowledge it, then keep waiting until every expected Dispatch settles.\n- Group addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`, `@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`.\n- Message types include `status`, `dispatch`, `worker_done`, `merge_ready`, `escalation`, `handoff`, `question`, `decision_gate` (legacy/gates), and `heartbeat`.\n- Use group addresses only for messages that are genuinely useful to many terminals, such as `status` broadcasts or intentional fan-out questions. Do not send dispatch lifecycle messages to groups.\n- `worker_done` belongs to the active Dispatch and defaults to its Run mailbox; never target a group.\n- A valid `worker_done` for the active `taskId` + `dispatchId` marks the task and dispatch completed automatically. Do not follow it with `task-update --status completed`; reserve manual updates for explicit recovery or overrides.\n- `heartbeat` is also Dispatch-scoped. Include both IDs and omit `--to` so Orca uses the owning Run; use `status` for broad progress updates.\n\n## Tasks And Dispatch\n\nA Run is the namespace/inbox, a Task is the work item, and a Dispatch assigns one Task attempt to a terminal. Create or bind a Run once before the common loop.\n\n```bash\norca orchestration run-create --objective --json\norca orchestration task-create --spec [--deps ] [--parent ] [--json]\norca orchestration task-list [--status ] [--ready] [--brief] [--json]\norca orchestration task-update --id --status [--result ] [--json]\norca orchestration dispatch --task --to [--from ] [--inject] [--json]\norca orchestration dispatch-show --task [--json]\n```\n\nTask statuses: `pending`, `ready`, `dispatched`, `completed`, `failed`, `blocked`.\n\nDispatch rules:\n\n- `--inject` sends the task spec plus preamble into a recognized agent CLI so it can report `worker_done`.\n- If the target is a bare shell, omit `--inject`, dispatch for tracking if needed, then send the prompt manually with `orca terminal send --terminal --text --enter --json`.\n- After 3 consecutive failures on one task, the dispatch context circuit-breaks and the task is marked failed.\n- Use `task-list --brief --json` for coordinator sweeps; it collapses whitespace and caps each echoed spec at 160 characters (`spec_truncated` marks shortened rows). Omit `--brief` when the full spec is required, or when an older CLI rejects it as an unknown flag.\n\n`dispatch` and `worker-start` refuse the following preflight cases with a stable `error.code`; read it before choosing a recovery, and treat `error.data.nextSteps` as the exact recovery text. Older hosts may omit `data`, so treat every field as optional.\n\n| Code | Meaning | Recovery |\n| -------------------- | --------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------ |\n| `task_not_found` | No Task with that id, or not in the bound Run (`data.taskId`, `data.runId`) | Check `task-list --json`; create the Task with `task-create` if it does not exist |\n| `task_not_startable` | Task cannot start now: not `ready`, or invalid `--retry-of` (`data.status`, `data.unmetDependencies`, `data.retryOf`) | Wait for running dependencies with `check --wait`; retry or unblock failed ones; inspect `dispatch-show` if already dispatched |\n| `inject_rejected` | `--inject` refused because no recognized agent runs in the target (`data.terminal`, `data.reason`) | Start a recognized agent there or pick another terminal; or dispatch without `--inject` and use `terminal send` |\n| `runtime_error` | Any other failure, including a target terminal that already owns an active Dispatch | Read the message, inspect state, and do not retry unchanged |\n\n## How deep workers can nest\n\nA dispatched worker normally cannot dispatch sub-workers. Attempting it fails with\n`nested_worker_depth_exceeded` and a message telling the worker to complete the task\nitself. Do that — do not try to route around it.\n\nThe limit is a number, not an on/off switch. `Settings -> Orchestration -> Nested worker depth`\nsets how many generations are allowed:\n\n- `1` (default): a coordinator dispatches workers; those workers do not dispatch.\n- `2`: workers may dispatch one further generation.\n\nDepth is counted from the terminal that issues the command, not from the Run. Creating a\nnew Run does not reset it — a worker that runs `run-create` then `worker-start` is still a\nworker, and still counted. This is the part that changed: the old behaviour rejected\nsub-dispatch only because a worker's terminal was not bound to a Run, so creating a Run was\nenough to slip past it.\n\nTwo limits worth knowing:\n\n- **It is a guardrail, not a security boundary.** A caller that declares another terminal's\n handle while its own launch evidence is unverifiable (an ordinary restored terminal, for\n example) can be counted as that terminal instead. Orca does not treat workers as hostile.\n- **It applies while a Dispatch is active.** After `worker_done`, or after a coordinator\n settles the task, the terminal is no longer a worker and is counted as a root again. The\n process may still be alive; that is the documented boundary, not an accident.\n\n## Preferred Supervised Worker Loop\n\nUse `worker-start` for the normal supervised path. It composes the existing worktree, terminal, readiness, and dispatch primitives while returning exact created/reused effects. Agents still choose placement and concurrency; Orca does not schedule workers or infer conflicts.\n\nCreate the Run and every independent Task first, then start all independent workers before waiting:\n\n```bash\norca orchestration run-create --objective \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration worker-start --task --worktree current --agent codex --json\norca orchestration worker-start --task --worktree current --agent claude --json\n```\n\n`current` and exact existing worktrees create a fresh agent terminal and do not rerun setup. Reuse an existing agent only with `--terminal `.\n\nFor a per-invocation Claude, Codex, or Cursor launch, pass an opaque provider model id with `--model`; add `--effort` only when that agent/model supports the level. These options apply only to fresh agent terminals, override general agent default arguments, and are reported under `launch.requested` and `launch.effective` in the receipt:\n\n```bash\norca orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`, and neither option can combine with `--terminal`. A connected worker server must advertise launch-preference support before Orca forwards either option.\n\nFor a new worktree, setup runs by default and agent-first creation reuses the returned startup agent terminal:\n\n```bash\norca orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n# Independent/top-level:\norca orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nSetup normally starts alongside the agent. Only a repository explicitly configured with `wait-for-setup` delays agent launch until setup succeeds. Use `--setup skip` or `--setup inherit` only for a concrete reason.\n\nRead the returned receipt before continuing: `ready` plus setup `running` is normal for start-immediately, while wait-for-setup returns setup `succeeded` before accepting task input. A failed or unknown start exits nonzero; inspect its `stage`, `effects`, and `residualResources` instead of guessing or automatically retrying. A wait-for-setup timeout can honestly leave setup `running`, which is not proof of failure.\n\nTo run the worker on another connected Orca server, add `--on `. The Run and Tasks remain authoritative on the current server; later commands route by Dispatch ID, so never repeat `--on`:\n\n```bash\n# Mac Run home -> Windows worker (the reverse is identical from a Windows Run home)\norca orchestration worker-start --task --on windows --worktree new-top-level --repo --name --agent codex --setup run --json\norca orchestration worker-show --dispatch --json\norca orchestration worker-read --dispatch --limit 50 --json\norca orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nRemote `current` and `new-child` are intentionally invalid because those words are ambiguous across servers. Use an exact discovered remote worktree selector or `new-top-level` with an explicit remote repo selector.\n\nThe follow-up is structured inbox mail, not prompt injection. The worker's next\n`orchestration check` receives it even when the Dispatch is on another connected Orca server.\n\n`worker-read` defaults to `--source auto`: Orca returns the exact hook-reported Codex, Claude, OpenClaude, or Grok transcript when it can prove the worker session, otherwise it returns bounded terminal output with `source: \"terminal\"` and a typed `fallbackReason`. Continue with the returned top-level `cursor`; it stays pinned to that exact source. If Orca reports `source_changed`, start a fresh read without the old cursor. Never supply or guess a provider session ID or transcript path.\n\nWait until every expected Dispatch settles, not for a fixed number of batches:\n\n```bash\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n# Process every message. For each accepted worker_done that is not immediately reused:\norca orchestration worker-release --dispatch --json\n# Acknowledge only after every message and required release decision is handled:\norca orchestration check --ack --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\nAfter processing each accepted `worker_done`, choose the terminal's next owner before you acknowledge the Delivery or wait again. If the same exact agent has an immediate follow-up Task, read the `worker.agent_terminal_handle` field of `worker-show --dispatch --json`, then run `orca orchestration worker-start --task --terminal --json` so Orca transfers cleanup ownership to the new Dispatch. Otherwise run `orca orchestration worker-release --dispatch --json`.\n\nRun `worker-release` after both succeeded and failed `worker_done` reports unless the user explicitly asked to keep that worker live. Release is post-completion cleanup, not cancellation: Orca first preserves inspectable output, then closes only the exact agent terminal owned by that settled Dispatch. Reused or pre-existing terminals, setup terminals, coordinators, active workers, user-taken-over terminals, and identities Orca cannot prove are retained. If the user explicitly asks to keep the live terminal for debugging, record that exception with `orca orchestration worker-retain --dispatch --json` instead of silently skipping cleanup. When the user is finished, the same Dispatch can be passed to `worker-release`, which clears the requested retention and releases the terminal.\n\nDo not release a worker because of a timeout, TUI idle state, heartbeat, status, question, escalation, or rejected/stale `worker_done`. If release returns `release_pending` or `release_unknown`, do not substitute `terminal close`; follow the exact recovery action in the receipt. A replayed Delivery may repeat `worker-release` safely.\n\nWorkers report exactly once using the IDs and capability injected by Orca; they do not supply Run/server/terminal identity:\n\n```bash\norca orchestration send --type worker_done --subject \"\" --body \"\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a,path/b\" --json\n# On failure, use --outcome failed; never encode failure only in prose.\n```\n\nA worker question defaults to its owning Run. Timeout leaves it pending:\n\n```bash\norca orchestration ask --question \"\" --options \"yes,no\" --timeout-ms 600000 --json\norca orchestration ask --resume --timeout-ms 600000 --json\n# Coordinator:\norca orchestration reply --id --body \"\" --json\n```\n\nRecovery is conditional, never a fixed destructive sequence:\n\n- The response was lost and named no Dispatch: run `orca orchestration request-show --request --json` first. It is read-only. `completed` means the mutation already took effect. `pending` means the original mutation is still running or Orca restarted before recording its outcome. For either state, replaying the original command with `--retry-request ` reuses the same operation identity so Orca can replay, join, or safely recover it without starting a separate duplicate. `absent` means this runtime holds no receipt under your caller identity and is not proof that nothing happened; inspect the affected state before deciding whether to retry.\n- `worker-show --dispatch ` says `ready`: keep waiting or read bounded output.\n- It proves `failed` or `stopped`: start a replacement with `worker-start --task --retry-of ` plus an explicit `--on`/`--worktree` and `--agent`/`--terminal` choice. Retry does not silently inherit placement.\n- It remains `outcome_unknown`: either `worker-stop --dispatch ` and inspect again, or explicitly `worker-abandon --dispatch ` while accepting that resources may still be live. Abandon performs no remote, process, or filesystem action.\n- `worker-stop` closes only the exact supervised agent terminal. It never deletes the worktree, setup terminal, configured tabs, or unrelated processes.\n\nLow-level `worktree create`, `terminal create`, and `dispatch --inject` remain valid recipes for custom argv or topology that `worker-start` does not express.\n\n`dispatch --inject` deliberately keeps an operator-started terminal unsupervised: it never creates a `worker_dispatches` row and `worker-stop`/`worker-abandon` never close that process. The dispatch context is still authoritative, so `worker-show`, `worker-read`, and `worker-list` report it as `unsupervised`; settled `worker-retain` and `worker-release` report `retained` with `no_owned_resource` and take no process action. Use `worker-start --terminal ` when supervision and worker lifecycle state are required.\n\n## Gates And Legacy Inspection\n\n```bash\norca orchestration gate-create --task --question [--options ] [--json]\norca orchestration gate-resolve --id --resolution [--json]\norca orchestration gate-list [--task ] [--status ] [--json]\n```\n\nUse `ask` for worker-to-coordinator questions; it creates a `question` message that the coordinator answers with `reply`. Use `gate-create` only for coordinator-managed task DAG decisions, not for answering a worker's `ask`.\n\n`coordinator-start`, `coordinator-stop`, `run`, and `run-stop` are retired scheduler commands. They perform no effects and return the current-skill recovery action. They are not aliases for lightweight Run creation or binding.\n\nRecovery only: `orca orchestration reset --tasks|--messages|--all --json` clears the selected local orchestration database state. Do not run it during active coordination unless explicitly abandoning that state.\n\n## Full Handoffs\n\nFor full ownership transfer, use non-lifecycle terminal/worktree commands and then stop monitoring unless the user asks for supervision.\n\nTreat these as full handoff requests by default: \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"send this to another agent\", \"another agent\", \"another worktree\", or \"launch another agent to own this.\" Custom model or reasoning effort words such as `gpt-5.5`, `high`, or `xhigh` do not make the handoff supervised.\n\nSupervised orchestration remains available only when the user explicitly asks for supervision or coordination: \"supervise\", \"monitor\", \"wait for worker_done\", \"wait for results\", \"track completion\", \"DAG\", \"decision gate\", \"ask/reply\", or \"coordinate workers.\"\n\nDo not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. `task-create` is also forbidden because it records coordinator-owned tracking state; if a task row is needed, the user asked for supervised orchestration. Do not create a `taskId`/`dispatchId`, inject a lifecycle preamble, wait for completion, or read the worker terminal after prompt delivery except to avoid losing the initial prompt.\n\nNew top-level worktree handoff:\n\n```bash\norca worktree create --name --no-parent --agent codex --prompt \"\" --setup run --json\n```\n\nBefore creating a new worktree from an active feature branch, decide and state whether the desired Orca lineage is child or top-level. Use child worktree lineage only when the new work is conceptually stacked under or dependent on the active worktree. For independent repo-wide fixes, standalone feature work, or unrelated follow-up tasks, create a top-level worktree with `--no-parent`.\n\nExisting terminal handoff:\n\n```bash\norca terminal send --terminal --text \"\" --enter --json\n```\n\nCustom Codex model/effort handoff:\n\n`orca worktree create --agent codex --prompt ...` launches the known Codex agent but does not accept Codex-specific `--model` or `-c model_reasoning_effort=...` arguments. When the user asks for a specific Codex model or effort, create the independent worktree first, launch Codex with the requested command in that worktree, wait only for TUI readiness if prompt delivery would otherwise race startup, send the prompt, and stop.\n\nThe two-step custom-argv path cannot enforce a repository's explicit `wait-for-setup` startup policy because the later `terminal create` is not the startup owned by `worktree create`. Use it only when the repository starts agents immediately. If the repository requires `wait-for-setup`, use an agent-first configured launcher that can preserve sequencing, or stop and ask rather than silently bypassing the policy.\n\nNote: when no repo default-terminal configuration supplies a primary terminal, bare create opens a fallback shell before `terminal create` adds the agent. Configured default tabs are materialized instead and may run real commands. Prefer `--agent` whenever custom argv is not required. With the two-step path, target only the agent handle; close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell.\n\nUse the exact full `::` worktree id returned by `orca worktree create --json`; a bare repo id cannot target the new worktree.\n\n```bash\norca worktree create --name --no-parent --setup run --json\norca terminal create --worktree id: --title --command 'codex --model gpt-5.5 -c model_reasoning_effort=\"xhigh\"' --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca terminal send --terminal --text \"\" --enter --json\n```\n\nWait only for `tui-idle` when needed to avoid losing the prompt. Do not monitor task completion.\n\n`--no-parent` only controls Orca lineage; it does not choose the Git base. If the work should start from the repo default base, omit `--base-branch` so Orca uses that default, or explicitly pass the repo default base (`origin/main`, `origin/master`, or the `orca repo show --repo --json` value); never base it on the current feature branch unless the user explicitly asks for stacked work or \"branch from current\". Put current-branch context in the prompt instead.\n\n## Worker Terminals\n\nChoose the worker location before creating a terminal. `Fresh worker` means a fresh agent session, not a new git worktree. For parallel work, create one fresh agent terminal per worker in the same required worktree, falling back to the active worktree when none is named. If the task says current worktree only, depends on uncommitted files/artifacts, or must validate/PR the current branch, keep every worker in the active worktree:\n\n```bash\norca terminal create --worktree active --title --command \"codex\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nReuse an idle agent in the required worktree only if the prompt allows reuse; otherwise create a fresh terminal there. Create a new worktree only when the user explicitly requests one or a concrete checkout or filesystem conflict makes sharing unsafe or impossible; if the user did not request it, state that conflict before running `worktree create`. Independent tasks, parallel execution, convenience, or a preference for separate checkouts are not isolation requirements.\n\nWhen a new worktree is allowed, use child lineage for isolated work that is stacked under or dependent on the active worktree, and use `--no-parent` when it is not stacked. Decide the Git base separately: `--no-parent` makes the worktree top-level in Orca, while omitted `--base-branch` uses the repo default base.\n\nFor every new worktree, pass `--setup run` so any configured repository setup hook runs. This does not mean waiting for setup before agent launch: preserve the repository's startup policy, whose default starts setup and the agent side by side. Use `--setup skip` or `--setup inherit` only when there is a concrete task-specific reason, and state that reason before creating the worktree. This rule does not rerun setup for current or existing worktrees.\n\n```bash\norca worktree create --name --agent codex --setup run --json\n# or: --agent claude | omp | pi | grok | ...\n# Read from agentTerminalHandle, falling back to startupTerminal.handle.\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nFor new-worktree workers, read the id and `agentTerminalHandle` from `worktree create`, falling back to `startupTerminal.handle` for older runtimes. Use that as the sole worker handle when present; otherwise use `terminal list` to resolve the agent handle. Omit `--repo` only inside an Orca-managed worktree; otherwise pass `--repo `.\n\n**For an allowed new worktree, use agent-first:** `--agent` reveals the new worktree and launches the selected agent **in its first terminal**, without adding a separate fallback shell for that worker. Pass `--setup run`; repo setup and default-terminal settings may add intentional tabs or splits. Do **not** run bare `worktree create` and then `terminal create --command ` for the same worker when agent-first create is available: without configured default tabs, that two-step path leaves a fallback shell + agent pair. Only use it when custom agent argv is required (for example Codex model/effort flags) or when an older CLI rejects `--agent`; if you must, message only the agent handle. Configured default tabs are intentional surfaces, so close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell. Do not run `worktree create` when the task must stay in the current worktree.\n\nUse `orca worktree create --prompt ...` or `orca terminal send ...` for full handoffs or untracked/lightweight prompts. Those paths do not attach `taskId`/`dispatchId`; the worker should not send lifecycle messages unless the prompt supplies a live orchestration preamble.\n\nSidebar lineage and orchestration lifecycle are related but not identical. A same-worktree worker may appear as a peer under that worktree in the sidebar while remaining a child dispatch in orchestration state; only an actual child worktree creates visible parent/child worktree lineage.\n\nOther terminal commands coordinators often need:\n\n```bash\norca terminal list [--worktree ] [--include-visual-layouts] [--json]\norca terminal create [--worktree ] [--title ] [--command ] [--json]\norca terminal split --terminal [--direction horizontal|vertical] [--command ] [--json]\norca terminal wait --terminal --for tui-idle --timeout-ms --json\norca terminal read --terminal --json\norca terminal send --terminal --text --enter --json\n```\n\nIf an older CLI rejects `worktree create --agent`, create the worktree normally, then run `orca terminal create --worktree --command \"codex\" --json` or `--command \"claude\"`.\n\nWait for `tui-idle` before dispatching. Always pass `--timeout-ms`; real coding tasks can take 15-60 minutes. During supervision, use rolling `check --wait` windows. If a window returns no matching message, inspect `task-list`, `terminal read`, or `terminal wait --for tui-idle` as a liveness checkpoint; if the terminal is still working or producing activity, keep waiting instead of retrying the task.\n\n## Agent Guidance\n\n- Workers with a valid live preamble must send `worker_done` exactly once from their own terminal with an explicit `--outcome succeeded` or `--outcome failed`:\n `orca orchestration send --type worker_done --subject \"\" --body \"<3-sentence summary: what you did, what you found, what's left>\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a\" --report-path \"\" --json`\n- A failed outcome is still a terminal report, but Orca records both the Dispatch and Task as failed. Never encode failure only in the subject/body.\n- After sending `worker_done`, end that dispatched turn and idle at the agent prompt. Do not autonomously start more work, poll, or attempt to close the terminal yourself. A direct user instruction takes precedence and starts ordinary user-owned work: follow it without coordinator approval or a fresh Dispatch, never refuse it because of worker/coordinator roles, and do not reuse the settled Dispatch's lifecycle IDs. A coordinator-supervised follow-up still arrives with a fresh preamble + TASK block.\n- For long tasks, send heartbeat/status only when the preamble asks for it, including both IDs:\n `orca orchestration send --type heartbeat --subject \"alive\" --payload '{\"taskId\":\"\",\"dispatchId\":\"\",\"phase\":\"implementing\"}' --json`\n- If blocked before completion, use `ask`; use `escalation` only when ownership is valid and the coordinator must intervene.\n- Treat preambles inherited through terminal history or full handoffs as stale unless the current prompt explicitly keeps that coordinator in the loop.\n- Coordinators must account for every settled worker terminal before waiting again or ending the turn: immediately reuse the exact worker for a new Dispatch, explicitly retain it at the user's request with `worker-retain`, or run `worker-release`. Do not leave a completed worker live merely to inspect output; released workers remain readable through `worker-read`.\n- Coordinators should use `task-list --ready` as external memory, dispatch parallel waves, and avoid dependency chains deeper than 3-4 steps.\n\n## Example\n\n```bash\norca terminal create --worktree active --title login-css-worker --command \"claude\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration task-create --spec \"Fix the login button CSS\" --json\norca orchestration dispatch --task --to --inject --json\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\n## Next Action\n\nCoordinator: confirm `orca status --json`, create or bind a Run, inspect `task-list`/`dispatch-show` if inheriting state, then use the explicit supervised loop (`task-create` -> `worker-start` -> `check --wait`). Use low-level terminal creation plus `dispatch --inject` only when the composed start does not express the needed topology. After every accepted `worker_done`, either transfer the exact terminal to an immediate follow-up Dispatch or run `worker-release` before the next wait.\n\nWorker: if the current prompt contains a live dispatch preamble, do the task, use `ask` for blocking questions, and send `worker_done` once with the required payload. If the preamble is stale or absent, do not send lifecycle messages; inspect state or treat the prompt as an ordinary handoff.\n" // Why: no current guide has bundled reference documents, so --full is byte-identical for now. // oxfmt-ignore diff --git a/src/cli/orchestration-dispatch-refusal-format.test.ts b/src/cli/orchestration-dispatch-refusal-format.test.ts new file mode 100644 index 00000000000..0a4461f25f1 --- /dev/null +++ b/src/cli/orchestration-dispatch-refusal-format.test.ts @@ -0,0 +1,77 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + injectRejectedRefusal, + taskNotFoundRefusal, + taskNotStartableRefusal, + type DispatchRefusalReceipt +} from '../shared/orchestration-dispatch-refusal-contract' +import { formatCliError, reportCliError } from './format' +import { RuntimeRpcFailureError, type RuntimeRpcFailure } from './runtime/types' + +afterEach(() => { + vi.restoreAllMocks() +}) + +// Why: these are the exact envelopes the RPC dispatcher test proved the runtime emits. This +// checkout's formatter never enumerates codes (verified below with a code no build has defined), +// which is what lets a client that predates a new code still print its message and nextSteps. +describe('orchestration dispatch refusals through the CLI error boundary', () => { + it.each([ + { + receipt: taskNotFoundRefusal('Task not found: task_missing', { taskId: 'task_missing' }), + recovery: /task-create|task-list/ + }, + { + receipt: taskNotStartableRefusal( + 'Task task_child is pending; only ready tasks can be dispatched', + { taskId: 'task_child', status: 'pending', unmetDependencies: ['task_parent'] } + ), + recovery: /task_parent/ + }, + { + receipt: injectRejectedRefusal('term_worker', 'no_agent_detected'), + recovery: /without --inject/ + } + ])('prints $receipt.code with its recovery in human and JSON output', ({ receipt, recovery }) => { + const failure = envelope(receipt) + const error = new RuntimeRpcFailureError(failure) + expect(error.code).toBe(receipt.code) + + const human = formatCliError(error, { commandPath: ['orchestration', 'dispatch'] }) + expect(human).toContain(receipt.message) + expect(human).toMatch(recovery) + + const log = vi.spyOn(console, 'log').mockImplementation(() => {}) + reportCliError(error, true, { commandPath: ['orchestration', 'dispatch'] }) + const printed = JSON.parse(log.mock.calls[0]?.[0] as string) as RuntimeRpcFailure + expect(printed.ok).toBe(false) + expect(printed.error).toEqual(receipt) + }) +}) + +// Why: a code this build has never defined stands in for a future host's new code; if the +// formatter ever starts gating on known codes, this is the assertion that catches it. +it('prints an unknown code with its message and nextSteps unchanged', () => { + const failure: RuntimeRpcFailure = { + id: 'rpc_1', + ok: false, + error: { + code: 'code_from_a_newer_host', + message: 'Refused for a reason this CLI has never heard of.', + data: { nextSteps: ['Do the thing the newer host suggested.'] } + }, + _meta: { runtimeId: 'runtime_1' } + } + const error = new RuntimeRpcFailureError(failure) + + expect(formatCliError(error, { commandPath: ['orchestration', 'dispatch'] })).toBe( + 'Refused for a reason this CLI has never heard of.\nNext step: Do the thing the newer host suggested.' + ) + const log = vi.spyOn(console, 'log').mockImplementation(() => {}) + reportCliError(error, true, { commandPath: ['orchestration', 'dispatch'] }) + expect(JSON.parse(log.mock.calls[0]?.[0] as string)).toEqual(failure) +}) + +function envelope(receipt: DispatchRefusalReceipt): RuntimeRpcFailure { + return { id: 'rpc_1', ok: false, error: receipt, _meta: { runtimeId: 'runtime_1' } } +} diff --git a/src/main/runtime/orchestration/db/dispatch-context/dispatch-context-store.ts b/src/main/runtime/orchestration/db/dispatch-context/dispatch-context-store.ts index cd4083acc0a..ee78396292d 100644 --- a/src/main/runtime/orchestration/db/dispatch-context/dispatch-context-store.ts +++ b/src/main/runtime/orchestration/db/dispatch-context/dispatch-context-store.ts @@ -7,6 +7,7 @@ import { paneKeyMatchSuffix } from '../pane-key-match' import { claimDispatchContextRow } from '../dispatch-row-writer' import type { DispatchCreator } from '../dispatch-depth' import type { OrchestrationDb } from '../orchestration-db' +import { taskNotFoundError, taskNotStartableError } from '../../task-dispatch-refusal' export function createDispatchContext( this: OrchestrationDb, @@ -26,10 +27,14 @@ export function createDispatchContext( const depth = this.resolveChildDispatchDepth(params.creator, params.maxDepth) const task = this.getTask(taskId) if (!task) { - throw new Error(`Task not found: ${taskId}`) + throw taskNotFoundError(`Task not found: ${taskId}`, { taskId }) } if (task.status !== 'ready') { - throw new Error(`Task ${taskId} is ${task.status}; only ready tasks can be dispatched`) + throw taskNotStartableError( + this, + `Task ${taskId} is ${task.status}; only ready tasks can be dispatched`, + task + ) } // Why: lock on pane identity too, so a reminted handle can't open a second concurrent dispatch on the same pane. @@ -72,9 +77,12 @@ export function createDispatchContext( `Terminal ${assigneeHandle} already has an active dispatch (${occupied.id} for task ${occupied.task_id})` ) } - throw new Error( - `Task ${taskId} is ${current?.status ?? 'missing'}; only ready tasks can be dispatched` - ) + // Why: the atomic claim lost to a concurrent status change; report it with the same + // typed receipt as the precheck so the loser can recover instead of reading runtime_error. + const message = `Task ${taskId} is ${current?.status ?? 'missing'}; only ready tasks can be dispatched` + throw current + ? taskNotStartableError(this, message, current) + : taskNotFoundError(message, { taskId }) } this.db.prepare("UPDATE tasks SET status = 'dispatched' WHERE id = ?").run(taskId) const dispatch = this.db diff --git a/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-start.ts b/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-start.ts index e26369e7fc1..e472ea1c7e8 100644 --- a/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-start.ts +++ b/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-start.ts @@ -6,6 +6,7 @@ import { generateId } from '../generated-id' import type { OrchestrationDb } from '../orchestration-db' import { insertStartingDispatchContextRow } from '../dispatch-row-writer' import type { DispatchCreator } from '../dispatch-depth' +import { taskNotFoundError, taskNotStartableError } from '../../task-dispatch-refusal' export function createStartingWorkerDispatch( this: OrchestrationDb, @@ -60,7 +61,7 @@ export function createStartingWorkerDispatch( } const task = this.getTask(params.taskId) if (!task) { - throw new OrchestrationError('task_not_found', `Task ${params.taskId} was not found.`) + throw taskNotFoundError(`Task ${params.taskId} was not found.`, { taskId: params.taskId }) } if (params.retryOf) { const prior = this.getDispatchContextById(params.retryOf) @@ -74,15 +75,18 @@ export function createStartingWorkerDispatch( !['failed', 'stopped', 'abandoned'].includes(priorWorker.state) || !['failed', 'blocked'].includes(task.status) ) { - throw new OrchestrationError( - 'task_not_startable', - `Task ${task.id} cannot retry from Dispatch ${params.retryOf}.` + throw taskNotStartableError( + this, + `Task ${task.id} cannot retry from Dispatch ${params.retryOf}.`, + task, + params.retryOf ) } } else if (task.status !== 'ready') { - throw new OrchestrationError( - 'task_not_startable', - `Task ${task.id} is ${task.status}; only a ready Task can start.` + throw taskNotStartableError( + this, + `Task ${task.id} is ${task.status}; only a ready Task can start.`, + task ) } diff --git a/src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts b/src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts index 9e4a2768a1d..156d1427f01 100644 --- a/src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts +++ b/src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts @@ -168,7 +168,12 @@ describe('OrchestrationDb worker Dispatch state', () => { payloadHash: 'payload_hash' } }) - ).toThrow('was not found') + ).toThrowError( + expect.objectContaining({ + code: 'task_not_found', + message: 'Task task_missing was not found.' + }) + ) expect(d.getMutationReceipt('caller_fingerprint', 'invalid_worker_start')).toBeUndefined() }) diff --git a/src/main/runtime/orchestration/task-dispatch-refusal.ts b/src/main/runtime/orchestration/task-dispatch-refusal.ts new file mode 100644 index 00000000000..f3e026d0f73 --- /dev/null +++ b/src/main/runtime/orchestration/task-dispatch-refusal.ts @@ -0,0 +1,61 @@ +import type { OrchestrationDb } from './db' +import { OrchestrationError } from './orchestration-error' +import type { TaskRow } from './types' +import { + injectRejectedRefusal, + taskNotFoundRefusal, + taskNotStartableRefusal, + type DispatchRefusalReceipt, + type InjectRejectionReason +} from '../../../shared/orchestration-dispatch-refusal-contract' + +// Why: each site keeps the exact message it published before; only the code and data are shared. + +export function taskNotFoundError( + message: string, + detail: { taskId: string; runId?: string } +): OrchestrationError { + return toError(taskNotFoundRefusal(message, detail)) +} + +export function taskNotStartableError( + db: OrchestrationDb, + message: string, + task: TaskRow, + retryOf?: string +): OrchestrationError { + return toError( + taskNotStartableRefusal(message, { + taskId: task.id, + status: task.status, + unmetDependencies: unmetTaskDependencies(db, task), + ...(retryOf ? { retryOf } : {}) + }) + ) +} + +export function injectRejectedError( + terminal: string, + reason: InjectRejectionReason +): OrchestrationError { + return toError(injectRejectedRefusal(terminal, reason)) +} + +function toError(receipt: DispatchRefusalReceipt): OrchestrationError { + return new OrchestrationError(receipt.code, receipt.message, receipt.data) +} + +function unmetTaskDependencies(db: OrchestrationDb, task: TaskRow): string[] { + let deps: unknown + try { + deps = JSON.parse(task.deps) + } catch { + return [] + } + if (!Array.isArray(deps)) { + return [] + } + return deps.filter( + (dep): dep is string => typeof dep === 'string' && db.getTask(dep)?.status !== 'completed' + ) +} diff --git a/src/main/runtime/rpc/errors.ts b/src/main/runtime/rpc/errors.ts index f4b4768865b..1e4567f7f6f 100644 --- a/src/main/runtime/rpc/errors.ts +++ b/src/main/runtime/rpc/errors.ts @@ -85,6 +85,7 @@ const STRUCTURED_RUNTIME_PASSTHROUGH_CODES: ReadonlySet = new Set([ 'consumer_fenced', 'task_not_found', 'task_not_startable', + 'inject_rejected', 'dispatch_not_found', 'dispatch_run_mismatch', 'terminal_not_found', diff --git a/src/main/runtime/rpc/methods/orchestration-dispatch-error-codes.test.ts b/src/main/runtime/rpc/methods/orchestration-dispatch-error-codes.test.ts new file mode 100644 index 00000000000..918724e6aaf --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration-dispatch-error-codes.test.ts @@ -0,0 +1,242 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { ORCHESTRATION_CONTRACT_VERSION } from '../../../../shared/protocol-version' +import { + buildInjectRejectionMessage, + injectRejectedRefusal, + taskNotFoundRefusal, + taskNotStartableRefusal +} from '../../../../shared/orchestration-dispatch-refusal-contract' +import { OrcaRuntimeService } from '../../orca-runtime' +import { OrchestrationDb } from '../../orchestration/db' +import type { RpcFailure, RpcRequest, RpcResponse } from '../core' +import { RpcDispatcher } from '../dispatcher' +import { ORCHESTRATION_METHODS } from './orchestration' + +const COORDINATOR_HANDLE = 'term_codes_coordinator' +const COORDINATOR_PANE = 'tab_coord:cccccccc-cccc-4ccc-8ccc-cccccccccccc' +const WORKER_HANDLE = 'term_codes_worker' +const WORKER_PANE = 'tab_worker:dddddddd-dddd-4ddd-8ddd-dddddddddddd' + +type Harness = { db: OrchestrationDb; runtime: OrcaRuntimeService; dispatcher: RpcDispatcher } + +const harnesses: Harness[] = [] +let requestSequence = 0 + +afterEach(() => { + for (const harness of harnesses.splice(0)) { + harness.db.close() + } + vi.restoreAllMocks() +}) + +// Why: an agent reads the receipt code to pick a recovery; every case is driven from the real +// RPC dispatcher and checked against the shared contract the CLI-side test formats. +describe('orchestration dispatch failure codes through RpcDispatcher', () => { + it('reports task_not_found for a task id that does not exist', async () => { + const harness = createHarness() + + const response = await dispatch(harness, { task: 'task_missing', to: WORKER_HANDLE }) + + expect(expectFailure(response).error).toEqual( + taskNotFoundRefusal('Task not found: task_missing', { taskId: 'task_missing' }) + ) + }) + + it('reports task_not_startable with the unmet dependencies for a pending task', async () => { + const harness = createHarness() + const parent = harness.db.createTask({ spec: 'parent' }) + const child = harness.db.createTask({ spec: 'child', deps: [parent.id] }) + + const response = await dispatch(harness, { task: child.id, to: WORKER_HANDLE }) + + expect(expectFailure(response).error).toEqual( + taskNotStartableRefusal(`Task ${child.id} is pending; only ready tasks can be dispatched`, { + taskId: child.id, + status: 'pending', + unmetDependencies: [parent.id] + }) + ) + expect(harness.db.getTask(child.id)?.status).toBe('pending') + }) + + it('reports task_not_startable with the status for a completed task', async () => { + const harness = createHarness() + const task = harness.db.createTask({ spec: 'done' }) + harness.db.updateTaskStatus(task.id, 'completed') + + const response = await dispatch(harness, { task: task.id, to: WORKER_HANDLE }) + + expect(expectFailure(response).error).toEqual( + taskNotStartableRefusal(`Task ${task.id} is completed; only ready tasks can be dispatched`, { + taskId: task.id, + status: 'completed', + unmetDependencies: [] + }) + ) + }) + + it('reports inject_rejected when the target terminal runs no recognized agent', async () => { + const harness = createHarness() + const task = harness.db.createTask({ spec: 'work' }) + vi.spyOn(harness.runtime, 'isTerminalRunningAgent').mockResolvedValue(false) + + const response = await dispatch(harness, { task: task.id, to: WORKER_HANDLE, inject: true }) + + expect(expectFailure(response).error).toEqual( + injectRejectedRefusal(WORKER_HANDLE, 'no_agent_detected') + ) + expect(expectFailure(response).error.message).toBe(buildInjectRejectionMessage(WORKER_HANDLE)) + expect(harness.db.getTask(task.id)?.status).toBe('ready') + expect(harness.db.getDispatchContext(task.id)).toBeUndefined() + }) + + it('reports task_not_startable with dependency detail from worker-start', async () => { + const harness = createHarness() + const parent = harness.db.createTask({ spec: 'parent' }) + const child = harness.db.createTask({ spec: 'child', deps: [parent.id] }) + mockWorkerStartTopology(harness.runtime) + + const response = await harness.dispatcher.dispatch( + request('orchestration.workerStart', { + task: child.id, + from: COORDINATOR_HANDLE, + agent: 'claude' + }) + ) + + expect(expectFailure(response).error).toEqual( + taskNotStartableRefusal(`Task ${child.id} is pending; only a ready Task can start.`, { + taskId: child.id, + status: 'pending', + unmetDependencies: [parent.id] + }) + ) + expect(harness.db.getTask(child.id)?.status).toBe('pending') + }) + + it('reports task_not_startable with retry detail for an invalid --retry-of', async () => { + const harness = createHarness() + const task = harness.db.createTask({ spec: 'work' }) + mockWorkerStartTopology(harness.runtime) + + const response = await harness.dispatcher.dispatch( + request('orchestration.workerStart', { + task: task.id, + from: COORDINATOR_HANDLE, + agent: 'claude', + retryOf: 'ctx_missing' + }) + ) + + expect(expectFailure(response).error).toEqual( + taskNotStartableRefusal(`Task ${task.id} cannot retry from Dispatch ctx_missing.`, { + taskId: task.id, + status: 'ready', + unmetDependencies: [], + retryOf: 'ctx_missing' + }) + ) + }) + + it('types the atomic claim loser when the task changes after the ready precheck', async () => { + const harness = createHarness() + const task = harness.db.createTask({ spec: 'raced' }) + // Why: the pane lookup runs after the ready precheck and before the DB claim, so failing the + // task there is the interleaving a concurrent status change produces. The loser's DB refusal + // must carry the same typed receipt instead of the bare Error it used to throw. + vi.mocked(harness.runtime.getTerminalPaneKey).mockImplementation((handle) => { + if (handle === WORKER_HANDLE) { + harness.db.updateTaskStatus(task.id, 'failed', 'raced out') + return WORKER_PANE + } + return handle === COORDINATOR_HANDLE ? COORDINATOR_PANE : null + }) + + const response = await dispatch(harness, { task: task.id, to: WORKER_HANDLE }) + + expect(expectFailure(response).error).toEqual( + taskNotStartableRefusal(`Task ${task.id} is failed; only ready tasks can be dispatched`, { + taskId: task.id, + status: 'failed', + unmetDependencies: [] + }) + ) + expect(harness.db.getDispatchContext(task.id)).toBeUndefined() + }) + + it('keeps runtime_error for a genuinely unexpected dispatch failure', async () => { + const harness = createHarness() + const task = harness.db.createTask({ spec: 'work' }) + vi.spyOn(harness.runtime, 'isTerminalRunningAgent').mockRejectedValue( + new Error('probe exploded') + ) + + const response = await dispatch(harness, { task: task.id, to: WORKER_HANDLE, inject: true }) + + expect(expectFailure(response).error).toMatchObject({ + code: 'runtime_error', + message: 'probe exploded' + }) + }) +}) + +function expectFailure(response: RpcResponse): RpcFailure { + if (response.ok) { + throw new Error(`Expected a failure, got ${JSON.stringify(response.result)}`) + } + return response +} + +function createHarness(): Harness { + const db = new OrchestrationDb(':memory:') + const runtime = new OrcaRuntimeService() + runtime.setOrchestrationDb(db) + vi.spyOn(runtime, 'getTerminalPaneKey').mockImplementation((handle) => + handle === COORDINATOR_HANDLE ? COORDINATOR_PANE : handle === WORKER_HANDLE ? WORKER_PANE : null + ) + vi.spyOn(runtime, 'getTerminalProcessIncarnation').mockImplementation((handle) => + handle === WORKER_HANDLE ? 'pty-worker:incarnation-1' : null + ) + const runId = db.createRun({ + objective: 'Typed dispatch failures', + coordinatorHandle: COORDINATOR_HANDLE, + coordinatorPaneKey: COORDINATOR_PANE + }).id + const createTask = db.createTask.bind(db) + db.createTask = (task) => createTask({ ...task, runId: task.runId ?? runId }) + const harness = { + db, + runtime, + dispatcher: new RpcDispatcher({ runtime, methods: ORCHESTRATION_METHODS }) + } + harnesses.push(harness) + return harness +} + +function mockWorkerStartTopology(runtime: OrcaRuntimeService): void { + vi.spyOn(runtime, 'validateOrchestrationAgentLauncher').mockImplementation(() => {}) + vi.spyOn(runtime, 'showTerminal').mockImplementation( + async (handle) => ({ handle, worktreeId: 'repo::worktree', status: 'running' }) as never + ) + vi.spyOn(runtime, 'showManagedTerminalWorkspace').mockResolvedValue({ + id: 'repo::worktree' + } as never) +} + +function dispatch(harness: Harness, params: Record): Promise { + return harness.dispatcher.dispatch( + request('orchestration.dispatch', { from: COORDINATOR_HANDLE, ...params }) + ) +} + +function request(method: string, params: Record): RpcRequest { + requestSequence += 1 + return { + id: `rpc_dispatch_error_codes_${requestSequence}`, + authToken: 'test-token', + method, + params, + orchestrationContractVersion: ORCHESTRATION_CONTRACT_VERSION, + orchestrationRequestId: `dispatch_error_codes_${requestSequence}` + } +} diff --git a/src/main/runtime/rpc/methods/orchestration-dispatch-methods.ts b/src/main/runtime/rpc/methods/orchestration-dispatch-methods.ts index ae21a4e5a46..d573d944b2d 100644 --- a/src/main/runtime/rpc/methods/orchestration-dispatch-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration-dispatch-methods.ts @@ -2,7 +2,11 @@ import { defineMethod, type RpcMethod } from '../core' import { OrchestrationError } from '../../orchestration/orchestration-error' import { buildDispatchPreamble } from '../../orchestration/preamble' import { resolveDispatchCreator } from './orchestration-dispatch-creator' -import { buildInjectRejectionMessage } from './orchestration-inject-rejection-message' +import { + injectRejectedError, + taskNotFoundError, + taskNotStartableError +} from '../../orchestration/task-dispatch-refusal' import { resolveRunScope } from './orchestration-run-scope' import { DispatchParams, DispatchShowParams } from './orchestration-schemas' @@ -22,7 +26,7 @@ export const ORCHESTRATION_DISPATCH_METHODS: RpcMethod[] = [ const db = runtime.getOrchestrationDb() const task = db.getTask(params.task) if (!task) { - throw new Error(`Task not found: ${params.task}`) + throw taskNotFoundError(`Task not found: ${params.task}`, { taskId: params.task }) } const run = resolveRunScope(runtime, { runId: params.run, @@ -32,10 +36,10 @@ export const ORCHESTRATION_DISPATCH_METHODS: RpcMethod[] = [ callerEvidence: orchestrationCompatibilityEvidence }) if (task.run_id !== run.id) { - throw new OrchestrationError( - 'task_not_found', - `Task ${task.id} was not found in Run ${run.id}.` - ) + throw taskNotFoundError(`Task ${task.id} was not found in Run ${run.id}.`, { + taskId: task.id, + runId: run.id + }) } // Why: dry-run previews the preamble without mutating state, so it skips the ready-status check and uses a placeholder dispatchId. @@ -66,14 +70,18 @@ export const ORCHESTRATION_DISPATCH_METHODS: RpcMethod[] = [ const to = params.to if (task.status !== 'ready') { - throw new Error(`Task ${params.task} is ${task.status}; only ready tasks can be dispatched`) + throw taskNotStartableError( + db, + `Task ${params.task} is ${task.status}; only ready tasks can be dispatched`, + task + ) } // Why: injecting the preamble into a bare shell dumps it as shell commands (gibberish), so require a detected agent first. if (params.inject) { const hasAgent = await runtime.isTerminalRunningAgent(to) if (!hasAgent) { - throw new Error(buildInjectRejectionMessage(to)) + throw injectRejectedError(to, 'no_agent_detected') } } diff --git a/src/main/runtime/rpc/methods/orchestration-inject-rejection-message.test.ts b/src/main/runtime/rpc/methods/orchestration-inject-rejection-message.test.ts deleted file mode 100644 index a7ae03b00a8..00000000000 --- a/src/main/runtime/rpc/methods/orchestration-inject-rejection-message.test.ts +++ /dev/null @@ -1,31 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { buildInjectRejectionMessage } from './orchestration-inject-rejection-message' -import { TUI_AGENT_CONFIG } from '../../../../shared/tui-agent-config' -import { recognizeAgentProcess } from '../../../../shared/agent-process-recognition' - -describe('buildInjectRejectionMessage', () => { - const message = buildInjectRejectionMessage('term_a') - - it('keeps the substring callers and scripts match on', () => { - expect(message).toContain('Cannot dispatch --inject to terminal term_a') - expect(message).toContain('no recognized agent detected') - }) - - it('names every agent Orca recognizes, including agy', () => { - expect(message).toMatch(/\bagy\b/) - for (const config of Object.values(TUI_AGENT_CONFIG)) { - expect(message).toContain(config.expectedProcess) - } - }) - - it('lists only names detection actually resolves, deduped and sorted', () => { - const listed = (/\(([^)]+)\)/.exec(message)?.[1] ?? '').split(', ') - - expect(listed.length).toBeGreaterThan(0) - expect(new Set(listed).size).toBe(listed.length) - expect([...listed].sort()).toEqual(listed) - for (const name of listed) { - expect(recognizeAgentProcess(name)).not.toBeNull() - } - }) -}) diff --git a/src/main/runtime/rpc/methods/orchestration-inject-rejection-message.ts b/src/main/runtime/rpc/methods/orchestration-inject-rejection-message.ts deleted file mode 100644 index 33d622ca80e..00000000000 --- a/src/main/runtime/rpc/methods/orchestration-inject-rejection-message.ts +++ /dev/null @@ -1,16 +0,0 @@ -import { TUI_AGENT_CONFIG } from '../../../../shared/tui-agent-config' - -// Why: the old five-name example read as an allowlist (#15125); derive from the field detection keys on so it cannot drift. -// Not filtered by `disabledTuiAgents` — that gates Orca's launchers, not detection, so a hand-started disabled agent still injects. -const RECOGNIZED_AGENT_PROCESS_NAMES = [ - ...new Set(Object.values(TUI_AGENT_CONFIG).map((config) => config.expectedProcess)) -].sort() - -export function buildInjectRejectionMessage(terminal: string): string { - return ( - `Cannot dispatch --inject to terminal ${terminal}: no recognized agent detected. ` + - `Orca detects these agent CLIs (${RECOGNIZED_AGENT_PROCESS_NAMES.join(', ')}). ` + - 'Start one in the terminal and let it finish launching, ' + - 'or dispatch without --inject and send the prompt manually.' - ) -} diff --git a/src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts b/src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts index a216b4c7a4d..cd6d79c9fd5 100644 --- a/src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts +++ b/src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts @@ -3,7 +3,7 @@ import type { RpcContext } from '../core' import { createOrchestrationRpcHarness } from './orchestration-rpc-test-harness' import type { OrchestrationDb } from '../../orchestration/db' import type { OrcaRuntimeService } from '../../orca-runtime' -import { buildInjectRejectionMessage } from './orchestration-inject-rejection-message' +import { buildInjectRejectionMessage } from '../../../../shared/orchestration-dispatch-refusal-contract' import { createRootDispatch } from '../../orchestration/db/root-dispatch-test-fixture' describe('orchestration RPC methods', () => { diff --git a/src/main/runtime/rpc/methods/orchestration-workers.ts b/src/main/runtime/rpc/methods/orchestration-workers.ts index 61271525939..632b34cc1b7 100644 --- a/src/main/runtime/rpc/methods/orchestration-workers.ts +++ b/src/main/runtime/rpc/methods/orchestration-workers.ts @@ -21,6 +21,7 @@ import { import { failWorkerStartWithReceipt } from './orchestration-worker-start-receipt' import { prepareLocalWorkerStart } from './orchestration-worker-start-validation' import { resolveDispatchCreator } from './orchestration-dispatch-creator' +import { taskNotFoundError } from '../../orchestration/task-dispatch-refusal' import { resolveOrchestrationCaller } from './orchestration-run-scope' import { isWorkerStartTimeoutWithinTimerLimit, @@ -58,10 +59,10 @@ export const ORCHESTRATION_WORKER_START_METHODS: RpcMethod[] = [ } const task = db.getTask(params.task) if (!task || task.run_id !== run.id) { - throw new OrchestrationError( - 'task_not_found', - `Task ${params.task} was not found in Run ${run.id}.` - ) + throw taskNotFoundError(`Task ${params.task} was not found in Run ${run.id}.`, { + taskId: params.task, + runId: run.id + }) } if (params.on) { diff --git a/src/main/ssh/ssh-remote-cli-dispatch-refusal-passthrough.test.ts b/src/main/ssh/ssh-remote-cli-dispatch-refusal-passthrough.test.ts new file mode 100644 index 00000000000..cd0503348d0 --- /dev/null +++ b/src/main/ssh/ssh-remote-cli-dispatch-refusal-passthrough.test.ts @@ -0,0 +1,64 @@ +import { EventEmitter } from 'node:events' +import { expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ + app: { + isPackaged: false, + getAppPath: () => '/host/app' + } +})) +vi.mock('../persistence', () => ({ + getCanonicalUserDataPath: () => '/host/user-data' +})) + +import { OrcaRuntimeService } from '../runtime/orca-runtime' +import { runRemoteOrcaCli } from './ssh-remote-orca-cli' + +// Why: the SSH bridge captures the host CLI child's stdout and exit code without reparsing; this +// pins that a typed refusal envelope and its nonzero exit reach the remote agent unchanged. +it('relays typed dispatch refusal codes from the host CLI unchanged', async () => { + const child = new EventEmitter() as EventEmitter & { + stdout: EventEmitter + stderr: EventEmitter + stdin: { end: ReturnType; on: ReturnType } + kill: ReturnType + } + child.stdout = new EventEmitter() + child.stderr = new EventEmitter() + child.stdin = { end: vi.fn(), on: vi.fn() } + child.kill = vi.fn() + const spawn = vi.fn(() => child) + const refusal = { + id: 'rpc_1', + ok: false, + error: { + code: 'task_not_startable', + message: 'Task task_1 is pending; only ready tasks can be dispatched', + data: { taskId: 'task_1', status: 'pending', unmetDependencies: ['task_0'] } + }, + _meta: { runtimeId: 'runtime_1' } + } + + const resultPromise = runRemoteOrcaCli( + new OrcaRuntimeService(), + { + argv: ['orchestration', 'dispatch', '--task', 'task_1', '--to', 'term_w', '--json'], + cwd: '/home/alice/repo', + env: { ORCA_TERMINAL_HANDLE: 'term_ssh' } + }, + { + execPath: '/host/electron', + cliEntryPath: '/host/app/out/cli/index.js', + userDataPath: '/host/user-data', + entryExists: () => true, + spawn: spawn as never + } + ) + + const stdout = `${JSON.stringify(refusal, null, 2)}\n` + await Promise.resolve() + child.stdout.emit('data', Buffer.from(stdout)) + child.emit('close', 1) + + expect(await resultPromise).toEqual({ stdout, stderr: '', exitCode: 1 }) +}) diff --git a/src/shared/orchestration-dispatch-refusal-contract.test.ts b/src/shared/orchestration-dispatch-refusal-contract.test.ts new file mode 100644 index 00000000000..4f9fa4bf617 --- /dev/null +++ b/src/shared/orchestration-dispatch-refusal-contract.test.ts @@ -0,0 +1,67 @@ +import { describe, expect, it } from 'vitest' +import { + buildInjectRejectionMessage, + taskNotFoundRefusal, + taskNotStartableRefusal +} from './orchestration-dispatch-refusal-contract' +import { TUI_AGENT_CONFIG } from './tui-agent-config' +import { recognizeAgentProcess } from './agent-process-recognition' + +describe('buildInjectRejectionMessage', () => { + const message = buildInjectRejectionMessage('term_a') + + it('keeps the substring callers and scripts match on', () => { + expect(message).toContain('Cannot dispatch --inject to terminal term_a') + expect(message).toContain('no recognized agent detected') + }) + + it('names every agent Orca recognizes, including agy', () => { + expect(message).toMatch(/\bagy\b/) + for (const config of Object.values(TUI_AGENT_CONFIG)) { + expect(message).toContain(config.expectedProcess) + } + }) + + it('lists only names detection actually resolves, deduped and sorted', () => { + const listed = (/\(([^)]+)\)/.exec(message)?.[1] ?? '').split(', ') + + expect(listed.length).toBeGreaterThan(0) + expect(new Set(listed).size).toBe(listed.length) + expect([...listed].sort()).toEqual(listed) + for (const name of listed) { + expect(recognizeAgentProcess(name)).not.toBeNull() + } + }) +}) + +// Why: these strings are published receipts; they are pinned as literals, independent of the +// builders, so a refactor cannot silently rewrite them together with the expectation. +describe('dispatch refusal receipts keep their published messages', () => { + it('leaves the message exactly as each call site supplies it', () => { + expect(taskNotFoundRefusal('Task not found: task_1', { taskId: 'task_1' }).message).toBe( + 'Task not found: task_1' + ) + expect( + taskNotStartableRefusal('Task task_1 is pending; only a ready Task can start.', { + taskId: 'task_1', + status: 'pending', + unmetDependencies: [] + }).message + ).toBe('Task task_1 is pending; only a ready Task can start.') + }) + + it('tailors nextSteps to retry, dependency, occupancy, and terminal-status refusals', () => { + const base = { taskId: 'task_1', status: 'failed', unmetDependencies: [] } + expect(taskNotStartableRefusal('m', { ...base, retryOf: 'ctx_1' }).data.nextSteps[0]).toMatch( + /--retry-of.*ctx_1/ + ) + expect( + taskNotStartableRefusal('m', { ...base, status: 'pending', unmetDependencies: ['task_0'] }) + .data.nextSteps[0] + ).toMatch(/task_0.*unblock failed/) + expect( + taskNotStartableRefusal('m', { ...base, status: 'dispatched' }).data.nextSteps[0] + ).toMatch(/dispatch-show --task task_1/) + expect(taskNotStartableRefusal('m', base).data.nextSteps[0]).toMatch(/failed Task cannot/) + }) +}) diff --git a/src/shared/orchestration-dispatch-refusal-contract.ts b/src/shared/orchestration-dispatch-refusal-contract.ts new file mode 100644 index 00000000000..b764067402b --- /dev/null +++ b/src/shared/orchestration-dispatch-refusal-contract.ts @@ -0,0 +1,102 @@ +import { TUI_AGENT_CONFIG } from './tui-agent-config' + +// Why: one source for each dispatch refusal's code, message, and data, so the runtime emits and +// the CLI test formats the identical envelope. Messages are supplied per call site because each +// existing string is a published receipt an old consumer may match on. + +export type DispatchRefusalReceipt = { + code: 'task_not_found' | 'task_not_startable' | 'inject_rejected' + message: string + data: Record & { nextSteps: string[] } +} + +export function taskNotFoundRefusal( + message: string, + detail: { taskId: string; runId?: string } +): DispatchRefusalReceipt { + return { + code: 'task_not_found', + message, + data: { + ...detail, + nextSteps: [ + 'Run orca orchestration task-list --json in the bound Run to find the intended Task id.', + 'If the Task does not exist yet, create it with orca orchestration task-create --spec --json.' + ] + } + } +} + +export type TaskNotStartableDetail = { + taskId: string + status: string + unmetDependencies: string[] + retryOf?: string +} + +export function taskNotStartableRefusal( + message: string, + detail: TaskNotStartableDetail +): DispatchRefusalReceipt { + return { + code: 'task_not_startable', + message, + data: { ...detail, nextSteps: taskNotStartableNextSteps(detail) } + } +} + +function taskNotStartableNextSteps(detail: TaskNotStartableDetail): string[] { + if (detail.retryOf) { + return [ + `--retry-of must name the latest settled Dispatch of a failed or blocked Task; check orca orchestration dispatch-show --task ${detail.taskId} --json and orca orchestration worker-show --dispatch ${detail.retryOf} --json.` + ] + } + if (detail.unmetDependencies.length > 0) { + return [ + `Dependencies ${detail.unmetDependencies.join(', ')} are not completed. Wait for running ones with orca orchestration check --wait --json; retry or unblock failed ones before dispatching again.` + ] + } + if (detail.status === 'dispatched') { + return [ + `The Task already has an active Dispatch; inspect it with orca orchestration dispatch-show --task ${detail.taskId} --json.` + ] + } + return [ + `A ${detail.status} Task cannot be dispatched; create a new Task or use worker-start --retry-of for a failed attempt.` + ] +} + +// Why: the old five-name example read as an allowlist (#15125); derive from the field detection keys on so it cannot drift. +// Not filtered by `disabledTuiAgents` — that gates Orca's launchers, not detection, so a hand-started disabled agent still injects. +const RECOGNIZED_AGENT_PROCESS_NAMES = [ + ...new Set(Object.values(TUI_AGENT_CONFIG).map((config) => config.expectedProcess)) +].sort() + +export function buildInjectRejectionMessage(terminal: string): string { + return ( + `Cannot dispatch --inject to terminal ${terminal}: no recognized agent detected. ` + + `Orca detects these agent CLIs (${RECOGNIZED_AGENT_PROCESS_NAMES.join(', ')}). ` + + 'Start one in the terminal and let it finish launching, ' + + 'or dispatch without --inject and send the prompt manually.' + ) +} + +export type InjectRejectionReason = 'no_agent_detected' + +export function injectRejectedRefusal( + terminal: string, + reason: InjectRejectionReason +): DispatchRefusalReceipt { + return { + code: 'inject_rejected', + message: buildInjectRejectionMessage(terminal), + data: { + terminal, + reason, + nextSteps: [ + 'Start a recognized agent CLI in that terminal and wait for it to finish launching, or pick a terminal that already runs one.', + 'Alternatively dispatch without --inject and deliver the prompt with orca terminal send --terminal --text --enter --json.' + ] + } + } +}