From d19be485d380c8129b090a824a709e7ad7c1ea85 Mon Sep 17 00:00:00 2001
From: Neil <4138956+nwparker@users.noreply.github.com>
Date: Sun, 6 Sep 2026 02:48:34 -0700
Subject: [PATCH 01/32] test: reject unsupported app-server in golden agent
fixture (#19056)
---
tests/e2e/fixtures/golden-stub-agent/golden-stub-agent.js | 6 ++++++
1 file changed, 6 insertions(+)
diff --git a/tests/e2e/fixtures/golden-stub-agent/golden-stub-agent.js b/tests/e2e/fixtures/golden-stub-agent/golden-stub-agent.js
index d22cbf43b79..47810499db1 100644
--- a/tests/e2e/fixtures/golden-stub-agent/golden-stub-agent.js
+++ b/tests/e2e/fixtures/golden-stub-agent/golden-stub-agent.js
@@ -3,6 +3,12 @@
const READY_MARKER = 'GOLDEN_STUB_AGENT_READY'
const EXIT_MARKER = 'GOLDEN_STUB_AGENT_EXITED'
+// The interactive fixture does not implement Codex's JSONL app-server API.
+if (process.argv[2] === 'app-server') {
+ process.stderr.write("error: unrecognized subcommand 'app-server'\n")
+ process.exit(2)
+}
+
const ESC = '\x1b'
const keyboardProtocolMode = process.argv.includes('--keyboard-protocol')
const keyboardProtocolAgent = process.argv.includes('--grok') ? 'Grok' : 'Codex'
From ec64df335ee5790217a649dc3016f6d160282e8c Mon Sep 17 00:00:00 2001
From: Neil <4138956+nwparker@users.noreply.github.com>
Date: Sun, 6 Sep 2026 03:20:06 -0700
Subject: [PATCH 02/32] test: reject unsupported app-server in WSL golden stub
(#19062)
---
tests/e2e/helpers/wsl-golden-stub-agent.ts | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/tests/e2e/helpers/wsl-golden-stub-agent.ts b/tests/e2e/helpers/wsl-golden-stub-agent.ts
index 0b09644c14b..94ee5727d93 100644
--- a/tests/e2e/helpers/wsl-golden-stub-agent.ts
+++ b/tests/e2e/helpers/wsl-golden-stub-agent.ts
@@ -41,7 +41,7 @@ const BACKUP_EXISTING_STUB_SCRIPT =
// The marker is written first so stale-lock recovery only removes a stub this helper wrote.
const STAGE_SCRIPT =
`mkdir -p /usr/local/bin && : > ${WSL_STUB_STAGED_MARKER} && ` +
- `printf '#!/bin/sh\\necho GOLDEN_STUB_AGENT_READY\\nexec sleep 3600\\n' > ${WSL_STUB_PATH} && ` +
+ `printf '#!/bin/sh\\nif [ "$1" = app-server ]; then echo "error: unrecognized subcommand app-server" >&2; exit 2; fi\\necho GOLDEN_STUB_AGENT_READY\\nexec sleep 3600\\n' > ${WSL_STUB_PATH} && ` +
`chmod 0755 ${WSL_STUB_PATH}`
// The marker is written before the link so a crashed run over-reports rather than leaks a link.
From adcc30be3bc2a0bae7bc0f6cdd1bf55c7815f407 Mon Sep 17 00:00:00 2001
From: Neil <4138956+nwparker@users.noreply.github.com>
Date: Sun, 6 Sep 2026 03:29:05 -0700
Subject: [PATCH 03/32] test: canonicalize native Windows paths during
repository teardown (#19064)
---
tests/e2e/global-teardown.ts | 6 +++---
tests/e2e/global-teardown.unit.test.ts | 2 +-
2 files changed, 4 insertions(+), 4 deletions(-)
diff --git a/tests/e2e/global-teardown.ts b/tests/e2e/global-teardown.ts
index 63248cd36c8..0961eb0ee5c 100644
--- a/tests/e2e/global-teardown.ts
+++ b/tests/e2e/global-teardown.ts
@@ -10,7 +10,7 @@ import { readFileSync, existsSync, realpathSync, rmSync } from 'node:fs'
import { TEST_REPO_PATH_FILE } from './global-setup'
export function linkedWorktreePaths(testRepoDir: string): string[] {
- const root = realpathSync(testRepoDir)
+ const root = realpathSync.native(testRepoDir)
const output = execFileSync('git', ['-C', testRepoDir, 'worktree', 'list', '--porcelain'], {
encoding: 'utf8'
})
@@ -23,7 +23,7 @@ export function linkedWorktreePaths(testRepoDir: string): string[] {
if (!existsSync(recordedPath)) {
continue
}
- const canonicalPath = realpathSync(recordedPath)
+ const canonicalPath = realpathSync.native(recordedPath)
if (canonicalPath !== root) {
linked.add(canonicalPath)
}
@@ -32,7 +32,7 @@ export function linkedWorktreePaths(testRepoDir: string): string[] {
}
export function cleanupTestRepository(testRepoDir: string): void {
- const root = realpathSync(testRepoDir)
+ const root = realpathSync.native(testRepoDir)
let worktreePaths: string[] = []
try {
worktreePaths = linkedWorktreePaths(root)
diff --git a/tests/e2e/global-teardown.unit.test.ts b/tests/e2e/global-teardown.unit.test.ts
index b5fabfe3de8..fde77199434 100644
--- a/tests/e2e/global-teardown.unit.test.ts
+++ b/tests/e2e/global-teardown.unit.test.ts
@@ -39,7 +39,7 @@ describe('E2E global teardown ownership', () => {
git(repoPath, ['worktree', 'add', '-b', 'second-owned', secondWorktreePath])
expect(new Set(linkedWorktreePaths(repoPath))).toEqual(
- new Set([realpathSync(firstWorktreePath), realpathSync(secondWorktreePath)])
+ new Set([realpathSync.native(firstWorktreePath), realpathSync.native(secondWorktreePath)])
)
cleanupTestRepository(repoPath)
From f952f1ac96dce3d2f92c0aacf548f1ec6c53da69 Mon Sep 17 00:00:00 2001
From: Neil <4138956+nwparker@users.noreply.github.com>
Date: Sun, 6 Sep 2026 05:24:54 -0700
Subject: [PATCH 04/32] test: run real WSL terminal launch and paste in PR CI
(#19072)
* test: continuously exercise real WSL terminal launch and paste
* test: establish live WSL reader before changing default shell
* ci: pin WSL kernel installer and participation selectors
* ci: route deleted WSL paths and record immutable run evidence
* test: require exactly three WSL repetitions in lane contract
---
.../actions/setup-wsl-test-runtime/action.yml | 8 ++
.../actions/setup-wsl-test-runtime/setup.ps1 | 32 +++++++
.github/workflows/pr.yml | 14 ++++
.github/workflows/windows-wsl-e2e.yml | 74 ++++++++++++++++
config/reliability-gates.jsonc | 84 +++++++++++++++++++
config/scripts/pr-e2e-source-routing.mjs | 21 +++++
.../scripts/verify-wsl-e2e-participation.mjs | 54 ++++++++++++
.../verify-wsl-e2e-participation.test.mjs | 52 ++++++++++++
config/scripts/wsl-e2e-lane-contract.test.mjs | 69 +++++++++++++++
...inal-windows-shell-paste-ownership.spec.ts | 11 ++-
10 files changed, 415 insertions(+), 4 deletions(-)
create mode 100644 .github/actions/setup-wsl-test-runtime/action.yml
create mode 100644 .github/actions/setup-wsl-test-runtime/setup.ps1
create mode 100644 .github/workflows/windows-wsl-e2e.yml
create mode 100644 config/scripts/verify-wsl-e2e-participation.mjs
create mode 100644 config/scripts/verify-wsl-e2e-participation.test.mjs
create mode 100644 config/scripts/wsl-e2e-lane-contract.test.mjs
diff --git a/.github/actions/setup-wsl-test-runtime/action.yml b/.github/actions/setup-wsl-test-runtime/action.yml
new file mode 100644
index 00000000000..f919c2e75bc
--- /dev/null
+++ b/.github/actions/setup-wsl-test-runtime/action.yml
@@ -0,0 +1,8 @@
+name: Set up WSL test runtime
+description: Install a checksum-pinned Ubuntu WSL1 guest with executable Node and Git for real terminal tests.
+runs:
+ using: composite
+ steps:
+ - name: Provision Ubuntu WSL1
+ shell: pwsh
+ run: '& "${{ github.action_path }}/setup.ps1"'
diff --git a/.github/actions/setup-wsl-test-runtime/setup.ps1 b/.github/actions/setup-wsl-test-runtime/setup.ps1
new file mode 100644
index 00000000000..2fd012eb246
--- /dev/null
+++ b/.github/actions/setup-wsl-test-runtime/setup.ps1
@@ -0,0 +1,32 @@
+$ErrorActionPreference = 'Stop'
+if (-not $IsWindows) { throw 'WSL test provisioning requires a Windows runner' }
+
+$rootfs = Join-Path $env:RUNNER_TEMP 'noble-rootfs.tar.gz'
+Invoke-WebRequest 'https://releases.ubuntu.com/24.04.4/ubuntu-24.04.4-wsl-amd64.wsl' -OutFile $rootfs
+if ((Get-FileHash $rootfs -Algorithm SHA256).Hash.ToLowerInvariant() -ne '9b2f7730dc68227dd04a9f3e5eab86ad85caf556b8606ad94f1f29ff5c4fd3f5') { throw 'Ubuntu rootfs checksum mismatch' }
+$distroDir = Join-Path $env:RUNNER_TEMP 'orca-wsl-ubuntu'
+wsl.exe --import Ubuntu $distroDir $rootfs --version 1
+if ($LASTEXITCODE -ne 0) { throw "WSL import failed: $LASTEXITCODE" }
+wsl.exe --distribution Ubuntu --user root --exec /usr/bin/true
+if ($LASTEXITCODE -ne 0) { throw "WSL guest did not start: $LASTEXITCODE" }
+wsl.exe --distribution Ubuntu --user root --exec /usr/bin/apt-get update
+if ($LASTEXITCODE -ne 0) { throw "WSL apt update failed: $LASTEXITCODE" }
+wsl.exe --distribution Ubuntu --user root --exec /usr/bin/apt-get install --yes git curl xz-utils
+if ($LASTEXITCODE -ne 0) { throw "WSL git install failed: $LASTEXITCODE" }
+$kernelMsi = Join-Path $env:RUNNER_TEMP 'wsl_update_x64.msi'
+Invoke-WebRequest 'https://wslstorestorage.blob.core.windows.net/wslblob/wsl_update_x64.msi' -OutFile $kernelMsi
+if ((Get-FileHash $kernelMsi -Algorithm SHA256).Hash.ToLowerInvariant() -ne '4d09c776c8d45f70a202281d18e19be1118f53159b0c217a5274a31ce18525fe') { throw 'WSL kernel installer checksum mismatch' }
+$installer = Start-Process msiexec.exe -ArgumentList @('/i', $kernelMsi, '/quiet', '/norestart') -Wait -PassThru
+if ($installer.ExitCode -ne 0) { throw "WSL kernel installation failed: $($installer.ExitCode)" }
+wsl.exe --status
+if ($LASTEXITCODE -ne 0) { throw "WSL status failed: $LASTEXITCODE" }
+wsl.exe --distribution Ubuntu --user root --exec /usr/bin/curl --fail --silent --show-error --location https://nodejs.org/dist/v22.14.0/node-v22.14.0-linux-x64.tar.xz --output /tmp/orca-node.tar.xz
+if ($LASTEXITCODE -ne 0) { throw 'Node download failed' }
+$nodeHash = wsl.exe --distribution Ubuntu --user root --exec /usr/bin/sha256sum /tmp/orca-node.tar.xz
+if ($LASTEXITCODE -ne 0 -or -not ($nodeHash -match '^69b09dba5c8dcb05c4e4273a4340db1005abeafe3927efda2bc5b249e80437ec')) { throw 'Node checksum mismatch' }
+wsl.exe --distribution Ubuntu --user root --exec /usr/bin/tar -xJf /tmp/orca-node.tar.xz -C /usr/local --strip-components=1
+if ($LASTEXITCODE -ne 0) { throw 'Node extraction failed' }
+wsl.exe --distribution Ubuntu --user root --exec /usr/local/bin/node --version
+if ($LASTEXITCODE -ne 0) { throw 'Node cannot execute in WSL' }
+wsl.exe --list --verbose
+if ($LASTEXITCODE -ne 0) { throw "WSL enumeration failed: $LASTEXITCODE" }
diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml
index bd655eb801a..1fd141ee4a0 100644
--- a/.github/workflows/pr.yml
+++ b/.github/workflows/pr.yml
@@ -45,6 +45,7 @@ jobs:
test_files: ${{ steps.e2e_filter.outputs.test_files }}
ssh_source_changed: ${{ steps.e2e_filter.outputs.ssh_source_changed }}
native_ime_source_changed: ${{ steps.e2e_filter.outputs.native_ime_source_changed }}
+ wsl_source_changed: ${{ steps.e2e_filter.outputs.wsl_source_changed }}
steps:
- name: Checkout
uses: actions/checkout@v6
@@ -92,6 +93,9 @@ jobs:
# trigger on IME source rather than on a spec name in some route's list.
NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)"
echo "native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED" >> "$GITHUB_OUTPUT"
+ WSL_CHANGED="$(git diff --name-only --no-renames --diff-filter=ACDMR --merge-base "$BASE" "$HEAD")"
+ WSL_SOURCE_CHANGED="$(printf '%s\n' "$WSL_CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --wsl-source)"
+ echo "wsl_source_changed=$WSL_SOURCE_CHANGED" >> "$GITHUB_OUTPUT"
echo "Native IME source changed: $NATIVE_IME_SOURCE_CHANGED"
SHOULD_RUN="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --reusable-workflow)"
if [ "$SHOULD_RUN" = true ]; then
@@ -942,6 +946,16 @@ jobs:
contents: read
uses: ./.github/workflows/terminal-ime-e2e.yml
+ windows_wsl:
+ name: real WSL terminal
+ needs: code_paths
+ if: needs.code_paths.outputs.wsl_source_changed == 'true'
+ permissions:
+ contents: read
+ uses: ./.github/workflows/windows-wsl-e2e.yml
+ with:
+ ref: ${{ github.event.pull_request.head.sha }}
+
verify:
if: always()
needs:
diff --git a/.github/workflows/windows-wsl-e2e.yml b/.github/workflows/windows-wsl-e2e.yml
new file mode 100644
index 00000000000..fb781e25331
--- /dev/null
+++ b/.github/workflows/windows-wsl-e2e.yml
@@ -0,0 +1,74 @@
+name: Windows WSL terminal E2E
+
+on:
+ workflow_dispatch:
+ inputs:
+ ref:
+ description: Commit to validate
+ type: string
+ required: false
+ workflow_call:
+ inputs:
+ ref:
+ type: string
+ required: false
+
+permissions:
+ contents: read
+
+concurrency:
+ group: windows-wsl-e2e-${{ github.event.pull_request.number || github.ref }}
+ cancel-in-progress: true
+
+jobs:
+ wsl-terminal:
+ runs-on: windows-2022
+ timeout-minutes: 30
+ env:
+ NODE_OPTIONS: --max-old-space-size=4096
+ steps:
+ - uses: actions/checkout@v6
+ with:
+ ref: ${{ inputs.ref || github.sha }}
+ persist-credentials: false
+ - uses: ./.github/actions/setup-wsl-test-runtime
+ - uses: ./.github/actions/install-node-dependencies
+ with:
+ native-runtime: electron
+ - name: Build relay and Electron
+ run: |
+ pnpm run build:relay
+ if ($LASTEXITCODE -ne 0) { throw 'Relay build failed' }
+ pnpm exec electron-vite build --mode e2e
+ if ($LASTEXITCODE -ne 0) { throw 'Electron build failed' }
+ - name: Exercise real WSL launch and paste
+ env:
+ SKIP_BUILD: '1'
+ ORCA_E2E_FORWARD_APP_LOGS: '1'
+ PLAYWRIGHT_JSON_OUTPUT_FILE: test-results/wsl-results.json
+ run: >-
+ pnpm exec playwright test
+ tests/e2e/golden-tab-bar-agent-launch.spec.ts
+ tests/e2e/terminal-windows-shell-paste-ownership.spec.ts
+ --config tests/playwright.config.ts
+ --project=electron-headless
+ --grep "WSL"
+ --repeat-each=3
+ --workers=1
+ --reporter=list,json
+ - name: Require all nine WSL executions
+ if: always()
+ run: node config/scripts/verify-wsl-e2e-participation.mjs test-results/wsl-results.json
+ - name: Upload WSL participation report
+ uses: actions/upload-artifact@v7
+ if: always()
+ with:
+ name: windows-wsl-participation-report
+ path: test-results/wsl-results.json
+ retention-days: 3
+ - uses: actions/upload-artifact@v7
+ if: failure()
+ with:
+ name: windows-wsl-terminal-traces
+ path: test-results/
+ retention-days: 7
diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc
index d48a239354f..1153293f77c 100644
--- a/config/reliability-gates.jsonc
+++ b/config/reliability-gates.jsonc
@@ -18151,6 +18151,90 @@
"No p95 CI history or full product mutation proof."
],
"demotionRule": "Keep experimental while any recovery reproduction fails or any teardown, identity, resource-count, or rendered oracle flakes; never promote by extending sleeps or retries."
+ },
+ {
+ "id": "terminal.windows-wsl-launch-and-paste",
+ "title": "Real WSL terminal agent launch and paste ownership",
+ "maturity": "experimental",
+ "protection": "partial",
+ "owner": "terminal-runtime",
+ "layer": "electron-windows-wsl",
+ "surfaces": ["agent tab launch", "keyboard paste", "terminal runtime retention"],
+ "platforms": ["windows"],
+ "providers": ["wsl1", "wsl2"],
+ "coveredPlatforms": ["windows"],
+ "coveredProviders": ["wsl1"],
+ "coverageNotes": "Real WSL1 coverage: three scenarios each passed three times with no skips or retries; exact JSON report verified. Latest PR routing and installer-checksum follow-ups await CI. WSL2 remains untested.",
+ "motivatingLinks": ["https://github.com/stablyai/orca/actions/runs/34030832614"],
+ "invariant": "An agent launched into WSL runs in the guest; keyboard paste reaches exactly one owning PTY and preserves Linux content even after the default shell changes.",
+ "oracle": "Run the existing real WSL launch and two paste cases three times; require nine passes and zero skipped, unexpected, or flaky results in the Playwright JSON report.",
+ "commands": [
+ "gh workflow run windows-wsl-e2e.yml",
+ "pnpm exec playwright test tests/e2e/golden-tab-bar-agent-launch.spec.ts tests/e2e/terminal-windows-shell-paste-ownership.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1",
+ "node_modules/.bin/vitest run --config config/vitest.config.ts config/scripts/wsl-e2e-lane-contract.test.mjs config/scripts/verify-wsl-e2e-participation.test.mjs",
+ "gh run view 34031806291 --log"
+ ],
+ "testFiles": [
+ "tests/e2e/golden-tab-bar-agent-launch.spec.ts",
+ "tests/e2e/terminal-windows-shell-paste-ownership.spec.ts",
+ "config/scripts/wsl-e2e-lane-contract.test.mjs",
+ "config/scripts/verify-wsl-e2e-participation.test.mjs"
+ ],
+ "assertionRefs": [
+ {
+ "file": "tests/e2e/golden-tab-bar-agent-launch.spec.ts",
+ "assertions": ["requires a distro-only marker from the launched agent"]
+ },
+ {
+ "file": "tests/e2e/terminal-windows-shell-paste-ownership.spec.ts",
+ "assertions": [
+ "requires exact Linux pasted content and exactly one PTY write",
+ "retains WSL paste ownership after changing the default shell"
+ ]
+ },
+ {
+ "file": "config/scripts/verify-wsl-e2e-participation.test.mjs",
+ "assertions": ["rejects skipped, missing, substituted and retried scenarios"]
+ }
+ ],
+ "evidenceRuns": [
+ {
+ "date": "2026-09-06",
+ "runner": "ci",
+ "platform": "windows",
+ "result": "passed",
+ "command": "gh run view 34031806291 --log",
+ "durationSeconds": 210,
+ "summary": "Immutable run34031806291 at92fc5152: WSL1 launch3 and paste6 passed after reader-readiness correction; named-scenario verifier accepted actual JSON report with0skips0retries. Command retrieves recorded evidence; workflow_dispatch command above reruns current coverage."
+ }
+ ],
+ "runtimeBudget": {
+ "p95Seconds": 1800,
+ "scope": "CI job timeout; measured p95 is not established"
+ },
+ "flakeHistory": {
+ "status": "soaking",
+ "evidence": "Initial permanent-lane diagnostic8passed1failed on missing PTY before changing settings. After requiring guest-reader readiness before mutation, run34031806291 passed9/9. Two earlier setup validations also passed9/9. Long-term CI history remains missing."
+ },
+ "redGreenEvidence": {
+ "status": "partial",
+ "evidence": "Verifier rejects actual8pass1fail CI report and accepts actual9pass report. Unit contracts reject skips, missing or substituted scenarios and retried passes. No full application fault-mutation proof."
+ },
+ "performanceBudget": {
+ "required": false,
+ "evidence": "CI-only provisioning and routing; no application runtime changes."
+ },
+ "promotionCriteria": [
+ "Require all nine real WSL executions on the final workflow head.",
+ "Demonstrate missing or skipped WSL execution fails participation.",
+ "Collect repeated CI history before adding this experimental lane to required verification."
+ ],
+ "knownGaps": [
+ "WSL2 is not provisioned.",
+ "No SSH, folder-only workspace, packaged mixed-version, or live-service claim.",
+ "The new PR lane is outside verify until reliability is established."
+ ],
+ "demotionRule": "Keep experimental if provisioning or an execution flakes; never promote by skipping a case, raising timeouts, or retrying until green."
}
]
}
diff --git a/config/scripts/pr-e2e-source-routing.mjs b/config/scripts/pr-e2e-source-routing.mjs
index 5b698fb0b42..7e95869d13f 100644
--- a/config/scripts/pr-e2e-source-routing.mjs
+++ b/config/scripts/pr-e2e-source-routing.mjs
@@ -13,6 +13,18 @@ const NATIVE_IME_HARNESS =
/^(?:config\/scripts\/(?:run-terminal-ibus-hangul-e2e|terminal-ime-engagement-receipt)\.mjs$|tests\/e2e\/terminal-ime-(?:boundary-probe|byte-reader|engagement-receipt)\.ts$|tests\/e2e\/terminal-(?:ibus-hangul|hangul-terminating-digit|macos-2set-korean)-native\.spec\.ts$)/
export const PR_E2E_SOURCE_ROUTES = [
+ {
+ id: 'terminal.windows-wsl-launch-and-paste',
+ specs: [
+ 'tests/e2e/golden-tab-bar-agent-launch.spec.ts',
+ 'tests/e2e/terminal-windows-shell-paste-ownership.spec.ts'
+ ],
+ matches: (file) =>
+ isProductSource(file) &&
+ /^(?:config\/scripts\/verify-wsl-e2e-participation\.mjs$|src\/main\/(?:wsl[/-]|pty\/.*wsl|providers\/wsl)|src\/shared\/(?:wsl-|windows-terminal-shell)|src\/renderer\/src\/.*(?:terminal-paste|pty-paste)|tests\/e2e\/(?:golden-tab-bar-agent-launch\.spec|terminal-windows-shell-paste-ownership\.spec|helpers\/(?:wsl-golden-stub-agent|golden-stub-agent))|\.github\/(?:actions\/setup-wsl-test-runtime\/|workflows\/windows-wsl-e2e\.yml))/.test(
+ file
+ )
+ },
{
id: 'ephemeral-vm-runtime.rollback-readable-sidecar',
specs: ['tests/e2e/ephemeral-vm-provisioned-root.spec.ts'],
@@ -227,6 +239,13 @@ export function shouldRunReusablePrE2e(changedPaths) {
)
}
+export function hasWslSourceChange(changedPaths) {
+ const route = PR_E2E_SOURCE_ROUTES.find(
+ (candidate) => candidate.id === 'terminal.windows-wsl-launch-and-paste'
+ )
+ return changedPaths.some(route.matches)
+}
+
if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) {
let input = ''
process.stdin.setEncoding('utf8')
@@ -238,6 +257,8 @@ if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href)
process.stdout.write(`${hasSshSourceChange(changedPaths)}\n`)
} else if (process.argv.includes('--reusable-workflow')) {
process.stdout.write(`${shouldRunReusablePrE2e(changedPaths)}\n`)
+ } else if (process.argv.includes('--wsl-source')) {
+ process.stdout.write(`${hasWslSourceChange(changedPaths)}\n`)
} else if (process.argv.includes('--native-ime-source')) {
process.stdout.write(`${hasNativeImeSourceChange(changedPaths)}\n`)
} else {
diff --git a/config/scripts/verify-wsl-e2e-participation.mjs b/config/scripts/verify-wsl-e2e-participation.mjs
new file mode 100644
index 00000000000..21570ef7689
--- /dev/null
+++ b/config/scripts/verify-wsl-e2e-participation.mjs
@@ -0,0 +1,54 @@
+import { readFileSync } from 'node:fs'
+import { pathToFileURL } from 'node:url'
+
+export const WSL_TEST_TITLES = [
+ 'tab-bar + menu launches an agent inside WSL @tab-bar-agent-launch-golden',
+ 'WSL terminal keyboard paste preserves Linux shell content with one PTY owner',
+ 'existing WSL terminal keeps paste runtime after default shell changes'
+]
+
+export function verifyWslParticipation(report) {
+ const stats = report?.stats
+ if (
+ !stats ||
+ stats.expected !== 9 ||
+ stats.skipped !== 0 ||
+ stats.unexpected !== 0 ||
+ stats.flaky !== 0 ||
+ report.errors?.length
+ ) {
+ throw new Error(`WSL participation failed: ${JSON.stringify(stats)}`)
+ }
+ const counts = new Map(WSL_TEST_TITLES.map((title) => [title, 0]))
+ const visit = (suites) => {
+ for (const suite of suites ?? []) {
+ for (const spec of suite.specs ?? []) {
+ if (!counts.has(spec.title)) {
+ throw new Error(`Unexpected WSL scenario: ${spec.title}`)
+ }
+ for (const test of spec.tests ?? []) {
+ if (
+ test.expectedStatus !== 'passed' ||
+ test.results?.length !== 1 ||
+ test.results[0].status !== 'passed'
+ ) {
+ throw new Error(`WSL scenario did not pass without retries: ${spec.title}`)
+ }
+ counts.set(spec.title, counts.get(spec.title) + 1)
+ }
+ }
+ visit(suite.suites)
+ }
+ }
+ visit(report.suites)
+ for (const [title, count] of counts) {
+ if (count !== 3) {
+ throw new Error(`WSL scenario requires three executions: ${title} (${count})`)
+ }
+ }
+}
+
+if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) {
+ verifyWslParticipation(JSON.parse(readFileSync(process.argv[2], 'utf8')))
+ console.log('All three WSL scenarios passed three times without skips or retries.')
+}
diff --git a/config/scripts/verify-wsl-e2e-participation.test.mjs b/config/scripts/verify-wsl-e2e-participation.test.mjs
new file mode 100644
index 00000000000..ae2f0935879
--- /dev/null
+++ b/config/scripts/verify-wsl-e2e-participation.test.mjs
@@ -0,0 +1,52 @@
+import { describe, expect, it } from 'vitest'
+import { verifyWslParticipation, WSL_TEST_TITLES } from './verify-wsl-e2e-participation.mjs'
+
+function report() {
+ return {
+ stats: { expected: 9, skipped: 0, unexpected: 0, flaky: 0 },
+ suites: [
+ {
+ suites: [
+ {
+ specs: WSL_TEST_TITLES.map((title) => ({
+ title,
+ tests: Array.from({ length: 3 }, () => ({
+ expectedStatus: 'passed',
+ results: [{ status: 'passed' }]
+ }))
+ }))
+ }
+ ]
+ }
+ ]
+ }
+}
+
+describe('WSL participation', () => {
+ it('accepts all three named scenarios executed three times', () => {
+ expect(() => verifyWslParticipation(report())).not.toThrow()
+ })
+ it.each(['skipped', 'unexpected', 'flaky'])('rejects a nonzero %s result', (key) => {
+ const value = report()
+ value.stats[key] = 1
+ expect(() => verifyWslParticipation(value)).toThrow('participation failed')
+ })
+ it('rejects missing scenarios even when aggregate counts claim nine passes', () => {
+ const value = report()
+ value.suites[0].suites[0].specs.pop()
+ expect(() => verifyWslParticipation(value)).toThrow('requires three executions')
+ })
+ it('rejects an unrelated scenario substituted for an expected scenario', () => {
+ const value = report()
+ value.suites[0].suites[0].specs[0].title = 'native shell passes'
+ expect(() => verifyWslParticipation(value)).toThrow('Unexpected WSL scenario')
+ })
+ it('rejects a pass obtained after a failed attempt', () => {
+ const value = report()
+ value.suites[0].suites[0].specs[0].tests[0].results.unshift({ status: 'failed' })
+ expect(() => verifyWslParticipation(value)).toThrow('without retries')
+ })
+ it('rejects missing report content', () => {
+ expect(() => verifyWslParticipation({})).toThrow('participation failed')
+ })
+})
diff --git a/config/scripts/wsl-e2e-lane-contract.test.mjs b/config/scripts/wsl-e2e-lane-contract.test.mjs
new file mode 100644
index 00000000000..0369eb7c0c4
--- /dev/null
+++ b/config/scripts/wsl-e2e-lane-contract.test.mjs
@@ -0,0 +1,69 @@
+import { readFileSync } from 'node:fs'
+import { describe, expect, it } from 'vitest'
+import { parse } from 'yaml'
+import { hasWslSourceChange, selectPrE2eSpecs } from './pr-e2e-source-routing.mjs'
+
+const read = (path) => readFileSync(new URL(`../../${path}`, import.meta.url), 'utf8')
+
+describe('real WSL terminal lane', () => {
+ it.each([
+ 'config/scripts/verify-wsl-e2e-participation.mjs',
+ 'src/main/wsl-availability.ts',
+ 'src/main/wsl/wsl-runner.ts',
+ 'src/main/pty/wsl-orca-env.ts',
+ 'src/shared/wsl-login-shell-command.ts',
+ 'src/shared/windows-terminal-shell.ts',
+ 'tests/e2e/helpers/wsl-golden-stub-agent.ts',
+ 'tests/e2e/golden-tab-bar-agent-launch.spec.ts',
+ 'tests/e2e/terminal-windows-shell-paste-ownership.spec.ts',
+ '.github/actions/setup-wsl-test-runtime/setup.ps1',
+ '.github/workflows/windows-wsl-e2e.yml'
+ ])('routes %s to both WSL sentinels', (path) => {
+ expect(hasWslSourceChange([path])).toBe(true)
+ expect(selectPrE2eSpecs([path])).toEqual(
+ expect.arrayContaining([
+ 'tests/e2e/golden-tab-bar-agent-launch.spec.ts',
+ 'tests/e2e/terminal-windows-shell-paste-ownership.spec.ts'
+ ])
+ )
+ })
+
+ it.each([
+ 'docs/reference/wsl-command-execution.md',
+ 'src/main/wsl-availability.test.ts',
+ 'src/main/ssh/connection.ts'
+ ])('excludes unrelated or unit-only change %s', (path) => {
+ expect(hasWslSourceChange([path])).toBe(false)
+ })
+
+ it('runs the reusable lane at the immutable PR head', () => {
+ const pr = parse(read('.github/workflows/pr.yml'))
+ expect(pr.jobs.windows_wsl.if).toBe("needs.code_paths.outputs.wsl_source_changed == 'true'")
+ expect(pr.jobs.windows_wsl.with.ref).toBe('${{ github.event.pull_request.head.sha }}')
+ const detector = pr.jobs['code_paths'].steps.find(
+ (step) => step.name === 'Filter changed E2E specs'
+ )
+ expect(detector.run).toContain(
+ 'WSL_CHANGED="$(git diff --name-only --no-renames --diff-filter=ACDMR'
+ )
+ expect(detector.run).toContain(
+ '"$WSL_CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --wsl-source'
+ )
+ const workflow = parse(read('.github/workflows/windows-wsl-e2e.yml'))
+ const steps = workflow.jobs['wsl-terminal'].steps
+ expect(steps[0].with.ref).toBe('${{ inputs.ref || github.sha }}')
+ expect(steps.some((step) => step.uses === './.github/actions/setup-wsl-test-runtime')).toBe(
+ true
+ )
+ const exercise = steps.find((step) => step.name === 'Exercise real WSL launch and paste')
+ expect(exercise.run.split(/\s+/).filter((arg) => arg.startsWith('--repeat-each='))).toEqual([
+ '--repeat-each=3'
+ ])
+ expect(exercise.run).toContain('--grep "WSL"')
+ const receipt = steps.find((step) => step.name === 'Require all nine WSL executions')
+ expect(receipt.if).toBe('always()')
+ expect(receipt.run).toBe(
+ 'node config/scripts/verify-wsl-e2e-participation.mjs test-results/wsl-results.json'
+ )
+ })
+})
diff --git a/tests/e2e/terminal-windows-shell-paste-ownership.spec.ts b/tests/e2e/terminal-windows-shell-paste-ownership.spec.ts
index c94bf8a7f38..ecaf5baf8bf 100644
--- a/tests/e2e/terminal-windows-shell-paste-ownership.spec.ts
+++ b/tests/e2e/terminal-windows-shell-paste-ownership.spec.ts
@@ -423,10 +423,6 @@ test.describe('Windows terminal shell paste ownership', () => {
const wslDistro = await configureActiveProjectWslRuntime(orcaPage)
test.skip(!wslDistro, 'No WSL distro is available on this Windows host')
const tabId = await createWindowsProjectRuntimeTerminalTab(orcaPage, 'wsl.exe')
- await updateWindowsDefaultShellSetting(orcaPage, 'cmd.exe')
- await expect(
- orcaPage.locator(`[data-testid="sortable-tab"][data-tab-id="${tabId}"] [data-shell-icon]`)
- ).toHaveAttribute('data-shell-icon', 'wsl.exe')
await waitForActiveTerminalManager(orcaPage, 30_000)
await installTerminalPtyWriteSpy(electronApp)
@@ -454,6 +450,13 @@ test.describe('Windows terminal shell paste ownership', () => {
scriptStarted = true
await waitForTerminalOutput(orcaPage, `PASTE_READY_${runId}`, 10_000)
+ // Exercise a live WSL process across the settings change.
+ await updateWindowsDefaultShellSetting(orcaPage, 'cmd.exe')
+ await expect(
+ orcaPage.locator(`[data-testid="sortable-tab"][data-tab-id="${tabId}"] [data-shell-icon]`)
+ ).toHaveAttribute('data-shell-icon', 'wsl.exe')
+ expect(await waitForActivePanePtyId(orcaPage)).toBe(ptyId)
+
await clearTerminalPtyWriteLog(electronApp)
await orcaPage.evaluate((text) => window.api.ui.writeClipboardText(text), payload)
await focusActiveTerminalInput(orcaPage)
From 1d2e00819ffe1197ce11ff8a7fd599891b9db019 Mon Sep 17 00:00:00 2001
From: Neil <4138956+nwparker@users.noreply.github.com>
Date: Sun, 6 Sep 2026 07:42:50 -0700
Subject: [PATCH 05/32] test: restore SSH bulk-open freeze coverage in headed
CI (#19081)
* test: restore SSH bulk-open freeze coverage in headed CI
* test: record ten passing headed SSH freeze repetitions
* test: record ten passing headed SSH freeze repetitions
* test: route changed SSH freeze spec only to its dedicated lane
---
.github/workflows/e2e.yml | 4 +-
config/reliability-gates.jsonc | 39 +++++++++++++++----
config/scripts/pr-e2e-gate-contract.test.mjs | 4 +-
.../run-ssh-docker-bulk-open-freeze-e2e.mjs | 2 +-
config/scripts/run-ssh-docker-e2e.mjs | 28 ++-----------
.../ssh-docker-bulk-open-freeze-repro.spec.ts | 29 ++------------
6 files changed, 44 insertions(+), 62 deletions(-)
diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml
index a94a7ea2ba5..abafb5cd6c8 100644
--- a/.github/workflows/e2e.yml
+++ b/.github/workflows/e2e.yml
@@ -227,6 +227,7 @@ jobs:
mapfile -t TEST_FILES < <(jq -r '.[] | select(
. != "tests/e2e/ssh-startup-exec-readiness.spec.ts" and
. != "tests/e2e/paired-startup-exec-readiness.spec.ts" and
+ . != "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts" and
. != "tests/e2e/terminal-ibus-hangul-native.spec.ts"
)' <<<"$TEST_FILES_JSON")
if [ "${#TEST_FILES[@]}" -eq 0 ]; then
@@ -262,12 +263,13 @@ jobs:
needs: [build, prepare-native-cache]
# effect of one route listing a startup-readiness spec — pruning that spec would have
# silently retired the whole lane. The signal is now derived from the SSH routes directly.
- # The two spec clauses stay for their honest purpose: changed-e2e hands these specs to this
+ # The explicit spec clauses stay for their honest purpose: changed-e2e hands these specs to this
# lane, so editing one must still run it here.
if: >-
inputs.test_files == '' ||
inputs.ssh_source_changed == 'true' ||
contains(inputs.test_files, 'tests/e2e/ssh-startup-exec-readiness.spec.ts') ||
+ contains(inputs.test_files, 'tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts') ||
contains(inputs.test_files, 'tests/e2e/paired-startup-exec-readiness.spec.ts')
runs-on: ubuntu-latest
# Why 60: this lane now also runs the remaining Docker-SSH specs serially. They average
diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc
index 1153293f77c..76ba45e0432 100644
--- a/config/reliability-gates.jsonc
+++ b/config/reliability-gates.jsonc
@@ -18035,19 +18035,27 @@
],
"platforms": ["macos", "linux", "windows"],
"providers": ["ssh"],
- "coveredPlatforms": ["macos"],
+ "coveredPlatforms": [
+ "macos",
+ "linux"
+ ],
"coveredProviders": ["ssh"],
- "coverageNotes": "A macOS Electron client drives a Linux Docker SSH execution host. The six-spec suite passed ten enabled cases with clean worker exit (5.2m). The formerly skipped frozen-host input case now waits for recovered authority before sending input and passed four separate executions (one initial and three repetitions). The existing flooded-shell fixme remains an explicitly reproduced application gap.",
+ "coverageNotes": "A macOS Electron client drives a Linux Docker SSH execution host. The six-spec suite passed ten enabled cases with clean worker exit (5.2m). The formerly skipped frozen-host input case now waits for recovered authority before sending input and passed four separate executions (one initial and three repetitions). The existing flooded-shell fixme remains an explicitly reproduced application gap. The bulk-open freeze reproduction runs in Linux headed CI with SwiftShader on Xvfb: headless Linux schedules idle animation frames about 1s apart, invalidating the foreground interaction measurement. Original uninstrumented five-pane workload passed all ten repetitions with zero retries/skips in 6.6m; bulk-open lag 79.3–147.8ms and interaction 127.1–155.9ms, unchanged 2500ms/5000ms budgets. Run 34037669843, head f25eab3fd7d723509ced026633f80b193a139b76, excludes unmerged replay-input application fix #19075.",
"motivatingLinks": [
"https://github.com/stablyai/orca/issues/18018",
"https://github.com/stablyai/orca/pull/18546",
- "https://github.com/stablyai/orca/issues/12547"
+ "https://github.com/stablyai/orca/issues/12547",
+ "https://github.com/stablyai/orca/issues/16764",
+ "https://github.com/stablyai/orca/actions/runs/34037450427",
+ "https://github.com/stablyai/orca/actions/runs/34037669843"
],
"invariant": "Transport loss and frozen-host silence must preserve the remote session; host relay loss may rebind a pane without accumulating reattachable leases. Reconnects must preserve usable terminal content, bounded PTYs/fds/processes, complete large listings, and independently recoverable watcher processes. Electron test shutdown must release inherited pipes after confirmed root exit without closing live-process pipes.",
"oracle": "Poll a changed connected SSH authority after injected faults, then require terminal output and appropriate PTY identity. Read remote process/fd state, listFiles replies, and rendered explorer rows. Resolve Playwright cleanup only after the root process exits and its inherited pipes close; live-process pipes remain untouched.",
"commands": [
"ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-transport-drop-recovery.spec.ts tests/e2e/ssh-docker-half-open-link.spec.ts tests/e2e/ssh-docker-quick-open-large-listing.spec.ts tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts tests/e2e/ssh-docker-resource-accumulation.spec.ts tests/e2e/ssh-docker-watcher-isolation.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1",
- "pnpm exec vitest run --config config/vitest.config.ts tests/e2e/helpers/electron-process-shutdown.unit.test.ts"
+ "pnpm exec vitest run --config config/vitest.config.ts tests/e2e/helpers/electron-process-shutdown.unit.test.ts",
+ "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts --config tests/playwright.config.ts --project=electron-headful --workers=1",
+ "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts --config tests/playwright.config.ts --project=electron-headful --workers=1 --repeat-each=10"
],
"testFiles": [
"tests/e2e/ssh-docker-transport-drop-recovery.spec.ts",
@@ -18056,7 +18064,8 @@
"tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts",
"tests/e2e/ssh-docker-resource-accumulation.spec.ts",
"tests/e2e/ssh-docker-watcher-isolation.spec.ts",
- "tests/e2e/helpers/electron-process-shutdown.unit.test.ts"
+ "tests/e2e/helpers/electron-process-shutdown.unit.test.ts",
+ "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts"
],
"assertionRefs": [
{
@@ -18101,6 +18110,12 @@
"releases inherited pipes after confirmed exit, including prior exit",
"retains live-process pipes on shutdown timeout"
]
+ },
+ {
+ "file": "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts",
+ "assertions": [
+ "five flooding SSH panes remain below unchanged 2500ms soft and 5000ms hard freeze budgets during bulk reopen and two double-animation-frame view changes"
+ ]
}
],
"evidenceRuns": [
@@ -18121,6 +18136,15 @@
"command": "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-transport-drop-recovery.spec.ts tests/e2e/ssh-docker-half-open-link.spec.ts tests/e2e/ssh-docker-quick-open-large-listing.spec.ts tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts tests/e2e/ssh-docker-resource-accumulation.spec.ts tests/e2e/ssh-docker-watcher-isolation.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1",
"durationSeconds": 312,
"summary": "Six specs: ten passed, two existing fixme skipped, clean worker shutdown. Baseline same enabled suite: ten passed but worker teardown timed out (7.3m)."
+ },
+ {
+ "date": "2026-09-06",
+ "runner": "ci",
+ "platform": "linux",
+ "result": "passed",
+ "command": "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts --config tests/playwright.config.ts --project=electron-headful --workers=1 --repeat-each=10",
+ "durationSeconds": 396,
+ "summary": "Original uninstrumented five-pane workload passed all ten repetitions with zero retries/skips in 6.6m; bulk-open lag 79.3–147.8ms and interaction 127.1–155.9ms, unchanged 2500ms/5000ms budgets. Run 34037669843, head f25eab3fd7d723509ced026633f80b193a139b76, excludes unmerged replay-input application fix #19075."
}
],
"runtimeBudget": {
@@ -18146,9 +18170,10 @@
],
"knownGaps": [
"The disconnected 48MB flood still loses its relay channel: original post-flood input marker failed in 60s, and waiting for the finite producer completion marker failed in 120s. It remains an explicit #18018 fixme reproduction; frozen-host input is re-enabled after four successful runs.",
- "Linux and Windows desktop clients, WSL, folder workspaces, paired runtimes and live agent CLIs are not exercised by these Docker specs.",
+ "Linux headed CI covers the bulk-open freeze reproduction; Windows clients, WSL, folder workspaces, paired runtimes and live agent CLIs are not covered by that result.",
"Some legacy assertions inspect terminal serialization or backing state rather than rendered DOM; no blanket visual coverage claim.",
- "No p95 CI history or full product mutation proof."
+ "No p95 CI history or full product mutation proof.",
+ "One headless bulk-open probe reached 6478.6ms in run 34035957303; animation-frame scheduling explains the consistent interaction failures, but does not directly explain that isolated timer-lag outlier. Long-term headed CI soak remains outstanding."
],
"demotionRule": "Keep experimental while any recovery reproduction fails or any teardown, identity, resource-count, or rendered oracle flakes; never promote by extending sleeps or retries."
},
diff --git a/config/scripts/pr-e2e-gate-contract.test.mjs b/config/scripts/pr-e2e-gate-contract.test.mjs
index 41f9338ab75..f5295faf1ad 100644
--- a/config/scripts/pr-e2e-gate-contract.test.mjs
+++ b/config/scripts/pr-e2e-gate-contract.test.mjs
@@ -168,6 +168,7 @@ describe('PR E2E gate contract', () => {
expect(changedRun.env.TEST_FILES_JSON).toBe('${{ inputs.test_files }}')
expect(changedRun.run).toContain('. != "tests/e2e/ssh-startup-exec-readiness.spec.ts"')
expect(changedRun.run).toContain('. != "tests/e2e/paired-startup-exec-readiness.spec.ts"')
+ expect(changedRun.run).toContain('. != "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts"')
expect(changedRun.run).toContain('if [ "${#TEST_FILES[@]}" -eq 0 ]')
expect(changedRun.run).toContain('grep -l \'@headful\' "${TEST_FILES[@]}"')
expect(changedRun.run).toContain('E2E_PROJECT_ARGS+=(--project=electron-headful)')
@@ -379,8 +380,7 @@ describe('PR E2E gate contract', () => {
// run-ssh-docker-e2e.mjs so the gap stays legible rather than looking like coverage.
const unreachableSpecs = new Set([
'tests/e2e/ssh-docker-relay-perf.spec.ts',
- 'tests/e2e/ssh-codex-display-artifacts-repro.spec.ts',
- 'tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts'
+ 'tests/e2e/ssh-codex-display-artifacts-repro.spec.ts'
])
// Why comments are stripped: this file's own runner lists the two exempt specs by name in a
// prose comment. A substring scan over raw text would count any spec merely *discussed* in a
diff --git a/config/scripts/run-ssh-docker-bulk-open-freeze-e2e.mjs b/config/scripts/run-ssh-docker-bulk-open-freeze-e2e.mjs
index 153f3fb5bc6..294bf7e2c7b 100644
--- a/config/scripts/run-ssh-docker-bulk-open-freeze-e2e.mjs
+++ b/config/scripts/run-ssh-docker-bulk-open-freeze-e2e.mjs
@@ -29,7 +29,7 @@ const result = spawnSync(
'--config',
'tests/playwright.config.ts',
'--project',
- 'electron-headless',
+ 'electron-headful',
'--workers=1',
...extraArgs
],
diff --git a/config/scripts/run-ssh-docker-e2e.mjs b/config/scripts/run-ssh-docker-e2e.mjs
index 9ab44b8457e..faa354689cb 100644
--- a/config/scripts/run-ssh-docker-e2e.mjs
+++ b/config/scripts/run-ssh-docker-e2e.mjs
@@ -33,31 +33,8 @@ if (runtime.status !== 0) {
// all. Recorded as a real gap, not as coverage living somewhere else.
// ssh-codex-display-artifacts-repro.spec.ts — installs a real remote codex binary that CI
// runners do not have (observed as `spawn codex ENOENT`). Runs in no CI lane at all.
-// ssh-docker-bulk-open-freeze-repro.spec.ts — un-rotted and now measurable, and marked
-// `test.fixme` because its oracle cannot gate. Absent from this list AND skipped, so the
-// two cannot drift: it is also reachable from the changed-specs lane whenever the spec
-// itself is edited, and a wall-clock oracle that fails there is worth no more than one
-// that fails here.
-// The rot (#16764) is fixed: the stale call sites are repaired, it connects after session
-// restore instead of before, and readiness keys on the repeating flood marker rather than
-// a one-shot READY line the flood buries within ~16ms. It runs end to end and prints a
-// measurement instead of dying on a call site.
-// What it is NOT is portable. Three runs of the same measurement path:
-// developer workstation: hiddenFlood 2.1ms bulkOpen 41.5ms interaction 53.6ms
-// GitHub ubuntu runner A: hiddenFlood 1.5ms bulkOpen 2575.6ms interaction 3464.2ms
-// GitHub ubuntu runner B: hiddenFlood 0.2ms bulkOpen 397.4ms interaction 3386.7ms
-// bulkOpen swings 6.5x between two CI runs of the same code, so a fixed threshold on it is
-// a coin flip; interaction sits stably ~64x over the workstation figure because it times a
-// view remount, not the renderer freeze the issue reports, and only shares the budget
-// constant because both are milliseconds. Every failure so far is the soft budget; hard
-// has never tripped, and the relay was still streaming each time — the budget failed, not
-// the product. Same rule as ssh-docker-relay-perf above. Gating needs a distribution
-// first, then a host-relative oracle; a bigger constant, or a ratio picked from three
-// samples, is the same arbitrary number in different clothes.
-// COVERAGE GAP, recorded as such: 5 simultaneously flooding SSH panes exercise writer
-// saturation, ACK/credit accounting and per-pane polling together, and nothing else covers
-// that combination. Flip `test.fixme` back to `test` to run it. Tracked in
-// stablyai/orca#16764.
+// The bulk-open frame probe runs headed: headless Linux compositing schedules idle RAFs
+// roughly 1s apart, so it cannot measure foreground interaction against the same budget.
//
// Why both projects: ssh-port-forward-lifecycle is @headful, which the headless project
// grep-inverts away.
@@ -87,6 +64,7 @@ const result = spawnSync(
'tests/e2e/ssh-ai-vault-session-history.spec.ts',
'tests/e2e/ssh-cold-activation-restore.spec.ts',
'tests/e2e/ssh-cold-hydration-gap-tab-seeding.spec.ts',
+ 'tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts',
'tests/e2e/ssh-docker-half-open-link.spec.ts',
'tests/e2e/ssh-docker-quick-open-large-listing.spec.ts',
'tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts',
diff --git a/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts b/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts
index 2de70c199d3..4f51b346b91 100644
--- a/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts
+++ b/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts
@@ -53,32 +53,9 @@ function continuousFloodCommand(runId: string, index: number): string {
test.describe('R2 Docker SSH bulk-open freeze', () => {
test.skip(!RUN_DOCKER_SSH, 'Set ORCA_E2E_SSH_DOCKER=1 to run Docker SSH freeze repro')
- // Fixme: un-rotted and measurable, but its oracle is wall-clock and does not survive a change of
- // host, so it cannot gate. Three runs of the same measurement path:
- //
- // host hiddenFlood bulkOpen interaction
- // developer workstation 2.1ms 41.5ms 53.6ms
- // GitHub ubuntu runner A 1.5ms 2575.6ms 3464.2ms
- // GitHub ubuntu runner B 0.2ms 397.4ms 3386.7ms
- //
- // Two separate problems, and neither is the product. `bulkOpenMaxLagMs` swings 6.5x between two
- // CI runs of the same code, so a fixed threshold on it is a coin flip; `interactionProbeMs` sits
- // stably ~64x over the workstation figure, because it times two `setActiveView` round trips
- // through a double rAF — a view remount cost, not the renderer freeze #16764 reports. It shares
- // SOFT/HARD_FREEZE_LAG_MS with the lag probe only because both are milliseconds. `hardFreeze`
- // has never tripped on any host; the failure is always the soft budget.
- //
- // Not converted to a ratio against a calibration run: with a 6.5x within-host swing on the very
- // quantity that would be normalized, a threshold picked from three samples is the same arbitrary
- // constant in dimensionless clothing. Gating needs a distribution first.
- //
- // Kept executable rather than deleted: flip `test.fixme` back to `test` to run it, which is how
- // the numbers above were taken. Tracked in stablyai/orca#16764.
- //
- // The cost is real and is recorded in run-ssh-docker-e2e.mjs: 5 simultaneously flooding SSH panes
- // exercise writer saturation, ACK/credit accounting and per-pane polling together, and nothing
- // else covers that combination. It is a gap, not coverage living somewhere else.
- test.fixme('bulk-open many flooding SSH terminals and measure renderer lag @freeze-repro', async ({
+ // Headless Linux disables compositing and schedules idle RAFs ~1s apart; use headed CI.
+ // Headed SwiftShader restores ~16ms frames without changing the freeze budgets.
+ test('bulk-open many flooding SSH terminals and measure renderer lag @freeze-repro @headful', async ({
orcaPage,
registerPostElectronShutdownCleanup
}, testInfo) => {
From 3631f886a7a4baf9a24cb525b68828281954631d Mon Sep 17 00:00:00 2001
From: Neil <4138956+nwparker@users.noreply.github.com>
Date: Sun, 6 Sep 2026 08:25:13 -0700
Subject: [PATCH 06/32] test: enable direct and client-hosted SSH browser
coverage (#19090)
* test: enable direct and client-hosted SSH browser coverage
* test: record twelve passing SSH browser journey repetitions
* test: distinguish SSH journey evidence from unit runtime budget
---
.github/workflows/e2e.yml | 4 ++
config/reliability-gates.jsonc | 63 ++++++++++++++++---
config/scripts/run-ssh-docker-e2e.mjs | 10 +--
.../scripts/ssh-browser-e2e-routing.test.mjs | 26 ++++++++
4 files changed, 90 insertions(+), 13 deletions(-)
create mode 100644 config/scripts/ssh-browser-e2e-routing.test.mjs
diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml
index abafb5cd6c8..0b529cde7e3 100644
--- a/.github/workflows/e2e.yml
+++ b/.github/workflows/e2e.yml
@@ -227,6 +227,8 @@ jobs:
mapfile -t TEST_FILES < <(jq -r '.[] | select(
. != "tests/e2e/ssh-startup-exec-readiness.spec.ts" and
. != "tests/e2e/paired-startup-exec-readiness.spec.ts" and
+ . != "tests/e2e/local-ssh-browser-routing.spec.ts" and
+ . != "tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts" and
. != "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts" and
. != "tests/e2e/terminal-ibus-hangul-native.spec.ts"
)' <<<"$TEST_FILES_JSON")
@@ -268,6 +270,8 @@ jobs:
if: >-
inputs.test_files == '' ||
inputs.ssh_source_changed == 'true' ||
+ contains(inputs.test_files, 'tests/e2e/local-ssh-browser-routing.spec.ts') ||
+ contains(inputs.test_files, 'tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts') ||
contains(inputs.test_files, 'tests/e2e/ssh-startup-exec-readiness.spec.ts') ||
contains(inputs.test_files, 'tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts') ||
contains(inputs.test_files, 'tests/e2e/paired-startup-exec-readiness.spec.ts')
diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc
index 76ba45e0432..5609318325c 100644
--- a/config/reliability-gates.jsonc
+++ b/config/reliability-gates.jsonc
@@ -4596,11 +4596,19 @@
],
"platforms": ["macos", "linux", "windows"],
"providers": ["remote-runtime", "ssh", "wsl"],
- "coveredPlatforms": ["macos"],
- "coveredProviders": [],
- "coverageNotes": "Deterministic main-process tests cover versioned delimiter-safe aggregate partition derivation, path-safe opaque names, durable collision metadata, oversized or corrupt metadata refusal, missing-sidecar Chromium-data refusal, bounded binding/live-page admission, immediate SOCKS5 setup with Chromium loopback bypass disabled, inherited-connection closure, exact proxy verification before allowlisting, concurrent setup coalescing, live proxy-retarget refusal, token-safe page replacement, existing browser-profile policy installation, active Orca-profile storage scoping, blank-only initial attachment, arbitrary initial-navigation denial, and fail-closed per-guest WebRTC policy through delayed or failed cleanup. The production client-page executor prepares, registers, and grants the exact route page. A real Electron A/B capture proves HTTP, HTTPS, WebSocket, redirects, subresources, downloads, and a `.test` hostname traverse SOCKS with no direct target connection. A two-launch control proves immediate setProxy routes a forced persisted-worker wake and later worker fetch. A separate capture proves the protected guest sends zero direct STUN packets. A further capture proves non-WebRTC UDP is also contained: a WebTransport session and a fetch forced onto QUIC both reach the desktop directly in the control arm and emit zero datagrams through the route partition, and the shipped disable-features list hides the Direct Sockets constructors whose mere construction kills a control-arm renderer. DNS prefetch is a tripwire over an accepted residual rather than a guard: Electron 43 inherits Chromium's PrefetchDNS, so a `` host resolves on the desktop resolver outside the tunnel, and a source census keeps any DoH host-resolver mode from widening that leak. Network-service restart and provider journeys remain uncovered.",
+ "coveredPlatforms": [
+ "macos",
+ "linux"
+ ],
+ "coveredProviders": [
+ "ssh",
+ "remote-runtime"
+ ],
+ "coverageNotes": "Deterministic main-process tests cover versioned delimiter-safe aggregate partition derivation, path-safe opaque names, durable collision metadata, oversized or corrupt metadata refusal, missing-sidecar Chromium-data refusal, bounded binding/live-page admission, immediate SOCKS5 setup with Chromium loopback bypass disabled, inherited-connection closure, exact proxy verification before allowlisting, concurrent setup coalescing, live proxy-retarget refusal, token-safe page replacement, existing browser-profile policy installation, active Orca-profile storage scoping, blank-only initial attachment, arbitrary initial-navigation denial, and fail-closed per-guest WebRTC policy through delayed or failed cleanup. The production client-page executor prepares, registers, and grants the exact route page. A real Electron A/B capture proves HTTP, HTTPS, WebSocket, redirects, subresources, downloads, and a `.test` hostname traverse SOCKS with no direct target connection. A two-launch control proves immediate setProxy routes a forced persisted-worker wake and later worker fetch. A separate capture proves the protected guest sends zero direct STUN packets. A further capture proves non-WebRTC UDP is also contained: a WebTransport session and a fetch forced onto QUIC both reach the desktop directly in the control arm and emit zero datagrams through the route partition, and the shipped disable-features list hides the Direct Sockets constructors whose mere construction kills a control-arm renderer. DNS prefetch is a tripwire over an accepted residual rather than a guard: Electron 43 inherits Chromium's PrefetchDNS, so a `` host resolves on the desktop resolver outside the tunnel, and a source census keeps any DoH host-resolver mode from widening that leak. Network-service restart and WSL provider journeys remain uncovered. Four Linux Docker SSH browser baseline scenarios passed: direct-host routing, unavailable-host local escape, forwarding refusal, and paired client-hosted reconnect. All four scenarios subsequently passed three repetitions each (12 passes, no skips or retries) in Linux CI run 34040309638, with unchanged assertions and timeouts.",
"motivatingLinks": [
- "https://linear.app/stably/issue/STA-4150/refactor-remote-browser-to-client-hosted-electron-webviews"
+ "https://linear.app/stably/issue/STA-4150/refactor-remote-browser-to-client-hosted-electron-webviews",
+ "https://github.com/stablyai/orca/actions/runs/34039986047",
+ "https://github.com/stablyai/orca/actions/runs/34040309638"
],
"invariant": "A client-hosted partition is derived only in main from stable Orca-profile, browser-profile, authority-connection, and execution-host identities. Raw identities and individually linkable component hashes never enter its path-safe partition name. Durable binding metadata must match and precede Chromium partition data before reuse. One live partition never changes execution host or proxy endpoint. Fixed SOCKS5 setup starts immediately after Session creation, before policy installation can yield or a persisted worker is awakened; no partition enters the webview allowlist until browser policy is installed, inherited connections are closed, and resolveProxy returns exactly that one listener. Initial route-partition attachment is blank-only. Its exact WebContents is quarantined before applying non-proxied WebRTC denial and remains navigation- and popup-denied if policy application or cleanup fails. Distinct live partitions, retained logical page generations, durable bindings, and binding-file reads remain bounded. No UDP transport a route-partition page can reach — WebRTC, WebTransport, or forced QUIC — emits a datagram to the desktop, the Direct Sockets constructors stay absent from every guest so no page can kill its renderer, and the process never enables a DoH host-resolver mode.",
"oracle": "Derive two delimiter-adversarial identities and require distinct full-digest path-safe partitions with no raw IDs or component hashes. Persist one binding, reload it, and reject replacement, malformed or oversized state, Chromium data without matching metadata, and the 513th binding. Prepare one partition and require setProxy with <-loopback> to be invoked immediately after getSession and before policy setup, then closeAllConnections and exact SOCKS5 resolveProxy while isAllowedPartition remains false; only then may it become live. Under real Electron, require direct controls for HTTP, HTTPS, WebSocket, redirects, subresources, and downloads, then require the fixed SOCKS session to route every equivalent request plus an otherwise-unresolvable `.test` hostname with zero direct target connections. Across two Electron launches, require immediate setProxy to route a forced worker wake and post-verification fetch. Reject DIRECT, endpoint retargeting, and capacity overflow. Require quarantine before disable_non_proxied_udp and admission; under real Electron require the unprotected control to emit STUN and the protected guest to emit zero direct UDP packets. Under real Electron require a direct control to emit WebTransport and forced-QUIC datagrams and the SOCKS partition to emit none, require an explicitly enabled Direct Sockets control to expose the constructors and die on construction, and require the shipped disable-features list to leave them undefined with the renderer alive. Capture a route partition's netLog across a dns-prefetch load and require the prefetched host to appear on a local resolver task while an unreferenced control host appears nowhere; require no source file to set a non-'off' secureDnsMode.",
@@ -4609,7 +4617,9 @@
"pnpm exec vitest run --config config/vitest.config.ts src/main/browser/browser-route-webcontents-registry.test.ts src/main/browser/browser-route-webrtc-egress.electron.test.ts",
"pnpm exec vitest run --config config/vitest.config.ts src/main/browser/browser-route-session-registry.test.ts src/main/browser/browser-route-persisted-worker-egress.electron.test.ts src/main/browser/browser-route-webrtc-egress.electron.test.ts",
"pnpm exec vitest run --config config/vitest.config.ts src/main/browser/browser-route-tcp-egress.electron.test.ts",
- "pnpm exec vitest run --config config/vitest.config.ts src/main/browser/browser-route-h3-egress.electron.test.ts src/main/browser/browser-route-dns-prefetch.electron.test.ts src/main/startup/secure-dns-census.test.ts"
+ "pnpm exec vitest run --config config/vitest.config.ts src/main/browser/browser-route-h3-egress.electron.test.ts src/main/browser/browser-route-dns-prefetch.electron.test.ts src/main/startup/secure-dns-census.test.ts",
+ "ORCA_E2E_SSH_DOCKER=1 ORCA_E2E_LOCAL_SSH_BROWSER=1 ORCA_E2E_SSH_CLIENT_HOSTED_BROWSER=1 ORCA_E2E_WEB_CLIENT=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/local-ssh-browser-routing.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1 --repeat-each=3",
+ "ORCA_E2E_SSH_DOCKER=1 ORCA_E2E_LOCAL_SSH_BROWSER=1 ORCA_E2E_SSH_CLIENT_HOSTED_BROWSER=1 ORCA_E2E_WEB_CLIENT=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1 --repeat-each=3"
],
"testFiles": [
"src/main/browser/browser-route-identity.test.ts",
@@ -4624,7 +4634,9 @@
"src/main/browser/browser-route-webcontents-registry.test.ts",
"src/main/browser/browser-session-registry.test.ts",
"src/main/browser/browser-session-startup.test.ts",
- "src/main/window/createMainWindow.test.ts"
+ "src/main/window/createMainWindow.test.ts",
+ "tests/e2e/local-ssh-browser-routing.spec.ts",
+ "tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts"
],
"assertionRefs": [
{
@@ -4719,6 +4731,20 @@
"a live route partition may attach only the normalized blank document",
"an arbitrary URL cannot be the initial route-partition document"
]
+ },
+ {
+ "file": "tests/e2e/local-ssh-browser-routing.spec.ts",
+ "assertions": [
+ "a remote-only origin renders through direct SSH routing; cookies survive transport recovery",
+ "unavailable SSH hosts prevent premature webview attachment and offer a working explicit local escape hatch",
+ "real AllowTcpForwarding refusal is classified and Try anyway preserves the SSH route"
+ ]
+ },
+ {
+ "file": "tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts",
+ "assertions": [
+ "paired client-hosted browser pages render an SSH-only origin, preserve cookies and retire superseded route pages across a real transport drop"
+ ]
}
],
"evidenceRuns": [
@@ -4767,15 +4793,33 @@
"result": "passed",
"durationSeconds": 0.5,
"summary": "Six files passed 150 opaque identity, durable collision binding, bounded partition/page, proxy-before-allowlist, policy reuse, profile startup, and blank-only attach tests."
+ },
+ {
+ "date": "2026-09-06",
+ "runner": "ci",
+ "platform": "linux",
+ "result": "passed",
+ "command": "ORCA_E2E_SSH_DOCKER=1 ORCA_E2E_LOCAL_SSH_BROWSER=1 ORCA_E2E_SSH_CLIENT_HOSTED_BROWSER=1 ORCA_E2E_WEB_CLIENT=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/local-ssh-browser-routing.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1 --repeat-each=3",
+ "durationSeconds": 252,
+ "summary": "Nine direct SSH cases passed: three repetitions each of routing/reconnect, unavailable-host local escape, and real TCP-forwarding refusal. Run 34040309638, head 259a5f6; unchanged tests and timeouts, zero skips or retries."
+ },
+ {
+ "date": "2026-09-06",
+ "runner": "ci",
+ "platform": "linux",
+ "result": "passed",
+ "command": "ORCA_E2E_SSH_DOCKER=1 ORCA_E2E_LOCAL_SSH_BROWSER=1 ORCA_E2E_SSH_CLIENT_HOSTED_BROWSER=1 ORCA_E2E_WEB_CLIENT=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1 --repeat-each=3",
+ "durationSeconds": 138,
+ "summary": "Three paired client-hosted reconnect cases passed with remote-only origin and cookie-preservation assertions. Run 34040309638, head 259a5f6; unchanged tests and timeouts, zero skips or retries."
}
],
"runtimeBudget": {
"p95Seconds": 2,
- "scope": "deterministic identity, binding-store, session-policy, and window-boundary tests"
+ "scope": "deterministic identity, binding-store, session-policy, and window-boundary tests; this unit-test budget excludes SSH browser journeys, whose CI p95 is not yet established"
},
"flakeHistory": {
"status": "unknown",
- "evidence": "The deterministic suite passes locally; CI and real Electron soak history have not started."
+ "evidence": "The deterministic suite passes locally. Linux SSH provider journeys passed four baseline cases and twelve repeated cases in CI runs 34039986047 and 34040309638 with zero skips or retries. Long-term and cross-platform soak history remains incomplete."
},
"redGreenEvidence": {
"status": "partial",
@@ -4801,7 +4845,8 @@
"Partition deletion, download/transfer draining, idle route release, disk quotas, and browser-profile cloning are later lifecycle stages.",
"Binding writes serialize in Electron main, and packaged hosts rely on Orca's per-userData single-instance lock. Activation still needs an explicit guard for dev instances that share userData or a cross-process CAS/lock.",
"Sequential proxy or policy setup failures retain durable bindings and can exhaust the 512-binding ledger. Activation requires bounded tombstone recovery and partition garbage collection.",
- "Each preparePage synchronously reads and parses bounded binding metadata on Electron main; activation requires latency evidence or a safely invalidated cache before this becomes frequent."
+ "Each preparePage synchronously reads and parses bounded binding metadata on Electron main; activation requires latency evidence or a safely invalidated cache before this becomes frequent.",
+ "New SSH browser journey evidence is limited to Linux CI with Docker; native macOS/Windows clients and WSL providers remain unverified by these scenarios."
],
"demotionRule": "Keep experimental or demote if raw identities enter a partition path, a durable binding mismatch is reused, a partition retargets to another execution host or live listener, a route partition becomes attachable before exact proxy verification, initial attachment can navigate beyond blank, stale cleanup retires a replacement, admission exceeds a declared cap, or any browser request reaches desktop DNS, TCP, UDP, localhost, or system proxy outside the selected route."
},
diff --git a/config/scripts/run-ssh-docker-e2e.mjs b/config/scripts/run-ssh-docker-e2e.mjs
index faa354689cb..b4d9ddcafdd 100644
--- a/config/scripts/run-ssh-docker-e2e.mjs
+++ b/config/scripts/run-ssh-docker-e2e.mjs
@@ -6,6 +6,8 @@ const pnpm = process.platform === 'win32' ? 'pnpm.cmd' : 'pnpm'
const env = {
...process.env,
ORCA_E2E_SSH_DOCKER: '1',
+ ORCA_E2E_LOCAL_SSH_BROWSER: '1',
+ ORCA_E2E_SSH_CLIENT_HOSTED_BROWSER: '1',
ORCA_E2E_WEB_CLIENT: '1'
}
@@ -46,20 +48,20 @@ if (runtime.status !== 0) {
// - E2E does not gate merges: `verify.needs` in pr.yml omits `e2e` while the suite is red on
// main. Nothing in this lane blocks a PR yet. pr.yml's Require-successful-checks comment
// has the exact wiring to flip it, and the gate contract asserts the current state.
-// - Five specs and one unit test are gated on env vars no workflow sets, so they run nowhere
+// - Three specs and one unit test are gated on env vars no workflow sets, so they run nowhere
// and are not Docker-gated, which puts them outside this file's contract:
-// local-ssh-browser-routing (ORCA_E2E_LOCAL_SSH_BROWSER)
-// ssh-client-hosted-browser-drop-reconnect (ORCA_E2E_SSH_CLIENT_HOSTED_BROWSER)
// nested-runtime-ssh-lifecycle, nested-runtime-ssh-routing (ORCA_E2E_NESTED_RUNTIME_SSH)
// ssh-localhost (ORCA_E2E_SSH_LOCALHOST)
// ssh-browser-network-execution-route.docker.unit.test.ts (ORCA_RUN_DOCKER_SSH_BROWSER_E2E)
-// Runner scripts for the first four sit unused in package.json; no workflow calls them.
+// The nested-runtime runner remains unused by CI.
const result = spawnSync(
pnpm,
[
'exec',
'playwright',
'test',
+ 'tests/e2e/local-ssh-browser-routing.spec.ts',
+ 'tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts',
'tests/e2e/pty-input-write-queue-ssh.spec.ts',
'tests/e2e/ssh-ai-vault-session-history.spec.ts',
'tests/e2e/ssh-cold-activation-restore.spec.ts',
diff --git a/config/scripts/ssh-browser-e2e-routing.test.mjs b/config/scripts/ssh-browser-e2e-routing.test.mjs
new file mode 100644
index 00000000000..56537fd95e9
--- /dev/null
+++ b/config/scripts/ssh-browser-e2e-routing.test.mjs
@@ -0,0 +1,26 @@
+import { readFileSync } from 'node:fs'
+import { join, resolve } from 'node:path'
+import { parse } from 'yaml'
+import { expect, it } from 'vitest'
+
+const root = resolve(import.meta.dirname, '../..')
+const workflow = parse(readFileSync(join(root, '.github/workflows/e2e.yml'), 'utf8'))
+const runner = readFileSync(join(root, 'config/scripts/run-ssh-docker-e2e.mjs'), 'utf8')
+
+it('routes SSH browser specs to a lane that enables their opt-ins', () => {
+ const changedRun = workflow.jobs['changed-e2e'].steps.find(
+ (step) => step.name === 'Run changed E2E specs'
+ )
+ for (const [spec, flag] of [
+ ['tests/e2e/local-ssh-browser-routing.spec.ts', 'ORCA_E2E_LOCAL_SSH_BROWSER'],
+ [
+ 'tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts',
+ 'ORCA_E2E_SSH_CLIENT_HOSTED_BROWSER'
+ ]
+ ]) {
+ expect(runner).toContain(`'${spec}'`)
+ expect(runner).toContain(`${flag}: '1'`)
+ expect(workflow.jobs['ssh-docker-watcher-isolation'].if).toContain(spec)
+ expect(changedRun.run).toContain(`. != "${spec}"`)
+ }
+})
From 9837adaa07c2cf4c9ab22c4b02ba32f8aa6a1239 Mon Sep 17 00:00:00 2001
From: Neil <4138956+nwparker@users.noreply.github.com>
Date: Sun, 6 Sep 2026 09:05:08 -0700
Subject: [PATCH 07/32] test: reconnect after replacing same-ID runtime pairing
(#19094)
---
tests/e2e/helpers/nested-runtime-same-id-pairing.ts | 4 ++++
1 file changed, 4 insertions(+)
diff --git a/tests/e2e/helpers/nested-runtime-same-id-pairing.ts b/tests/e2e/helpers/nested-runtime-same-id-pairing.ts
index 5e13630de80..2d8e7d99d6d 100644
--- a/tests/e2e/helpers/nested-runtime-same-id-pairing.ts
+++ b/tests/e2e/helpers/nested-runtime-same-id-pairing.ts
@@ -28,6 +28,10 @@ export async function replaceRuntimePairingInPlace(args: {
if (!store) {
throw new Error('Paired desktop store is unavailable during same-ID re-pair')
}
+ const connection = await window.api.runtimeEnvironments.connect({ selector })
+ if (!connection.ok) {
+ throw new Error(`Same-ID re-pair reconnect failed: ${JSON.stringify(connection.error)}`)
+ }
const environments = await window.api.runtimeEnvironments.list()
store.getState().setRuntimeEnvironments(environments)
if (!(await store.getState().refreshRuntimeEnvironmentStatus(selector))) {
From f5960cec00f5809d2f02a19ce886faf4b8982a90 Mon Sep 17 00:00:00 2001
From: Neil <4138956+nwparker@users.noreply.github.com>
Date: Sun, 6 Sep 2026 09:29:56 -0700
Subject: [PATCH 08/32] test: enable Docker SSH browser network route coverage
in CI (#19095)
* test: enable Docker SSH browser network route journeys in CI
* test: register Docker browser job in token permissions contract
* test: declare SSH client dependency and narrow browser fixture routing
---
.github/workflows/e2e.yml | 21 ++++++++
config/reliability-gates.jsonc | 49 ++++++++++++-------
config/scripts/pr-e2e-source-routing.mjs | 11 +++++
.../release-cut-token-permissions.test.mjs | 1 +
config/scripts/run-ssh-docker-e2e.mjs | 3 +-
.../scripts/ssh-browser-e2e-routing.test.mjs | 38 ++++++++++++++
6 files changed, 102 insertions(+), 21 deletions(-)
diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml
index 0b529cde7e3..e000cdab4ba 100644
--- a/.github/workflows/e2e.yml
+++ b/.github/workflows/e2e.yml
@@ -228,6 +228,7 @@ jobs:
. != "tests/e2e/ssh-startup-exec-readiness.spec.ts" and
. != "tests/e2e/paired-startup-exec-readiness.spec.ts" and
. != "tests/e2e/local-ssh-browser-routing.spec.ts" and
+ . != "tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts" and
. != "tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts" and
. != "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts" and
. != "tests/e2e/terminal-ibus-hangul-native.spec.ts"
@@ -354,3 +355,23 @@ jobs:
path: e2e-traces/
retention-days: 7
if-no-files-found: ignore
+
+ ssh-browser-network-route:
+ name: ssh browser network route
+ if: inputs.test_files == '' || contains(inputs.test_files, 'tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts')
+ runs-on: ubuntu-latest
+ timeout-minutes: 15
+ steps:
+ - uses: actions/checkout@v6
+ with:
+ ref: ${{ inputs.ref || github.ref }}
+ - uses: ./.github/actions/install-node-dependencies
+ with:
+ native-runtime: node
+ - name: Install SSH client
+ run: sudo apt-get update && sudo apt-get install -y openssh-client
+ - name: Run Docker SSH browser network route journeys
+ env:
+ ORCA_BACKGROUND_LAUNCH: '1'
+ ORCA_RUN_DOCKER_SSH_BROWSER_E2E: '1'
+ run: node_modules/.bin/vitest run --config config/vitest.config.ts tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts
diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc
index 5609318325c..b794519c9b7 100644
--- a/config/reliability-gates.jsonc
+++ b/config/reliability-gates.jsonc
@@ -3731,9 +3731,9 @@
],
"platforms": ["macos", "linux", "windows"],
"providers": ["local", "remote-runtime", "ssh", "wsl"],
- "coveredPlatforms": ["macos"],
+ "coveredPlatforms": ["macos", "linux"],
"coveredProviders": ["remote-runtime", "ssh"],
- "coverageNotes": "Deterministic protocol, registry, injected-socket, and loopback-listener tests cover strict framing, remote-DNS targets, exact destination-write and source-consumption credit, at most 16 pending opens, 128 admitted opens per 10-second monotonic window, an 8 MiB per-route application-buffer ledger, shared 32 MiB browser-host and 128 MiB process ledgers across application copies, encrypted client queues, and native WebSocket bufferedAmount, bounded byte/claim/socket-source counts, a four-frame queued-drain quantum, authority epochs, exact host selection, one host per authenticated connection, four hosts per paired device, eight global browser-host polls with four per authenticated paired device, a shared ask/host ceiling that retains one quarter for waits, bounded initial and reconnect runtime_busy recovery, long-poll metering and disconnect abort, monotonic host/page/route generations without page tombstones, two-phase exact page retirement with cancellation, connection-owned cleanup, exact client revocation, stale and replaced fences, retired stream IDs, half-close and close ordering, SOCKS CONNECT, bind/close races, listener wildcard normalization, unsupported commands, unavailable routes, and raw/terminal binary-handler isolation. Page commands use a separately echoed v1 attach negotiation, exact authority/host/page generations, bounded command IDs and sequences, and bounded create/navigate payloads; legacy attaches still receive only the unchanged ready/revoked event shapes. A second optional reconciliation subprotocol gates bounded reclaim, close, and restore payloads behind exact attach/ready echo, complete inventory, command negotiation, reconnect authority, and command-result authority; the production client advertises it only with the matching command and inventory capabilities. Exact guest or app-renderer loss marks one page generation outcome-unknown, coalesces a bounded negotiated inventory reattach, closes or retires the dead generation, and allocates a fresh generation before URL restore; explicit close is not misclassified as a crash. Mixed-version mutation tests project hidden client pages before activate, close, split, reorder, and move-to-group admission, preserve hidden raw order slots, translate visible insertion indices, and project mutation snapshots. A production server orchestrator consumes each immutable inventory once, reserves target generations without exposing placement, emits only negotiated ledger commands, commits after exact completed proof, preserves unrelated and server placements, aborts an attempt when connection authority enters reconnect grace, and requires fresh inventory after failure or abort. The production client dispatcher additionally proves per-page FIFO execution, exact payload-matched duplicate replay, frozen command/result snapshots, a global retired-generation floor, transactional admission, bounded pages/active commands/queues/per-page and global result cache/concurrency, create dependency failure, cancellation, deduplicated retirement joining, and bounded close without late-result overwrite. The server ledger owns issue order and immutable command/result snapshots, bounds outstanding commands, active pages, and per-page/global replay caches, releases active-page capacity after an exact completed close while retaining bounded result replay, validates the shared wire payload before admission, requires live delivery and exact placement, authenticates results to the negotiated connection and paired lease, rejects gaps and conflicting replay, and fences outstanding outcomes at exact retirement. Negotiated command results reuse the authenticated attach socket through bounded nested JSON requests; exact ID routing, reverse-order replies, unknown and duplicate IDs, timeout teardown, serialization failure, aggregate queue accounting, acknowledgement validation, and the unchanged non-v1 path are deterministic. A stable local listener rejects CONNECT while offline or reconnecting, retains its address across replacement, requires a strictly increasing tunnel generation, ignores late superseded callbacks, propagates tunnel protocol failure to the route owner, and recycles exhausted stream IDs only after generation replacement. Reconnect uses the unchanged native v1 attach payload and capability pair; SSH descriptors alone add an execution-host capability and require a runtime-minted grant bound to the exact browser-host lease. Exact SSH provider epoch and connection generation fence ssh2 forwardOut and one non-interactive standalone system-SSH dynamic forward per route. Unit tests preserve domain-form SOCKS requests, sanitize remote errors, bound stderr, cancel startup, release timed-out and synchronously failed sockets, and release routes once. An ephemeral Docker sshd resolves a container-only domain and returns a unique HTTP marker through both ssh2 and the actual system-OpenSSH dynamic-forward adapter without touching the user's SSH files; authority loss fences the ssh2 route. The execution runtime charges route application bytes to the same per-host/process policy; its existing E2EE owner separately caps native outbound buffers process-wide. Production-registered browser-host and paired-runtime methods lease one exact host, prove attach, command delivery, and result settlement share one exact connection identity, then carry SOCKS and HTTP bytes over a dedicated E2EE socket to the fenced execution-host revision and prove route close destroys the destination socket. The production desktop adapter now composes one exact host per environment pairing revision with the page executor, current renderer selector, route Session/WebContents registries, and one reference-counted route per canonical execution-host key. Negotiated same-client control reconnect retains exact authority, placements, grants, dispatcher dedupe, executor guests, and listener addresses; it fences tunnels immediately, blocks route admission, reattaches command delivery only after ready, and replays unsettled commands without repeating completed mutations. Terminal release, replacement, legacy disconnect, and reconnect-grace expiry make only the exact host generation's client placements non-cancellable retirement-pending while retaining capacity until exact cleanup; reconnect grace preserves them. Environment replacement and app shutdown still serialize transport closure before page cleanup and force-close every remaining route. The production placement preparation starts the exact desktop adapter and advertises host/tunnel capabilities only when the paired Electron client is eligible. Node stream-internal high-water bytes, strict cross-route scheduling, and physical cross-platform evidence remain uncovered.",
+ "coverageNotes": "Deterministic protocol, registry, injected-socket, and loopback-listener tests cover strict framing, remote-DNS targets, exact destination-write and source-consumption credit, at most 16 pending opens, 128 admitted opens per 10-second monotonic window, an 8 MiB per-route application-buffer ledger, shared 32 MiB browser-host and 128 MiB process ledgers across application copies, encrypted client queues, and native WebSocket bufferedAmount, bounded byte/claim/socket-source counts, a four-frame queued-drain quantum, authority epochs, exact host selection, one host per authenticated connection, four hosts per paired device, eight global browser-host polls with four per authenticated paired device, a shared ask/host ceiling that retains one quarter for waits, bounded initial and reconnect runtime_busy recovery, long-poll metering and disconnect abort, monotonic host/page/route generations without page tombstones, two-phase exact page retirement with cancellation, connection-owned cleanup, exact client revocation, stale and replaced fences, retired stream IDs, half-close and close ordering, SOCKS CONNECT, bind/close races, listener wildcard normalization, unsupported commands, unavailable routes, and raw/terminal binary-handler isolation. Page commands use a separately echoed v1 attach negotiation, exact authority/host/page generations, bounded command IDs and sequences, and bounded create/navigate payloads; legacy attaches still receive only the unchanged ready/revoked event shapes. A second optional reconciliation subprotocol gates bounded reclaim, close, and restore payloads behind exact attach/ready echo, complete inventory, command negotiation, reconnect authority, and command-result authority; the production client advertises it only with the matching command and inventory capabilities. Exact guest or app-renderer loss marks one page generation outcome-unknown, coalesces a bounded negotiated inventory reattach, closes or retires the dead generation, and allocates a fresh generation before URL restore; explicit close is not misclassified as a crash. Mixed-version mutation tests project hidden client pages before activate, close, split, reorder, and move-to-group admission, preserve hidden raw order slots, translate visible insertion indices, and project mutation snapshots. A production server orchestrator consumes each immutable inventory once, reserves target generations without exposing placement, emits only negotiated ledger commands, commits after exact completed proof, preserves unrelated and server placements, aborts an attempt when connection authority enters reconnect grace, and requires fresh inventory after failure or abort. The production client dispatcher additionally proves per-page FIFO execution, exact payload-matched duplicate replay, frozen command/result snapshots, a global retired-generation floor, transactional admission, bounded pages/active commands/queues/per-page and global result cache/concurrency, create dependency failure, cancellation, deduplicated retirement joining, and bounded close without late-result overwrite. The server ledger owns issue order and immutable command/result snapshots, bounds outstanding commands, active pages, and per-page/global replay caches, releases active-page capacity after an exact completed close while retaining bounded result replay, validates the shared wire payload before admission, requires live delivery and exact placement, authenticates results to the negotiated connection and paired lease, rejects gaps and conflicting replay, and fences outstanding outcomes at exact retirement. Negotiated command results reuse the authenticated attach socket through bounded nested JSON requests; exact ID routing, reverse-order replies, unknown and duplicate IDs, timeout teardown, serialization failure, aggregate queue accounting, acknowledgement validation, and the unchanged non-v1 path are deterministic. A stable local listener rejects CONNECT while offline or reconnecting, retains its address across replacement, requires a strictly increasing tunnel generation, ignores late superseded callbacks, propagates tunnel protocol failure to the route owner, and recycles exhausted stream IDs only after generation replacement. Reconnect uses the unchanged native v1 attach payload and capability pair; SSH descriptors alone add an execution-host capability and require a runtime-minted grant bound to the exact browser-host lease. Exact SSH provider epoch and connection generation fence ssh2 forwardOut and one non-interactive standalone system-SSH dynamic forward per route. Unit tests preserve domain-form SOCKS requests, sanitize remote errors, bound stderr, cancel startup, release timed-out and synchronously failed sockets, and release routes once. An ephemeral Docker sshd resolves a container-only domain and returns a unique HTTP marker through both ssh2 and the actual system-OpenSSH dynamic-forward adapter without touching the user's SSH files; authority loss fences the ssh2 route. The execution runtime charges route application bytes to the same per-host/process policy; its existing E2EE owner separately caps native outbound buffers process-wide. Production-registered browser-host and paired-runtime methods lease one exact host, prove attach, command delivery, and result settlement share one exact connection identity, then carry SOCKS and HTTP bytes over a dedicated E2EE socket to the fenced execution-host revision and prove route close destroys the destination socket. The production desktop adapter now composes one exact host per environment pairing revision with the page executor, current renderer selector, route Session/WebContents registries, and one reference-counted route per canonical execution-host key. Negotiated same-client control reconnect retains exact authority, placements, grants, dispatcher dedupe, executor guests, and listener addresses; it fences tunnels immediately, blocks route admission, reattaches command delivery only after ready, and replays unsettled commands without repeating completed mutations. Terminal release, replacement, legacy disconnect, and reconnect-grace expiry make only the exact host generation's client placements non-cancellable retirement-pending while retaining capacity until exact cleanup; reconnect grace preserves them. Environment replacement and app shutdown still serialize transport closure before page cleanup and force-close every remaining route. The production placement preparation starts the exact desktop adapter and advertises host/tunnel capabilities only when the paired Electron client is eligible. Node stream-internal high-water bytes, strict cross-route scheduling, and physical cross-platform evidence remain uncovered. The two Docker remote-only SSH browser routing journeys now run in the dedicated Linux ssh-browser-network-route CI job on full runs and their mapped source/test changes; 2 baseline and 6 repeated cases passed with no skips/retries on 2026-09-06.",
"motivatingLinks": [
"https://linear.app/stably/issue/STA-4150/refactor-remote-browser-to-client-hosted-electron-webviews"
],
@@ -3765,7 +3765,8 @@
"pnpm exec vitest run --config config/vitest.config.ts src/main/browser/browser-client-network-route-registry.test.ts src/main/browser/paired-runtime-browser-client-host-composition.test.ts src/main/browser/paired-runtime-browser-client-host-registry.test.ts src/main/browser/paired-runtime-browser-client-host-runtime.test.ts src/main/browser/browser-client-page-command-executor.test.ts src/main/browser/browser-session-startup.test.ts src/main/ipc/runtime-environments-subscription-teardown.test.ts",
"pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/browser-host-command-ledger.test.ts src/main/runtime/browser-host-command-ledger-capacity.test.ts src/main/runtime/browser-host-lease-registry.test.ts src/main/runtime/rpc/methods/browser-client-host.test.ts",
"pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/browser-host-lease-registry.test.ts src/main/browser/browser-network-deferred-socket.test.ts src/main/browser/browser-network-execution-route.test.ts src/main/browser/paired-runtime-browser-network-route.test.ts src/main/runtime/rpc/methods/browser-network-tunnel.test.ts src/main/browser/ssh-browser-network-execution-route.test.ts src/main/browser/system-ssh-socks-client-socket.test.ts src/main/ssh/system-ssh-dynamic-forward-process.test.ts src/shared/browser-client-host-protocol.test.ts src/shared/browser-network-capabilities.test.ts src/main/ssh/system-ssh-forward-process.test.ts src/main/ssh/ssh-system-fallback.test.ts src/main/ssh/ssh-port-forward.test.ts",
- "ORCA_RUN_DOCKER_SSH_BROWSER_E2E=1 pnpm exec vitest run --config config/vitest.config.ts tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts"
+ "ORCA_RUN_DOCKER_SSH_BROWSER_E2E=1 pnpm exec vitest run --config config/vitest.config.ts tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts",
+ "ORCA_RUN_DOCKER_SSH_BROWSER_E2E=1 node_modules/.bin/vitest run --config config/vitest.config.ts tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts"
],
"testFiles": [
"src/main/browser/browser-route-webcontents-registry.test.ts",
@@ -4539,8 +4540,17 @@
"platform": "macos",
"command": "pnpm exec vitest run --config config/vitest.config.ts src/shared/remote-runtime-client.test.ts",
"result": "passed",
- "durationSeconds": 3.0,
+ "durationSeconds": 3,
"summary": "Seventeen authenticated subscription tests passed, including tunnel capability binding and hard outbound-queue overflow rejection."
+ },
+ {
+ "date": "2026-09-06",
+ "runner": "ci",
+ "platform": "linux",
+ "command": "ORCA_RUN_DOCKER_SSH_BROWSER_E2E=1 node_modules/.bin/vitest run --config config/vitest.config.ts tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts",
+ "result": "passed",
+ "durationSeconds": 28.29,
+ "summary": "Previously excluded Docker SSH2 and system-OpenSSH remote-only domain journeys: 2/2 baseline cases (34043512327) plus 6/6 across three independent CI jobs (34043659504), zero skips/retries. Dedicated ssh-browser-network-route job now executes them for full E2E runs and matched source/test edits; preserves all original route and authority assertions."
}
],
"runtimeBudget": {
@@ -4596,14 +4606,8 @@
],
"platforms": ["macos", "linux", "windows"],
"providers": ["remote-runtime", "ssh", "wsl"],
- "coveredPlatforms": [
- "macos",
- "linux"
- ],
- "coveredProviders": [
- "ssh",
- "remote-runtime"
- ],
+ "coveredPlatforms": ["macos", "linux"],
+ "coveredProviders": ["ssh", "remote-runtime"],
"coverageNotes": "Deterministic main-process tests cover versioned delimiter-safe aggregate partition derivation, path-safe opaque names, durable collision metadata, oversized or corrupt metadata refusal, missing-sidecar Chromium-data refusal, bounded binding/live-page admission, immediate SOCKS5 setup with Chromium loopback bypass disabled, inherited-connection closure, exact proxy verification before allowlisting, concurrent setup coalescing, live proxy-retarget refusal, token-safe page replacement, existing browser-profile policy installation, active Orca-profile storage scoping, blank-only initial attachment, arbitrary initial-navigation denial, and fail-closed per-guest WebRTC policy through delayed or failed cleanup. The production client-page executor prepares, registers, and grants the exact route page. A real Electron A/B capture proves HTTP, HTTPS, WebSocket, redirects, subresources, downloads, and a `.test` hostname traverse SOCKS with no direct target connection. A two-launch control proves immediate setProxy routes a forced persisted-worker wake and later worker fetch. A separate capture proves the protected guest sends zero direct STUN packets. A further capture proves non-WebRTC UDP is also contained: a WebTransport session and a fetch forced onto QUIC both reach the desktop directly in the control arm and emit zero datagrams through the route partition, and the shipped disable-features list hides the Direct Sockets constructors whose mere construction kills a control-arm renderer. DNS prefetch is a tripwire over an accepted residual rather than a guard: Electron 43 inherits Chromium's PrefetchDNS, so a `` host resolves on the desktop resolver outside the tunnel, and a source census keeps any DoH host-resolver mode from widening that leak. Network-service restart and WSL provider journeys remain uncovered. Four Linux Docker SSH browser baseline scenarios passed: direct-host routing, unavailable-host local escape, forwarding refusal, and paired client-hosted reconnect. All four scenarios subsequently passed three repetitions each (12 passes, no skips or retries) in Linux CI run 34040309638, with unchanged assertions and timeouts.",
"motivatingLinks": [
"https://linear.app/stably/issue/STA-4150/refactor-remote-browser-to-client-hosted-electron-webviews",
@@ -5089,9 +5093,9 @@
],
"platforms": ["macos", "linux", "windows", "ios"],
"providers": ["remote-runtime", "ssh", "wsl"],
- "coveredPlatforms": ["macos", "ios"],
+ "coveredPlatforms": ["macos", "ios", "linux"],
"coveredProviders": ["remote-runtime", "ssh", "wsl"],
- "coverageNotes": "Fresh-build Playwright journeys run the same production store action against an isolated headed Electron server and a real headless orca serve host. They prove one immutable client placement owns one real retained guest on the viewing desktop, the server owns no duplicate guest, no screencast frame renders, browser.snapshot reaches the client guest, disabling the setting preserves that guest, and the next page uses the legacy server engine. Deterministic contracts cover omitted placement, missing capabilities, explicit server placement, exact renderer-store materialization after delayed publication, no fallback after client-create failure, folder workspaces, git worktrees, browserless hosts, native and WSL routes, exact connected SSH authority, reconnect command replay, lease replacement, imported-inventory cleanup, bounded retirement, and shared remote screencast fanout for multiple independent viewers. A published v1.4.184 package runs both skew directions: an old client omits placement against the current host, while a current client capability-downgrades against the old host; each creates one server guest, no client guest, and returns the exact snapshot marker. A current iOS Simulator client paired to that legacy packaged host visibly loads Example Domain through the preserved server-hosted surface. A Docker OpenSSH target proves container-only DNS and localhost through both ssh2 and system-SSH routes. Physical Windows/Linux Electron and physical mobile journeys remain gaps.",
+ "coverageNotes": "Fresh-build Playwright journeys run the same production store action against an isolated headed Electron server and a real headless orca serve host. They prove one immutable client placement owns one real retained guest on the viewing desktop, the server owns no duplicate guest, no screencast frame renders, browser.snapshot reaches the client guest, disabling the setting preserves that guest, and the next page uses the legacy server engine. Deterministic contracts cover omitted placement, missing capabilities, explicit server placement, exact renderer-store materialization after delayed publication, no fallback after client-create failure, folder workspaces, git worktrees, browserless hosts, native and WSL routes, exact connected SSH authority, reconnect command replay, lease replacement, imported-inventory cleanup, bounded retirement, and shared remote screencast fanout for multiple independent viewers. A published v1.4.184 package runs both skew directions: an old client omits placement against the current host, while a current client capability-downgrades against the old host; each creates one server guest, no client guest, and returns the exact snapshot marker. A current iOS Simulator client paired to that legacy packaged host visibly loads Example Domain through the preserved server-hosted surface. A Docker OpenSSH target proves container-only DNS and localhost through both ssh2 and system-SSH routes. Physical Windows/Linux Electron and physical mobile journeys remain gaps. The two Docker remote-only SSH browser routing journeys now run in the dedicated Linux ssh-browser-network-route CI job on full runs and their mapped source/test changes; 2 baseline and 6 repeated cases passed with no skips/retries on 2026-09-06.",
"motivatingLinks": [
"https://linear.app/stably/issue/STA-4150/refactor-remote-browser-to-client-hosted-electron-webviews"
],
@@ -5106,7 +5110,8 @@
"pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/browser-network-tunnel-paired-runtime.integration.test.ts src/main/browser/paired-runtime-browser-network-route.test.ts src/main/browser/browser-network-execution-route.test.ts src/main/browser/wsl-browser-network-execution-route.test.ts src/main/browser/wsl-browser-network-relay-launch.test.ts src/main/runtime/runtime-browser-network-execution-host.test.ts",
"pnpm exec vitest run --config config/vitest.config.ts src/main/browser/browser-screencast-lifecycle.test.ts src/main/browser/browser-screencast-stream.test.ts src/main/runtime/orca-runtime-browser-screencast-fanout.test.ts",
"ORCA_RUN_DOCKER_SSH_BROWSER_E2E=1 pnpm exec vitest run --config config/vitest.config.ts tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts",
- "Manual iOS 26.5 simulator: pair current mobile code to packaged Orca 1.4.184; create Browser; navigate to https://example.com; require one visible Example Domain tab on the server-hosted surface"
+ "Manual iOS 26.5 simulator: pair current mobile code to packaged Orca 1.4.184; create Browser; navigate to https://example.com; require one visible Example Domain tab on the server-hosted surface",
+ "ORCA_RUN_DOCKER_SSH_BROWSER_E2E=1 node_modules/.bin/vitest run --config config/vitest.config.ts tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts"
],
"testFiles": [
"tests/e2e/paired-client-hosted-browser.spec.ts",
@@ -5242,6 +5247,15 @@
"result": "passed",
"durationSeconds": 1.2,
"summary": "22 screencast lifecycle, stream, and shared-fanout tests passed; the suite confirms one physical CDP stream fans out independently to multiple viewers, preserves viewport ownership, and cleans up without cross-viewer eviction."
+ },
+ {
+ "date": "2026-09-06",
+ "runner": "ci",
+ "platform": "linux",
+ "command": "ORCA_RUN_DOCKER_SSH_BROWSER_E2E=1 node_modules/.bin/vitest run --config config/vitest.config.ts tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts",
+ "result": "passed",
+ "durationSeconds": 28.29,
+ "summary": "Previously excluded Docker SSH2 and system-OpenSSH remote-only domain journeys: 2/2 baseline cases (34043512327) plus 6/6 across three independent CI jobs (34043659504), zero skips/retries. Dedicated ssh-browser-network-route job now executes them for full E2E runs and matched source/test edits; preserves all original route and authority assertions."
}
],
"runtimeBudget": {
@@ -18080,10 +18094,7 @@
],
"platforms": ["macos", "linux", "windows"],
"providers": ["ssh"],
- "coveredPlatforms": [
- "macos",
- "linux"
- ],
+ "coveredPlatforms": ["macos", "linux"],
"coveredProviders": ["ssh"],
"coverageNotes": "A macOS Electron client drives a Linux Docker SSH execution host. The six-spec suite passed ten enabled cases with clean worker exit (5.2m). The formerly skipped frozen-host input case now waits for recovered authority before sending input and passed four separate executions (one initial and three repetitions). The existing flooded-shell fixme remains an explicitly reproduced application gap. The bulk-open freeze reproduction runs in Linux headed CI with SwiftShader on Xvfb: headless Linux schedules idle animation frames about 1s apart, invalidating the foreground interaction measurement. Original uninstrumented five-pane workload passed all ten repetitions with zero retries/skips in 6.6m; bulk-open lag 79.3–147.8ms and interaction 127.1–155.9ms, unchanged 2500ms/5000ms budgets. Run 34037669843, head f25eab3fd7d723509ced026633f80b193a139b76, excludes unmerged replay-input application fix #19075.",
"motivatingLinks": [
diff --git a/config/scripts/pr-e2e-source-routing.mjs b/config/scripts/pr-e2e-source-routing.mjs
index 7e95869d13f..4b1c1930892 100644
--- a/config/scripts/pr-e2e-source-routing.mjs
+++ b/config/scripts/pr-e2e-source-routing.mjs
@@ -13,6 +13,17 @@ const NATIVE_IME_HARNESS =
/^(?:config\/scripts\/(?:run-terminal-ibus-hangul-e2e|terminal-ime-engagement-receipt)\.mjs$|tests\/e2e\/terminal-ime-(?:boundary-probe|byte-reader|engagement-receipt)\.ts$|tests\/e2e\/terminal-(?:ibus-hangul|hangul-terminating-digit|macos-2set-korean)-native\.spec\.ts$)/
export const PR_E2E_SOURCE_ROUTES = [
+ {
+ id: 'browser-network.ssh-docker-route',
+ specs: ['tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts'],
+ matches: (file) =>
+ file === 'tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts' ||
+ /^tests\/e2e\/helpers\/docker-ssh-relay-(?:image|target)\.ts$/.test(file) ||
+ (isProductSource(file) &&
+ /^src\/main\/(?:browser\/(?:ssh-browser-network-execution-route|browser-network-deferred-socket|browser-network-execution-route|system-ssh-socks-client-socket)|ssh\/system-ssh-dynamic-forward-process)\.ts$/.test(
+ file
+ ))
+ },
{
id: 'terminal.windows-wsl-launch-and-paste',
specs: [
diff --git a/config/scripts/release-cut-token-permissions.test.mjs b/config/scripts/release-cut-token-permissions.test.mjs
index f2f544a8f27..85f36dea3a0 100644
--- a/config/scripts/release-cut-token-permissions.test.mjs
+++ b/config/scripts/release-cut-token-permissions.test.mjs
@@ -12,6 +12,7 @@ const EXPECTED_MATRIX = {
'.github/workflows/e2e.yml#changed-e2e': { contents: 'read' },
'.github/workflows/e2e.yml#e2e': { contents: 'read' },
'.github/workflows/e2e.yml#prepare-native-cache': { contents: 'read' },
+ '.github/workflows/e2e.yml#ssh-browser-network-route': { contents: 'read' },
'.github/workflows/e2e.yml#ssh-docker-watcher-isolation': { contents: 'read' },
'.github/workflows/homebrew-bump.yml#bump-cask': { contents: 'read' },
'.github/workflows/release-mac-build.yml#build-mac': { contents: 'write' },
diff --git a/config/scripts/run-ssh-docker-e2e.mjs b/config/scripts/run-ssh-docker-e2e.mjs
index b4d9ddcafdd..4b9d51137f1 100644
--- a/config/scripts/run-ssh-docker-e2e.mjs
+++ b/config/scripts/run-ssh-docker-e2e.mjs
@@ -48,11 +48,10 @@ if (runtime.status !== 0) {
// - E2E does not gate merges: `verify.needs` in pr.yml omits `e2e` while the suite is red on
// main. Nothing in this lane blocks a PR yet. pr.yml's Require-successful-checks comment
// has the exact wiring to flip it, and the gate contract asserts the current state.
-// - Three specs and one unit test are gated on env vars no workflow sets, so they run nowhere
+// - Three specs are gated on env vars no workflow sets, so they run nowhere
// and are not Docker-gated, which puts them outside this file's contract:
// nested-runtime-ssh-lifecycle, nested-runtime-ssh-routing (ORCA_E2E_NESTED_RUNTIME_SSH)
// ssh-localhost (ORCA_E2E_SSH_LOCALHOST)
-// ssh-browser-network-execution-route.docker.unit.test.ts (ORCA_RUN_DOCKER_SSH_BROWSER_E2E)
// The nested-runtime runner remains unused by CI.
const result = spawnSync(
pnpm,
diff --git a/config/scripts/ssh-browser-e2e-routing.test.mjs b/config/scripts/ssh-browser-e2e-routing.test.mjs
index 56537fd95e9..0b46131cf14 100644
--- a/config/scripts/ssh-browser-e2e-routing.test.mjs
+++ b/config/scripts/ssh-browser-e2e-routing.test.mjs
@@ -2,6 +2,7 @@ import { readFileSync } from 'node:fs'
import { join, resolve } from 'node:path'
import { parse } from 'yaml'
import { expect, it } from 'vitest'
+import { selectPrE2eSpecs } from './pr-e2e-source-routing.mjs'
const root = resolve(import.meta.dirname, '../..')
const workflow = parse(readFileSync(join(root, '.github/workflows/e2e.yml'), 'utf8'))
@@ -24,3 +25,40 @@ it('routes SSH browser specs to a lane that enables their opt-ins', () => {
expect(changedRun.run).toContain(`. != "${spec}"`)
}
})
+
+it('executes both Docker network routes in a Node job with their opt-in enabled', () => {
+ const spec = 'tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts'
+ const job = workflow.jobs['ssh-browser-network-route']
+ const install = job.steps.find(
+ (step) => step.uses === './.github/actions/install-node-dependencies'
+ )
+ const run = job.steps.find(
+ (step) => step.name === 'Run Docker SSH browser network route journeys'
+ )
+ expect(job['runs-on']).toBe('ubuntu-latest')
+ expect(job.if).toContain("inputs.test_files == ''")
+ expect(job.if).toContain(spec)
+ expect(install.with['native-runtime']).toBe('node')
+ expect(run.env.ORCA_RUN_DOCKER_SSH_BROWSER_E2E).toBe('1')
+ expect(run.run).toContain(`vitest run --config config/vitest.config.ts ${spec}`)
+ expect(run['continue-on-error']).toBeUndefined()
+ expect(
+ workflow.jobs['changed-e2e'].steps.find((step) => step.name === 'Run changed E2E specs').run
+ ).toContain(`. != "${spec}"`)
+ for (const changed of [
+ spec,
+ 'src/main/browser/ssh-browser-network-execution-route.ts',
+ 'src/main/browser/browser-network-deferred-socket.ts',
+ 'src/main/browser/browser-network-execution-route.ts',
+ 'src/main/browser/system-ssh-socks-client-socket.ts',
+ 'src/main/ssh/system-ssh-dynamic-forward-process.ts',
+ 'tests/e2e/helpers/docker-ssh-relay-target.ts',
+ 'tests/e2e/helpers/docker-ssh-relay-image.ts'
+ ]) {
+ expect(selectPrE2eSpecs([changed])).toContain(spec)
+ }
+ expect(selectPrE2eSpecs(['src/renderer/src/components/Unrelated.tsx'])).not.toContain(spec)
+ expect(selectPrE2eSpecs(['tests/e2e/helpers/docker-ssh-relay-terminal-tabs.ts'])).not.toContain(
+ spec
+ )
+})
From 5dab49565507c7d2fdc02e92013a0d01321a1c39 Mon Sep 17 00:00:00 2001
From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com>
Date: Sun, 6 Sep 2026 12:37:11 -0400
Subject: [PATCH 09/32] docs(relay): record Roll 2 cell roll (4916ed67
fleet-wide) and tick checklist (#19096)
All 19 general cells on 4916ed67, selector gen 148 -> 186, 0 serving-process
exits across the roll. Three waves used the no-restart mode=rollback resume
(c13 transient trust-probe 409; c26/c21 post-apply runtime-status 503 shed).
Checklist: 2.3, 4.1, 4.3 relay side deployed; status header 2026-09-06.
---
.../relay-improvement-checklist-2026-09.md | 19 ++++++-----
.../docs/relay-reconnect-2026-09-findings.md | 32 ++++++++++++++++++-
cloud/docs/relay-roll2-plan-2026-09.md | 5 +++
3 files changed, 47 insertions(+), 9 deletions(-)
diff --git a/cloud/docs/relay-improvement-checklist-2026-09.md b/cloud/docs/relay-improvement-checklist-2026-09.md
index 91f1cc742ef..8d86afd4699 100644
--- a/cloud/docs/relay-improvement-checklist-2026-09.md
+++ b/cloud/docs/relay-improvement-checklist-2026-09.md
@@ -4,18 +4,18 @@ Companion to [`relay-improvement-roadmap-2026-09.md`](./relay-improvement-roadma
match). This file answers three questions per item: what are the concrete steps, what can run in parallel,
and will a user notice.
-## Status as of 2026-09-04 22:30Z
+## Status as of 2026-09-06 16:30Z
Three buckets. "Merged" means the code is on `main` and nothing in production has changed yet. "Deployed" means users are already getting it. "Awaiting owner" means I will not touch production without a go.
**Deployed to production**
+- Roll 2 relay image `4916ed67` (stablyai/orca #18959 + #18722 + #18720 flag unset): director since 2026-09-06 01:02Z, all 19 general cells by 16:29Z. Control lease 6 h ± 30 min, accept abandonment, per-cell inventory locks, pool `statement_timeout`. Record: findings doc, "Roll 2" section.
- Auth instance cap 20 + dead-family audit fix (orca-cloud #474) as revision `orca-cloud-auth-00031-tox`.
- Dynamic NAT ports in both regions (stablyai/orca #18693). Zero drops and zero proxy dial errors since.
- Nine alert policies with log metrics: 4 auth (#475), 3 relay Cloud SQL/NAT (#18693), 1 cell process-exit (#18717), all on the relay Slack channel.
-**Merged, ships with the next relay cell image roll (Roll 1 carries `519f4914`; Roll 2 needs a fresh image build)**
-- Per-cell inventory locks, delta counters, pool `statement_timeout` (#18722). Roll 2.
-- Cells dial Cloud SQL with `--private-ip` when configured (#18720). Inert until 2.1 applies.
+**Merged, not yet live**
+- Cells dial Cloud SQL with `--private-ip` when configured (#18720). Deployed in Roll 2 with the flag unset; inert until 2.1 applies.
- Phone shows a clear "sign in on the desktop again" state when the desktop is signed out (#18698).
**Merged, ships with the next auth deploy**
@@ -158,9 +158,10 @@ independent. (2.2 deferred; if revived, do it after 2.1 so the new instance is p
- [ ] Production: announce a window; same steps; verify `orca_relay_runtime_metrics` controls recover to pre-cutover count.
- [ ] Update `production-cloud-sql-app-consumers` budget test and both alert policies' `database_id`.
-### 2.3 Relay pool statement timeout (merged stablyai/orca #18722; ships Roll 2)
+### 2.3 Relay pool statement timeout (deployed in Roll 2, 2026-09-06)
- [x] `statement_timeout` on the relay `pg.Pool` (5 s, env-configurable; schema pool untimed; `57014` retryable), below the control-renewal deadline; DDL on an untimed connection (same pattern as auth #476).
- [x] Postgres test on 55440: a held lock fails the query fast and the bounded retry takes over.
+- [x] Deployed fleet-wide in Roll 2 (`4916ed67`), 2026-09-06.
### 3.1 Refresh rotation grace window (orca-cloud #478 merged 2026-09-04; deploy pending owner go)
- [ ] Fix the deploy-script env strip for `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` (pre-existing; found by #478).
@@ -169,14 +170,16 @@ independent. (2.2 deferred; if revived, do it after 2.1 so the new instance is p
- [x] Tests: replay inside window returns same successor; outside revokes; concurrent double-present yields one successor.
- [x] Deploy via `deploy-auth-production` (candidate → smoke → promote). Deployed 2026-09-04 23:15Z as `orca-cloud-auth-00035-gos`, cap 20 kept, 0 5xx; `successor_material` column present; sealed successors being written. (candidate → smoke → promote).
-### 3.2 / 4.3 Desktop (merged stablyai/orca #18719; ships next desktop release)
+### 3.2 / 4.3 Desktop (merged stablyai/orca #18719; ships next desktop release; relay side of 4.3 deployed in Roll 2)
- [x] 3.2: on refresh timeout, re-read stored session before retrying; do not re-send a token already rotated locally.
- [x] 4.3: ±10 % jitter on control lease renewal; unit test on the distribution; wire-compatible (server accepts early renewals already).
+- [x] 4.3 relay side: control lease 55 min → 6 h ± 30 min (#18959), deployed in Roll 2, 2026-09-06.
-### 4.1 Lock contention (partial: stablyai/orca #18722 merged; ships Roll 2)
+### 4.1 Lock contention (partial: stablyai/orca #18722 deployed in Roll 2, 2026-09-06)
- [x] Replace the global `FOR UPDATE` over `relay_cells` with per-cell row locks; counters delta-only. Remaining: `assignOnce` placement lock is still global (optimistic snapshot follow-up). with per-cell row locks or `pg_advisory_xact_lock(cell)`; counters delta-only.
- [x] Postgres tests on 55440 with concurrent probes (in #18722). Staging load run still owed; `postgres_retries` per hour drops in staging load run.
-- [ ] Ships in Roll 2; then 4.4 recalibrates the retries bar from a week of data.
+- [x] Shipped in Roll 2 (2026-09-06). Director retries first 6 h on the new image: 13 vs 85 on the predecessor's prior 6 h.
+- [ ] 4.4: recalibrate the retries bar from a week of data (after 2026-09-13).
### 4.2 Region preference
- [ ] Director: honor requested region when the preferred region has headroom, else sticky. Behind the existing flag.
diff --git a/cloud/docs/relay-reconnect-2026-09-findings.md b/cloud/docs/relay-reconnect-2026-09-findings.md
index 580a4da84d8..efb23f380bd 100644
--- a/cloud/docs/relay-reconnect-2026-09-findings.md
+++ b/cloud/docs/relay-reconnect-2026-09-findings.md
@@ -999,4 +999,34 @@ Owner: "sure, feel free to drive these." Sequence chosen: Roll 1 first (highest
| Image publish | run 34002233801 → `sha256:4916ed676d8389f694a648e750f1112d9002d68c84a1e0c7af828d5af129de62`; mirrored to staging (run 34002326150). | |
| Staging cell smoke | **Dropped.** Staging C4 is pinned to the Asia launch digest by `relay-staging-c4-refresh-workflow.test.mjs` (with production c27–c29 tfvars and the C4 recovery workflow) and the only C4 image-refresh path pins its accepted predecessor to an older digest. Re-pinning all of it for a smoke widens into the Asia launch machinery; #18969 closed. Roll 2 follows the Roll 1 path: director first, c7 as the rehearsal cell. | |
| Director deploy | run 34002673626 **success** 01:02Z: serving `orca-cloud-relay-00575-leq` on `4916ed67`, `00574-wag` (same image) tagged `selector-rollback`, `00569-ret` (`519f4914`) still deployable. Baseline before: 1 director Postgres retry in the prior hour, 0 `container die`. | |
-| c7 `verify` (read-only) | run 34002885408 dispatched 01:03Z, target `4916ed67`, rollback `85bf6799`, protocol 1, gen 148. | |
+| c7 `verify` (read-only) | run 34002885408 **success** (gate success, cell_1 rollout success, release_lease success), target `4916ed67`, rollback `85bf6799`, protocol 1, gen 148. | |
+| Director go/no-go (01:02Z–07:00Z, 6 h on `00575-leq`) | **Go.** Presence confirmed (13.8k assign 200s, 410 cell + 90 director `runtime_metrics` rows/30 min). Postgres retries 13 (all `55P03` lock_timeout) vs 85 on `00570-siv` in the prior 6 h. `/v1/assign` mix 200/401/503 = 13820/5557/623 vs 14081/5256/663 before the deploy; 503s are the placement/sticky admission `Retry-After` path and cluster by source (top source 351), same shape as before. 0 `container die`, cell `sqlFailuresDelta` sum 0. The earlier all-zero read at 01:28Z was a dead gcloud credential, not a quiet fleet, and was discarded. | |
+| Monitor dry-run (Roll 2 gate 1) | run 34018071984 dispatched 07:03Z at gen 148, **green** 07:18Z at `1326d6b40c`; main had moved to `b51bbf3fc6` with identical trusted code. | |
+| c7 `canary-apply` (run 34018804481) | **Succeeded** 07:18–07:31Z, protocol 1, rollback `85bf6799`: gate, rollout, seal_canary, release_lease all success. Template `…-20260906072156…` on `4916ed67`; selector gen 148 → 150. Four `container die` at 07:29:16–25Z were the new container exiting during boot (`applyPostgresSchema`/`backfillRelayCellRegions` → `Connection terminated due to connection timeout`, exit 1, 2 s runtime each) while the `cloud-sql-proxy` sidecar warmed up; fifth start at 07:29:26 listening, readiness check passed 07:29:27. Same boot-order race as c13 in Roll 1 batch 1, no serving impact (cell was still drained). 139 controls by 07:34Z and climbing, `sqlFailuresDelta` 0, `sqlLatencyMsMax` ~40 ms. | |
+| Monitor dry-run (Roll 2 gate 2) | run 34019568779 dispatched 07:36Z at gen 150, **green** 07:51Z at `57e34c7f03` (main `6494f2a4f0`, identical trusted code). | |
+| c8 `canary-apply` (run 34020284092) | **Succeeded** 07:52–08:09Z, protocol 1, rollback `519f4914`: all jobs success. Template `…-20260906075820…` on `4916ed67`; gen 150 → 152. One boot-race `container die` at 08:05:51Z (2 s, exit 1), next start served. 101 controls by 08:10Z, `sqlFailuresDelta` 0. | |
+| Monitor dry-run (Roll 2 gate 3) | run 34021119905 dispatched 08:11Z at gen 152, **green** 08:26Z at `ffbf35e0d2`. | |
+| Batch 1 `batch-apply` c9,c10,c13,c14 (run 34021868303, canary 34020284092) | **Failed on cell 3 (c13); c9 and c10 succeeded.** c9 08:27–08:43Z → gen 154, c10 08:43–08:58Z → gen 156, both trust-proven and restored general. c13: isolate → gen 157, drain, template `…-20260906090225…` on `4916ed67`, one boot-race exit 09:09:50Z, readiness 09:09:51Z, transition verifier passed at migration-only 09:11:17Z (2 680 assignments, heartbeat fresh, image `4916ed67`), then `probe-relay-rehome-trust` got **409** from the director at 09:11:18Z (157 ms; c9/c10 got 200 in ~178 ms). Failsafe re-asserted migration-only at gen 157 (no change). c14 skipped, lease released. c13 is **serving on the new image but isolated**: 151 controls by 09:18Z, `sqlFailuresDelta` 0, no exits fleet-wide after 09:12Z. The probe script prints only the status, not the director's `error` body, and neither the director nor c13 logs the 409 reason; candidates are the director's source check (`runtime.ready`/`heartbeatFresh`/incarnation read ~1 s after the verifier passed) or c13's `host-drain` rejecting the probe (incarnation mismatch, shared-runtime-identity proof, or the probe host unexpectedly present). Monitor residual: the probe should print the error body. | |
+| Monitor dry-run (Roll 2 gate 4) + c13 recovery | Gate run 34024459585 dispatched 09:26Z at gen 157 with c13 in migration-only. On green: `mode=rollback` for c13 with rollback digest `4916ed67` (what it already runs) and target `519f4914`, protocol 1 both ways: `ROLLBACK_RESUME=true` path, no restart, verify + trust probe + restore general. As in Roll 1 (c8 recovery), the rollback mode seals no canary authority, so c14 runs as its own `canary-apply` and the next batch is c15,c16,c19,c20 behind that. | |
+| c13 recovery (run 34025225328, `mode=rollback`) | Gate 4 **green** 09:38Z. Recovery **succeeded** 09:38–09:42Z: `ROLLBACK_RESUME=true`, no restart, verifier passed at migration-only (2 679 assignments, heartbeat fresh, `4916ed67`), **trust probe passed** (`host-not-connected` ×2, idempotent, shared runtime identity rejected), activate → **gen 158**, c13 general, verifier passed again. 154 controls, `sqlFailuresDelta` 0, no exits fleet-wide since 09:12Z. The 09:11Z 409 was therefore transient: same cell, same incarnation, same image, ~30 min later the identical probe passed. Most likely the director's source check reading the runtime row within ~1 s of the verifier's pass (a `ready`/heartbeat edge), which a retry in the workflow step would absorb. Residual: retry the trust probe once on 409 and print the error body. | |
+| Monitor dry-run (Roll 2 gate 5) | run 34025450523 dispatched 09:44Z at gen 158, **green** 09:59Z at `6933fd70d7` (main `d19be485d3`, identical trusted code). | |
+| c14 `canary-apply` (run 34026157631) | **Succeeded** 09:59–10:20Z, protocol 1: trust-proven, gen 158 → 160, canary authority sealed. No boot exits, 102 controls by 10:22Z, fleet `sqlFailuresDelta` 0 over 30 min. | |
+| Monitor dry-run (Roll 2 gate 6) | run 34027238190 dispatched 10:23Z at gen 160, **green** 10:38Z at `ec64df335e` (main `adcc30be3b`, identical trusted code). | |
+| Batch 2 `batch-apply` c15,c16,c19,c20 (run 34027985784, canary 34026157631) | **All four succeeded** 10:38–11:31Z, protocol 1, four trust proofs, gen 160 → 168. Boot-race exits only: 3 at 10:50Z (c16) and 5 at 11:02Z (c19), all 2–4 s, exit 1, next start served. Controls at 11:32Z: c15 160, c16 164, c19 164, c20 87 (still refilling). Fleet `sqlFailuresDelta` 1 over 30 min. | |
+| Monitor dry-run (Roll 2 gate 7) | run 34030557166 dispatched 11:33Z at gen 168, **green** 11:48Z at `adcc30be3b`. | |
+| c22 `canary-apply` (run 34031304526) | **Succeeded** 11:48–12:02Z, protocol 1, trust-proven, gen 168 → 170, canary authority sealed. No boot exits, 134 controls by 12:03Z. One correlated 1 s lock-timeout blip at 11:35:17–27Z (c10, c13, c19, c25, c28: one `sqlFailuresDelta` each, `sqlLatencyMsMax` ≈1 000 ms) spanning old and new images, the known lock-wait shape, not roll-related. Director retries 4 in the last hour. | |
+| Monitor dry-run (Roll 2 gate 8) | run 34032011250 dispatched 12:05Z at gen 170, **green** 12:20Z at `adcc30be3b`. | |
+| Batch 3 `batch-apply` c23,c24,c25,c26 (run 34032799574, canary 34031304526) | **Failed on cell 4 (c26); c23, c24, c25 succeeded** (12:20–13:11Z, gen 170 → 176, three trust proofs). c26: isolate → gen 177, drain, template `…-20260906131159…` on `4916ed67`, one boot-race exit 13:19:17Z, readiness 13:19:19Z, transition verifier passed at migration-only 13:20:42Z (2 604 assignments, heartbeat fresh, `4916ed67`), then the very next call, `admin_post target-runtime` to `c26.relay.onorca.dev/v1/admin/runtime-status`, got **503 `unconditional drop overload`** (27-byte body) and the step failed. That string is not in the relay codebase and c26 logged nothing at 13:20:42Z (readiness at 13:19:19Z, metrics steady), so it is a front-end/LB shed on one request; curl's `--retry 3` logged no retry attempt. Failsafe re-asserted migration-only at gen 177 (no change). c26 is serving on the new image but isolated: 166 controls by 13:25Z and climbing, `sqlFailuresDelta` 0. Residual: the post-apply `admin_post` should retry on 503 (the pre-apply one already tolerates a transient 5xx by comment). | |
+| c26 recovery (run 34036875433, `mode=rollback`) | Gate 9 (run 34036059275) **green** 13:41Z at gen 177 with c26 migration-only. Recovery **succeeded** 13:42–13:46Z: `ROLLBACK_RESUME=true`, no restart, verifier + trust probe passed, activate → **gen 178**, c26 general. 176 controls, `sqlFailuresDelta` 0, no exits since 13:25Z. **All 16 US general cells are on `4916ed67`.** | |
+| Monitor dry-run (Roll 2 gate 10) | run 34037169783 dispatched 13:48Z at gen 178, **green** 14:03Z at `f952f1ac96`. | |
+| c27 `canary-apply` (run 34037973681, Asia, protocol 0) | **Succeeded** 14:03–14:19Z, gen 178 → 180, canary authority sealed (unused; Asia cells roll as single canaries). Template on `4916ed67`, no boot exits, 51 controls by 14:20Z (Asia cell, refilling), `sqlFailuresDelta` 0, `sqlLatencyMsMax` ~1 040 ms (cross-region baseline, c28 on the old image reads ~1 055 ms). Fleet `sqlFailuresDelta` 5 over 30 min: c28 ×3 (~1.17 s), c8 and c9 ×1 (1 s bar), the known lock-wait singles. | |
+| Monitor dry-run (Roll 2 gate 11) | run 34038869552 dispatched 14:21Z at gen 180, **green** 14:36Z at `f952f1ac96`. | |
+| c28 `canary-apply` (run 34039710735, Asia, protocol 0) | **Succeeded** 14:36–14:53Z, gen 180 → 182. Template on `4916ed67`, no boot exits, 37 controls by 14:55Z (refilling), `sqlFailuresDelta` 0, `sqlLatencyMsMax` ~1 045 ms. Fleet `sqlFailuresDelta` 3 over 30 min. | |
+| Monitor dry-run (Roll 2 gate 12) | run 34040698172 dispatched 14:56Z at gen 182, **green** 15:12Z at `1d2e00819f`. | |
+| c29 `canary-apply` (run 34041558414, Asia, protocol 0) | **Succeeded** 15:12–15:28Z, gen 182 → 184. No boot exits, 55 controls by 15:29Z. | |
+| Census 15:29Z | MIG templates: 18 of 19 general cells on `4916ed67`; **c21 still on `519f4914`**. When c13's recovery re-sealed the canary at c14, batch 2 took c15,c16,c19,c20 and c21 dropped out of the plan's wave (`c15 canary + c16,c19,c20,c21`). Fleet 23 cells, 2 971 controls. Roll 2 exits since 07:00Z: 20, all boot-race (<10 s), 0 serving. Director retries 5 in the last hour. c21 rolls next as a single canary. | |
+| Monitor dry-run (Roll 2 gate 13) | run 34042460176 dispatched 15:30Z at gen 184, **green** 15:45Z at `3631f886a7`. | |
+| c21 `canary-apply` (run 34043296422, protocol 1) | **Failed at the same post-apply step as c26.** Isolate → gen 185, drain, template `…-20260906155550…` on `4916ed67`, verifier passed at migration-only 16:04:46Z (2 607 assignments, heartbeat fresh, `4916ed67`), then `admin_post target-runtime` to c21 got **503 `unconditional drop overload`** again (27-byte body, ~160 ms after the verifier's own successful read). Failsafe held migration-only at gen 185. c21 serving on the new image, isolated, 111 controls by 16:07Z. Second occurrence in ~3 h on two different cells, both ~1.3 min after readiness: consistent with an edge shed on the first admin request after the LB backend flips healthy. The step needs the same transient-5xx tolerance as the pre-apply read. | |
+| Monitor dry-run (Roll 2 gate 14) + c21 recovery | Gate run 34044440616 dispatched 16:08Z at gen 185 with c21 migration-only. On green: `mode=rollback` resume for c21 (rollback digest `4916ed67`, protocol 1). | |
+| c21 recovery (run 34045296151, `mode=rollback`) | Gate 14 **green** 16:23Z. Recovery **succeeded** 16:24–16:28Z: no restart, verifier + trust probe passed, activate → **gen 186**, c21 general. 164 controls, `sqlFailuresDelta` 0. | |
+| **Roll 2 complete** 16:29Z | **All 19 general cells on `4916ed67`** (c7–c10, c13–c16, c19–c29); existing-only c1–c6, c11, c12 and migration-only c17, c18 untouched. Selector gen 148 → 186. Fleet 23 cells, 2 927 controls. Container exits 07:00–16:29Z: 20, every one a boot-race exit (<10 s, `cloud-sql-proxy` sidecar not yet listening), **0 serving-process exits**. Director on `00575-leq` (`4916ed67`) since 01:02Z: Postgres retries 0 in the last hour (13 over the first 6 h vs 85 on the predecessor), 5xx in the last hour 104 `/v1/assign` 503s (admission `Retry-After` path, at the pre-roll rate). Three waves needed the no-restart `mode=rollback` resume (c13: transient trust-probe 409; c26 and c21: post-apply `runtime-status` 503 `unconditional drop overload`), each recovered in ~4 min with no drain. 14 monitor gates, 14 green, 0 freezes. | |
diff --git a/cloud/docs/relay-roll2-plan-2026-09.md b/cloud/docs/relay-roll2-plan-2026-09.md
index f84de39163f..ab275039fe9 100644
--- a/cloud/docs/relay-roll2-plan-2026-09.md
+++ b/cloud/docs/relay-roll2-plan-2026-09.md
@@ -133,6 +133,11 @@ Record every gate and wave in the findings doc as in Roll 1.
dropped.
- **Monitor residuals** already in the checklist: `probeEndpointHealth` retry decision still uses the
flat 2 000 ms bar; operator protocol unbound for Asia; `probe-relay-rehome-trust` regex.
+- **Same-cap job residuals found in Roll 2** (three of eleven mutating runs needed the resume path):
+ the post-apply `admin_post target-runtime` read has no transient-5xx tolerance and failed twice on a
+ one-request 503 `unconditional drop overload` from the edge ~80 s after readiness (c26, c21); and
+ `probe-relay-rehome-trust` prints only the status on a 409, so the transient c13 failure left no
+ reason on record. Retry both once and print the error body.
- Update the checklist status header; tick 2.3, 4.1, 4.3 relay-side as deployed.
## Deferred, owner decision required
From b459b8f16d3edfe44c9d5dc1a79d0401bb42b4cc Mon Sep 17 00:00:00 2001
From: Neil <4138956+nwparker@users.noreply.github.com>
Date: Sun, 6 Sep 2026 09:51:31 -0700
Subject: [PATCH 10/32] test: repair nested SSH fixture after HUB restart
(#19098)
* test: restore paired nested SSH fixture after HUB restart
* test: cover failed re-pair selection and background window safety
* test: use required braces in re-pair regression fixture
* test: use current paired runtime identity after re-pairing
---
.../paired-client-runtime-environment.ts | 70 ++++++++++++++++++
...ed-client-runtime-environment.unit.test.ts | 74 +++++++++++++++++++
tests/e2e/helpers/paired-electron-client.ts | 63 ++--------------
.../e2e/nested-runtime-ssh-lifecycle.spec.ts | 2 +-
4 files changed, 150 insertions(+), 59 deletions(-)
create mode 100644 tests/e2e/helpers/paired-client-runtime-environment.unit.test.ts
diff --git a/tests/e2e/helpers/paired-client-runtime-environment.ts b/tests/e2e/helpers/paired-client-runtime-environment.ts
index 2bb64d02974..771499a3ed8 100644
--- a/tests/e2e/helpers/paired-client-runtime-environment.ts
+++ b/tests/e2e/helpers/paired-client-runtime-environment.ts
@@ -1,4 +1,6 @@
import type { Page } from '@stablyai/playwright-test'
+import type { PairedElectronClient, RuntimeDesktopPairingOffer } from './paired-electron-client'
+import { revealPairedClientWindow } from './paired-client-window-reveal'
/**
* Points a freshly launched paired desktop client at the HUB runtime and makes it the active
@@ -35,3 +37,71 @@ export async function selectPairedRuntimeEnvironment(
return environmentId
}, args)
}
+
+export async function rePairPairedElectronClient(
+ client: PairedElectronClient,
+ offer: RuntimeDesktopPairingOffer,
+ name: string
+): Promise {
+ await client.captureDirectSshAttempts()
+ const environmentId = await client.page.evaluate(
+ async ({ currentEnvironmentId, name, pairingUrl }) => {
+ const store = window.__store
+ if (!store) {
+ throw new Error('Paired desktop store is unavailable')
+ }
+ if (!(await store.getState().setActiveRuntimeEnvironmentPreference(null))) {
+ throw new Error('Paired desktop could not select local before replacing the HUB')
+ }
+ await window.api.runtimeEnvironments.remove({ selector: currentEnvironmentId })
+ const result = await window.api.runtimeEnvironments.addFromPairingCode({
+ name,
+ pairingCode: pairingUrl
+ })
+ store.getState().setRuntimeEnvironments(await window.api.runtimeEnvironments.list())
+ if (!(await store.getState().refreshRuntimeEnvironmentStatus(result.environment.id))) {
+ throw new Error('Re-paired desktop could not reach the HUB runtime')
+ }
+ if (!(await store.getState().setActiveRuntimeEnvironmentPreference(result.environment.id))) {
+ throw new Error('Re-paired desktop could not select the HUB runtime')
+ }
+ return result.environment.id
+ },
+ {
+ currentEnvironmentId: client.environmentId,
+ name,
+ pairingUrl: offer.pairingUrl
+ }
+ )
+ client.environmentId = environmentId
+ // Why: removing and re-adding the same HUB changes the environment identity; remount so no pane keeps the retired transport wrapper.
+ await client.page.reload()
+ // Xvfb needs a mapped window to resume actionability frames after reload.
+ if (
+ process.env.GITHUB_ACTIONS === 'true' &&
+ process.platform === 'linux' &&
+ process.env.DISPLAY &&
+ process.env.ORCA_BACKGROUND_LAUNCH !== '1'
+ ) {
+ await revealPairedClientWindow(client)
+ }
+ await client.page.waitForFunction(
+ () => window.__store?.getState().workspaceSessionReady === true,
+ null,
+ { timeout: 30_000, polling: 100 }
+ )
+ await client.installDirectSshAttemptProbe()
+ const reachable = await client.page.evaluate(async (nextEnvironmentId) => {
+ const store = window.__store
+ if (!store) {
+ throw new Error('Re-paired desktop store is unavailable after reload')
+ }
+ if (!(await store.getState().refreshRuntimeEnvironmentStatus(nextEnvironmentId))) {
+ return false
+ }
+ return store.getState().setActiveRuntimeEnvironmentPreference(nextEnvironmentId)
+ }, environmentId)
+ if (!reachable) {
+ throw new Error('Re-paired desktop could not reach the HUB after reload')
+ }
+}
diff --git a/tests/e2e/helpers/paired-client-runtime-environment.unit.test.ts b/tests/e2e/helpers/paired-client-runtime-environment.unit.test.ts
new file mode 100644
index 00000000000..3851a9184c1
--- /dev/null
+++ b/tests/e2e/helpers/paired-client-runtime-environment.unit.test.ts
@@ -0,0 +1,74 @@
+import { afterEach, expect, it, vi } from 'vitest'
+import { rePairPairedElectronClient } from './paired-client-runtime-environment'
+import type { PairedElectronClient } from './paired-electron-client'
+
+afterEach(() => {
+ vi.unstubAllGlobals()
+ vi.unstubAllEnvs()
+})
+
+function fixture(canSelectLocal: boolean) {
+ let selected: string | null = 'old-hub'
+ const remove = vi.fn(async () => {
+ if (selected !== null) {
+ throw new Error('Cannot remove the selected runtime')
+ }
+ })
+ const state = {
+ setActiveRuntimeEnvironmentPreference: vi.fn(async (id: string | null) => {
+ if (id === null && !canSelectLocal) {
+ return false
+ }
+ selected = id
+ return true
+ }),
+ setRuntimeEnvironments: vi.fn(),
+ refreshRuntimeEnvironmentStatus: vi.fn(async () => true)
+ }
+ vi.stubGlobal('window', {
+ __store: { getState: () => state },
+ api: {
+ runtimeEnvironments: {
+ remove,
+ addFromPairingCode: vi.fn(async () => ({ environment: { id: 'new-hub' } })),
+ list: vi.fn(async () => [{ id: 'new-hub' }])
+ }
+ }
+ })
+ const nativeEvaluate = vi.fn()
+ const reload = vi.fn(async () => undefined)
+ const client = {
+ environmentId: 'old-hub',
+ captureDirectSshAttempts: vi.fn(async () => undefined),
+ installDirectSshAttemptProbe: vi.fn(async () => undefined),
+ app: { evaluate: nativeEvaluate },
+ page: {
+ evaluate: async (callback: (args: unknown) => unknown, args: unknown) => callback(args),
+ reload,
+ waitForFunction: vi.fn(async () => undefined)
+ }
+ } as unknown as PairedElectronClient
+ return { client, remove, reload, nativeEvaluate }
+}
+
+it('keeps the old pairing when selecting local fails', async () => {
+ const { client, remove, reload } = fixture(false)
+ await expect(rePairPairedElectronClient(client, { pairingUrl: 'code' }, 'HUB')).rejects.toThrow(
+ 'could not select local'
+ )
+ expect(remove).not.toHaveBeenCalled()
+ expect(reload).not.toHaveBeenCalled()
+ expect(client.environmentId).toBe('old-hub')
+})
+
+it('replaces the active pairing without touching native windows in background mode', async () => {
+ vi.stubEnv('ORCA_BACKGROUND_LAUNCH', '1')
+ vi.stubEnv('GITHUB_ACTIONS', 'true')
+ vi.stubEnv('DISPLAY', ':99')
+ const { client, remove, reload, nativeEvaluate } = fixture(true)
+ await rePairPairedElectronClient(client, { pairingUrl: 'code' }, 'HUB')
+ expect(remove).toHaveBeenCalledWith({ selector: 'old-hub' })
+ expect(client.environmentId).toBe('new-hub')
+ expect(reload).toHaveBeenCalledOnce()
+ expect(nativeEvaluate).not.toHaveBeenCalled()
+})
diff --git a/tests/e2e/helpers/paired-electron-client.ts b/tests/e2e/helpers/paired-electron-client.ts
index d08aaebe5f9..a5947bc1228 100644
--- a/tests/e2e/helpers/paired-electron-client.ts
+++ b/tests/e2e/helpers/paired-electron-client.ts
@@ -25,6 +25,8 @@ import {
import { createPairedWebClientUrl, type PairedWebClientOptions } from './paired-web-client-url'
import { selectPairedRuntimeEnvironment } from './paired-client-runtime-environment'
+export { rePairPairedElectronClient } from './paired-client-runtime-environment'
+
export type { SameIdPairingReplacement } from './nested-runtime-same-id-pairing'
export type PairedElectronClient = {
@@ -222,13 +224,13 @@ export async function launchPairedElectronClient(
replacementOffer: RuntimeDesktopPairingOffer
): Promise =>
replaceRuntimePairingInPlace({
- environmentId,
+ environmentId: client.environmentId,
page,
pairingUrl: replacementOffer.pairingUrl,
userDataDir
})
- return {
+ const client: PairedElectronClient = {
app,
page,
environmentId,
@@ -247,6 +249,7 @@ export async function launchPairedElectronClient(
replacePairingInPlace,
userDataDir
}
+ return client
} catch (error) {
await closeElectronAppForE2E(app)
await cleanupE2EDaemons(userDataDir)
@@ -254,59 +257,3 @@ export async function launchPairedElectronClient(
throw error
}
}
-
-export async function rePairPairedElectronClient(
- client: PairedElectronClient,
- offer: RuntimeDesktopPairingOffer,
- name: string
-): Promise {
- await client.captureDirectSshAttempts()
- const environmentId = await client.page.evaluate(
- async ({ currentEnvironmentId, name, pairingUrl }) => {
- const store = window.__store
- if (!store) {
- throw new Error('Paired desktop store is unavailable')
- }
- await window.api.runtimeEnvironments.remove({ selector: currentEnvironmentId })
- const result = await window.api.runtimeEnvironments.addFromPairingCode({
- name,
- pairingCode: pairingUrl
- })
- store.getState().setRuntimeEnvironments(await window.api.runtimeEnvironments.list())
- if (!(await store.getState().refreshRuntimeEnvironmentStatus(result.environment.id))) {
- throw new Error('Re-paired desktop could not reach the HUB runtime')
- }
- if (!(await store.getState().setActiveRuntimeEnvironmentPreference(result.environment.id))) {
- throw new Error('Re-paired desktop could not select the HUB runtime')
- }
- return result.environment.id
- },
- {
- currentEnvironmentId: client.environmentId,
- name,
- pairingUrl: offer.pairingUrl
- }
- )
- client.environmentId = environmentId
- // Why: removing and re-adding the same HUB changes the environment identity; remount so no pane keeps the retired transport wrapper.
- await client.page.reload()
- await client.page.waitForFunction(
- () => window.__store?.getState().workspaceSessionReady === true,
- null,
- { timeout: 30_000 }
- )
- await client.installDirectSshAttemptProbe()
- const reachable = await client.page.evaluate(async (nextEnvironmentId) => {
- const store = window.__store
- if (!store) {
- throw new Error('Re-paired desktop store is unavailable after reload')
- }
- if (!(await store.getState().refreshRuntimeEnvironmentStatus(nextEnvironmentId))) {
- return false
- }
- return store.getState().setActiveRuntimeEnvironmentPreference(nextEnvironmentId)
- }, environmentId)
- if (!reachable) {
- throw new Error('Re-paired desktop could not reach the HUB after reload')
- }
-}
diff --git a/tests/e2e/nested-runtime-ssh-lifecycle.spec.ts b/tests/e2e/nested-runtime-ssh-lifecycle.spec.ts
index 4b4eb7f21c6..2680d4c3204 100644
--- a/tests/e2e/nested-runtime-ssh-lifecycle.spec.ts
+++ b/tests/e2e/nested-runtime-ssh-lifecycle.spec.ts
@@ -722,7 +722,7 @@ test('restores a paired nested SSH route after the HUB restarts', async ({
if (!(await store.getState().refreshRuntimeEnvironmentStatus(environmentId))) {
return false
}
- return store.getState().switchRuntimeEnvironment(environmentId)
+ return store.getState().setActiveRuntimeEnvironmentPreference(environmentId)
}, preRestartEnvironmentId)
expect(existingPairingRecovered).toBe(true)
await reconnectDisconnectedDockerSshRelayTarget(hubLaunch.page, remote.targetId)
From 4d9e963ffd3e2da5c079f0ccee3a4ac065b4eade Mon Sep 17 00:00:00 2001
From: Neil <4138956+nwparker@users.noreply.github.com>
Date: Sun, 6 Sep 2026 10:05:26 -0700
Subject: [PATCH 11/32] test: enable localhost SSH terminal and hook journey in
CI (#19097)
* test: run localhost SSH terminal and hooks in CI
* test: isolate localhost SSH session fixtures across repetitions
* test: route remote agent hook source changes to localhost journey
* test: record localhost SSH reliability evidence and remaining gaps
* test: route the real SSH session hook authority
---
.github/workflows/e2e.yml | 65 +++++++++++++
config/reliability-gates.jsonc | 93 +++++++++++++++++++
config/scripts/pr-e2e-source-routing.mjs | 9 ++
.../release-cut-token-permissions.test.mjs | 1 +
config/scripts/run-ssh-docker-e2e.mjs | 3 +-
.../ssh-localhost-e2e-routing.test.mjs | 52 +++++++++++
tests/e2e/ssh-localhost.spec.ts | 7 +-
7 files changed, 227 insertions(+), 3 deletions(-)
create mode 100644 config/scripts/ssh-localhost-e2e-routing.test.mjs
diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml
index e000cdab4ba..f75d7ba00bb 100644
--- a/.github/workflows/e2e.yml
+++ b/.github/workflows/e2e.yml
@@ -229,6 +229,7 @@ jobs:
. != "tests/e2e/paired-startup-exec-readiness.spec.ts" and
. != "tests/e2e/local-ssh-browser-routing.spec.ts" and
. != "tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts" and
+ . != "tests/e2e/ssh-localhost.spec.ts" and
. != "tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts" and
. != "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts" and
. != "tests/e2e/terminal-ibus-hangul-native.spec.ts"
@@ -375,3 +376,67 @@ jobs:
ORCA_BACKGROUND_LAUNCH: '1'
ORCA_RUN_DOCKER_SSH_BROWSER_E2E: '1'
run: node_modules/.bin/vitest run --config config/vitest.config.ts tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts
+
+ ssh-localhost:
+ name: localhost SSH terminal and hooks
+ needs: [build, prepare-native-cache]
+ if: inputs.test_files == '' || contains(inputs.test_files, 'tests/e2e/ssh-localhost.spec.ts')
+ runs-on: ubuntu-latest
+ timeout-minutes: 20
+ steps:
+ - uses: actions/checkout@v6
+ with:
+ ref: ${{ inputs.ref || github.ref }}
+ - name: Install SSH server and headless tools
+ run: sudo apt-get update && sudo apt-get install -y build-essential openssh-client openssh-server python3 ripgrep xvfb zsh openbox x11-utils
+ - uses: ./.github/actions/install-node-dependencies
+ with:
+ native-runtime: electron
+ - uses: actions/download-artifact@v8
+ with:
+ name: e2e-build-out
+ path: out/
+ - name: Start isolated localhost SSH server
+ shell: bash
+ run: |
+ # Bare shells install Pi extensions only for an existing agent home.
+ mkdir -p "$HOME/.pi/agent"
+ fixture="$RUNNER_TEMP/orca-localhost-sshd"
+ mkdir -p "$fixture"
+ ssh-keygen -q -t ed25519 -N '' -f "$fixture/host_key"
+ ssh-keygen -q -t ed25519 -N '' -f "$fixture/client_key"
+ cat > "$fixture/sshd_config" <> "$GITHUB_ENV"
+ - name: Run localhost SSH terminal and hook journey
+ env:
+ SKIP_BUILD: '1'
+ ORCA_E2E_SSH_LOCALHOST: '1'
+ ORCA_FEATURE_REMOTE_AGENT_HOOKS: '1'
+ ORCA_E2E_FORWARD_APP_LOGS: '1'
+ run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh pnpm exec playwright test --config tests/playwright.config.ts tests/e2e/ssh-localhost.spec.ts --project=electron-headless --workers=1
+ - uses: actions/upload-artifact@v7
+ if: failure()
+ with:
+ name: localhost-ssh-traces
+ path: test-results/
+ retention-days: 7
+ if-no-files-found: ignore
diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc
index b794519c9b7..21209935826 100644
--- a/config/reliability-gates.jsonc
+++ b/config/reliability-gates.jsonc
@@ -10,6 +10,99 @@
}
},
"gates": [
+ {
+ "id": "ssh.localhost-terminal-agent-hooks",
+ "title": "Localhost SSH terminal and agent hooks reach the owning pane",
+ "maturity": "experimental",
+ "protection": "partial",
+ "owner": "terminal-runtime",
+ "layer": "electron-ssh-e2e",
+ "surfaces": ["SSH terminal", "remote agent status", "remote plugin installation"],
+ "platforms": ["macos", "linux", "windows"],
+ "providers": ["ssh"],
+ "coveredPlatforms": ["linux"],
+ "coveredProviders": ["ssh"],
+ "coverageNotes": "Ubuntu CI loopback sshd shares the runner filesystem. Fresh per-test repositories isolate retained relay workspace snapshots; existing Pi home supplies the documented bare-shell plugin prerequisite.",
+ "motivatingLinks": ["https://github.com/stablyai/orca/pull/19097"],
+ "invariant": "A localhost SSH terminal executes on the SSH host and routes authenticated hook status to its owning pane without treating idle keyboard input as agent interruption.",
+ "oracle": "Require terminal output markers, exported hook identity, actual OpenCode/Pi plugin files, and matching pane/worktree/connection hook events; Ctrl-C and Escape in an idle shell must not interrupt a hook-owned agent.",
+ "commands": [
+ "gh run view 34045578306 --log",
+ "gh run view 34045975180 --log",
+ "gh run view 34046230389 --log",
+ "ORCA_E2E_SSH_LOCALHOST=1 ORCA_FEATURE_REMOTE_AGENT_HOOKS=1 pnpm exec playwright test --config tests/playwright.config.ts tests/e2e/ssh-localhost.spec.ts --project=electron-headless --workers=1",
+ "node_modules/.bin/vitest run --config config/vitest.config.ts config/scripts/ssh-localhost-e2e-routing.test.mjs"
+ ],
+ "testFiles": [
+ "tests/e2e/ssh-localhost.spec.ts",
+ "config/scripts/ssh-localhost-e2e-routing.test.mjs"
+ ],
+ "assertionRefs": [
+ {
+ "file": "tests/e2e/ssh-localhost.spec.ts",
+ "assertions": ["routes a terminal and agent-hook status over localhost SSH"]
+ },
+ {
+ "file": "config/scripts/ssh-localhost-e2e-routing.test.mjs",
+ "assertions": ["selects the localhost journey for its remote hook authorities"]
+ }
+ ],
+ "evidenceRuns": [
+ {
+ "date": "2026-09-06",
+ "runner": "ci",
+ "platform": "linux",
+ "command": "gh run view 34045578306 --log",
+ "result": "failed",
+ "summary": "Shared repository:2passed1failed, active pane PTY binding timed out amid old SSH target ownership conflicts.",
+ "durationSeconds": 150
+ },
+ {
+ "date": "2026-09-06",
+ "runner": "ci",
+ "platform": "linux",
+ "command": "gh run view 34045975180 --log",
+ "result": "passed",
+ "durationSeconds": 114,
+ "summary": "Fresh per-test repository:3passed,0skips0retries; original assertions retained."
+ },
+ {
+ "date": "2026-09-06",
+ "runner": "ci",
+ "platform": "linux",
+ "command": "gh run view 34046230389 --log",
+ "result": "passed",
+ "summary": "Normal selective workflow with isolated repository executed the localhost journey successfully; generic lane filtered it out.",
+ "durationSeconds": 36.4
+ }
+ ],
+ "runtimeBudget": {
+ "p95Seconds": 1200,
+ "scope": "CI job timeout; measured p95 not established"
+ },
+ "flakeHistory": {
+ "status": "soaking",
+ "evidence": "Single baseline passed, shared-path repetitions exposed state leakage; isolated-path3/3 and normal workflow passed. Long-term history missing."
+ },
+ "redGreenEvidence": {
+ "status": "partial",
+ "evidence": "Same original scenario failed across shared-path repetitions and passed with unique paths; no application fault-mutation proof."
+ },
+ "performanceBudget": {
+ "required": false,
+ "evidence": "Functional terminal and hook routing coverage, not a performance oracle."
+ },
+ "promotionCriteria": [
+ "Collect repeated scheduled Linux runs without unexplained failures.",
+ "Preserve all original terminal, environment, plugin-file, and hook-status assertions."
+ ],
+ "knownGaps": [
+ "Different client profiles reopening one existing remote workspace can encounter old target-qualified PTY IDs; the fixture isolation does not fix that application behavior.",
+ "No macOS/Windows, remote network failure, folder-only, packaged, or mixed-version claim.",
+ "PR E2E is not part of required verify while broader reliability remains unresolved."
+ ],
+ "demotionRule": "Keep experimental on unexplained failures; do not mask them with retries, skips, or longer timeouts."
+ },
{
"id": "terminal-output.prestarted-shell-snapshot-adoption",
"title": "Prestarted shell adoption paints covered output once",
diff --git a/config/scripts/pr-e2e-source-routing.mjs b/config/scripts/pr-e2e-source-routing.mjs
index 4b1c1930892..18c6b032788 100644
--- a/config/scripts/pr-e2e-source-routing.mjs
+++ b/config/scripts/pr-e2e-source-routing.mjs
@@ -13,6 +13,15 @@ const NATIVE_IME_HARNESS =
/^(?:config\/scripts\/(?:run-terminal-ibus-hangul-e2e|terminal-ime-engagement-receipt)\.mjs$|tests\/e2e\/terminal-ime-(?:boundary-probe|byte-reader|engagement-receipt)\.ts$|tests\/e2e\/terminal-(?:ibus-hangul|hangul-terminating-digit|macos-2set-korean)-native\.spec\.ts$)/
export const PR_E2E_SOURCE_ROUTES = [
+ {
+ id: 'ssh.localhost-agent-hooks',
+ specs: ['tests/e2e/ssh-localhost.spec.ts'],
+ matches: (file) =>
+ isProductSource(file) &&
+ /^src\/(?:relay\/(?:agent-hook|relay-agent-hook-runtime|plugin-overlay)|main\/(?:agent-hooks\/|ssh\/ssh-relay-session\.ts$)|shared\/agent-hook)/.test(
+ file
+ )
+ },
{
id: 'browser-network.ssh-docker-route',
specs: ['tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts'],
diff --git a/config/scripts/release-cut-token-permissions.test.mjs b/config/scripts/release-cut-token-permissions.test.mjs
index 85f36dea3a0..0fc1e5f8448 100644
--- a/config/scripts/release-cut-token-permissions.test.mjs
+++ b/config/scripts/release-cut-token-permissions.test.mjs
@@ -13,6 +13,7 @@ const EXPECTED_MATRIX = {
'.github/workflows/e2e.yml#e2e': { contents: 'read' },
'.github/workflows/e2e.yml#prepare-native-cache': { contents: 'read' },
'.github/workflows/e2e.yml#ssh-browser-network-route': { contents: 'read' },
+ '.github/workflows/e2e.yml#ssh-localhost': { contents: 'read' },
'.github/workflows/e2e.yml#ssh-docker-watcher-isolation': { contents: 'read' },
'.github/workflows/homebrew-bump.yml#bump-cask': { contents: 'read' },
'.github/workflows/release-mac-build.yml#build-mac': { contents: 'write' },
diff --git a/config/scripts/run-ssh-docker-e2e.mjs b/config/scripts/run-ssh-docker-e2e.mjs
index 4b9d51137f1..b88bde609bb 100644
--- a/config/scripts/run-ssh-docker-e2e.mjs
+++ b/config/scripts/run-ssh-docker-e2e.mjs
@@ -48,10 +48,9 @@ if (runtime.status !== 0) {
// - E2E does not gate merges: `verify.needs` in pr.yml omits `e2e` while the suite is red on
// main. Nothing in this lane blocks a PR yet. pr.yml's Require-successful-checks comment
// has the exact wiring to flip it, and the gate contract asserts the current state.
-// - Three specs are gated on env vars no workflow sets, so they run nowhere
+// - Two specs are gated on env vars no workflow sets, so they run nowhere
// and are not Docker-gated, which puts them outside this file's contract:
// nested-runtime-ssh-lifecycle, nested-runtime-ssh-routing (ORCA_E2E_NESTED_RUNTIME_SSH)
-// ssh-localhost (ORCA_E2E_SSH_LOCALHOST)
// The nested-runtime runner remains unused by CI.
const result = spawnSync(
pnpm,
diff --git a/config/scripts/ssh-localhost-e2e-routing.test.mjs b/config/scripts/ssh-localhost-e2e-routing.test.mjs
new file mode 100644
index 00000000000..b400e86153c
--- /dev/null
+++ b/config/scripts/ssh-localhost-e2e-routing.test.mjs
@@ -0,0 +1,52 @@
+import { existsSync, readFileSync } from 'node:fs'
+import { resolve } from 'node:path'
+import { parse } from 'yaml'
+import { expect, it } from 'vitest'
+import { selectPrE2eSpecs } from './pr-e2e-source-routing.mjs'
+
+const workflow = parse(
+ readFileSync(resolve(import.meta.dirname, '../../.github/workflows/e2e.yml'), 'utf8')
+)
+
+it('gives the localhost SSH journey its same-filesystem server and agent prerequisite', () => {
+ const spec = 'tests/e2e/ssh-localhost.spec.ts'
+ const job = workflow.jobs['ssh-localhost']
+ expect(job.if).toContain("inputs.test_files == ''")
+ expect(job.if).toContain(spec)
+ expect(job['runs-on']).toBe('ubuntu-latest')
+ expect(job.needs).toEqual(['build', 'prepare-native-cache'])
+ const setup = job.steps.find((step) => step.name === 'Start isolated localhost SSH server')
+ expect(setup.run).toContain('ListenAddress 127.0.0.1')
+ expect(setup.run).toContain('PasswordAuthentication no')
+ expect(setup.run).toContain('UsePAM yes')
+ expect(setup.run).toContain('mkdir -p "$HOME/.pi/agent"')
+ for (const key of ['ORCA_E2E_SSH_PORT', 'ORCA_E2E_SSH_USER', 'ORCA_E2E_SSH_IDENTITY_FILE']) {
+ expect(setup.run).toContain(key)
+ }
+ const run = job.steps.find((step) => step.name === 'Run localhost SSH terminal and hook journey')
+ expect(run.env.ORCA_E2E_SSH_LOCALHOST).toBe('1')
+ expect(run.env.ORCA_FEATURE_REMOTE_AGENT_HOOKS).toBe('1')
+ expect(run.run).toContain(spec)
+ expect(run.run).toContain('--project=electron-headless')
+ expect(run.run).not.toContain('--retries')
+ expect(run['continue-on-error']).toBeUndefined()
+ expect(
+ workflow.jobs['changed-e2e'].steps.find((step) => step.name === 'Run changed E2E specs').run
+ ).toContain(`. != "${spec}"`)
+})
+
+it('selects the localhost journey for its remote hook authorities', () => {
+ const spec = 'tests/e2e/ssh-localhost.spec.ts'
+ for (const file of [
+ 'src/relay/relay-agent-hook-runtime.ts',
+ 'src/relay/agent-hook-server.ts',
+ 'src/relay/plugin-overlay.ts',
+ 'src/main/agent-hooks/server.ts',
+ 'src/main/ssh/ssh-relay-session.ts',
+ 'src/shared/agent-hook-relay.ts'
+ ]) {
+ expect(existsSync(resolve(import.meta.dirname, '../..', file)), file).toBe(true)
+ expect(selectPrE2eSpecs([file])).toContain(spec)
+ }
+ expect(selectPrE2eSpecs(['src/renderer/src/components/Unrelated.tsx'])).not.toContain(spec)
+})
diff --git a/tests/e2e/ssh-localhost.spec.ts b/tests/e2e/ssh-localhost.spec.ts
index 1117fcb9409..afb3e780b95 100644
--- a/tests/e2e/ssh-localhost.spec.ts
+++ b/tests/e2e/ssh-localhost.spec.ts
@@ -1,4 +1,6 @@
import os from 'node:os'
+import { createSeededTestRepo } from './helpers/seeded-test-repo'
+import { cleanupTestRepository } from './global-teardown'
import type { Page } from '@stablyai/playwright-test'
import { test, expect } from './helpers/orca-app'
@@ -151,9 +153,12 @@ test.describe('Localhost SSH', () => {
test('routes a terminal and agent-hook status over localhost SSH', async ({
orcaPage,
- testRepoPath
+ registerPostElectronShutdownCleanup
}) => {
test.slow()
+ // The relay persists workspace sessions by path across fresh client profiles.
+ const testRepoPath = createSeededTestRepo({ publishPath: false })
+ registerPostElectronShutdownCleanup(async () => cleanupTestRepository(testRepoPath))
await waitForSessionReady(orcaPage)
await waitForActiveWorktree(orcaPage)
From da48ad2b47c46aefc1b44e27fadbdb30aa76bb62 Mon Sep 17 00:00:00 2001
From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com>
Date: Sun, 6 Sep 2026 10:24:38 -0700
Subject: [PATCH 12/32] Bump mobile Android versionCode to 16 to match the
0.0.48 release (#19101)
Co-authored-by: Merge Sim
---
mobile/app.json | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/mobile/app.json b/mobile/app.json
index 131e3899396..fc36687d74f 100644
--- a/mobile/app.json
+++ b/mobile/app.json
@@ -75,7 +75,7 @@
"allowBackup": false,
"permissions": ["RECORD_AUDIO", "MODIFY_AUDIO_SETTINGS"],
"package": "com.stably.orca.mobile",
- "versionCode": 15
+ "versionCode": 16
},
"plugins": [
"expo-router",
From b44aaf20c6454161c9839863834df898d0374a7f Mon Sep 17 00:00:00 2001
From: Neil <4138956+nwparker@users.noreply.github.com>
Date: Sun, 6 Sep 2026 10:37:04 -0700
Subject: [PATCH 13/32] test: reuse authoritative SSH connection readiness in
localhost fixture (#19102)
* test: reuse authoritative SSH connection readiness in localhost fixture
* test: retain localhost SSH setup diagnostics
---
.../helpers/docker-ssh-relay-connection.ts | 160 ++----------------
.../e2e/helpers/ssh-test-target-connection.ts | 155 +++++++++++++++++
tests/e2e/ssh-localhost.spec.ts | 91 ++--------
3 files changed, 185 insertions(+), 221 deletions(-)
create mode 100644 tests/e2e/helpers/ssh-test-target-connection.ts
diff --git a/tests/e2e/helpers/docker-ssh-relay-connection.ts b/tests/e2e/helpers/docker-ssh-relay-connection.ts
index 3e0c35f3c53..caf29d40ce5 100644
--- a/tests/e2e/helpers/docker-ssh-relay-connection.ts
+++ b/tests/e2e/helpers/docker-ssh-relay-connection.ts
@@ -1,3 +1,4 @@
+import { connectSshTestTarget } from './ssh-test-target-connection'
import { expect, type Page } from '@stablyai/playwright-test'
import {
@@ -30,153 +31,26 @@ export async function connectDockerSshRelayTarget(
target: DockerSshRelayTarget,
options: DockerSshRelayConnectionOptions = {}
): Promise {
- return page.evaluate(
- async ({ target, remotePath, relayGracePeriodSeconds, viaProxyJump, seedInitialTab }) => {
- const store = window.__store
- if (!store) {
- throw new Error('Store unavailable')
- }
- const credentialUnsub = window.api.ssh.onCredentialRequest((request) => {
- void window.api.ssh.submitCredential({ requestId: request.requestId, value: null })
- })
- try {
- const { target: createdTarget, repoReadoptions } = await window.api.ssh.addTarget({
- target: {
- label: `${viaProxyJump ? 'Docker SSH ProxyJump' : 'Docker SSH Relay'} E2E ${Date.now()}`,
- ...(viaProxyJump ? { configHost: 'orca-e2e-destination' } : {}),
- host: target.host,
- port: viaProxyJump ? 22 : target.port,
- username: 'root',
- identityFile: target.identityFile,
- identitiesOnly: true,
- ...(viaProxyJump ? { jumpHost: 'orca-e2e-jump' } : {}),
- relayGracePeriodSeconds
- }
- })
- store.getState().recordSshRepoReadoptions(repoReadoptions)
- const state = await window.api.ssh.connect({ targetId: createdTarget.id })
- if (!state || state.status !== 'connected') {
- throw new Error(`SSH target did not connect: ${JSON.stringify(state)}`)
- }
- if (
- !state.providerEpoch ||
- !Number.isSafeInteger(state.connectionGeneration) ||
- state.connectionGeneration === undefined ||
- state.connectionGeneration < 0
- ) {
- throw new Error(`SSH target returned incomplete authority: ${JSON.stringify(state)}`)
- }
- store.getState().setSshConnectionState(createdTarget.id, state)
- const labels = new Map(store.getState().sshTargetLabels)
- labels.set(createdTarget.id, createdTarget.label)
- store.getState().setSshTargetLabels(labels)
- const executionHostId = `ssh:${encodeURIComponent(createdTarget.id)}` as const
- const authority = {
- targetId: createdTarget.id,
- providerEpoch: state.providerEpoch,
- connectionGeneration: state.connectionGeneration
- }
-
- const result = await window.api.repos.addRemote({
- connectionId: createdTarget.id,
- remotePath,
- displayName: viaProxyJump ? 'Docker SSH ProxyJump E2E' : 'Docker SSH Relay E2E'
- })
- if ('error' in result) {
- throw new Error(result.error)
- }
- const hasExpectedRepoOwner = (): boolean =>
- store
- .getState()
- .repos.some(
- (repo) =>
- repo.id === result.repo.id &&
- repo.connectionId === createdTarget.id &&
- repo.executionHostId === executionHostId
- )
- const waitForRepoOwner = async (): Promise => {
- if (hasExpectedRepoOwner()) {
- return
- }
- await new Promise((resolve, reject) => {
- const timer = window.setTimeout(() => {
- unsubscribe()
- reject(new Error(`Remote repo owner did not hydrate for ${result.repo.path}`))
- }, 15_000)
- const unsubscribe = store.subscribe((next) => {
- if (
- !next.repos.some(
- (repo) =>
- repo.id === result.repo.id &&
- repo.connectionId === createdTarget.id &&
- repo.executionHostId === executionHostId
- )
- ) {
- return
- }
- window.clearTimeout(timer)
- unsubscribe()
- resolve()
- })
- })
- }
- await store.getState().fetchRepos()
- await waitForRepoOwner()
- const currentState = store.getState().sshConnectionStates.get(createdTarget.id)
- if (
- currentState?.providerEpoch !== authority.providerEpoch ||
- currentState.connectionGeneration !== authority.connectionGeneration
- ) {
- throw new Error(`SSH authority rotated before worktree hydration for ${result.repo.path}`)
- }
- const worktreeResult = await store.getState().fetchWorktrees(result.repo.id, {
- executionHostId,
- directSshAuthority: authority,
- requireAuthoritative: true
- })
- if (
- worktreeResult.status !== 'complete' ||
- worktreeResult.repoId !== result.repo.id ||
- worktreeResult.authority.kind !== 'direct-ssh' ||
- worktreeResult.authority.executionHostId !== executionHostId ||
- worktreeResult.authority.targetId !== authority.targetId ||
- worktreeResult.authority.providerEpoch !== authority.providerEpoch ||
- worktreeResult.authority.connectionGeneration !== authority.connectionGeneration
- ) {
- throw new Error(
- `Remote worktree hydration was not authoritative: ${JSON.stringify(worktreeResult)}`
- )
- }
- const worktree = (store.getState().worktreesByRepo[result.repo.id] ?? []).find(
- (candidate) => candidate.hostId === executionHostId
- )
- if (!worktree) {
- throw new Error(`No remote worktree found for ${result.repo.path}`)
- }
- store.getState().setActiveWorktree(worktree.id)
- if (seedInitialTab && (store.getState().tabsByWorktree[worktree.id] ?? []).length === 0) {
- store.getState().createTab(worktree.id)
- }
- store.getState().setActiveTabType('terminal')
- return {
- targetId: createdTarget.id,
- repoId: result.repo.id,
- worktreeId: worktree.id
- }
- } finally {
- credentialUnsub()
- }
+ const viaProxyJump = options.viaProxyJump ?? false
+ return connectSshTestTarget(
+ page,
+ {
+ label: `${viaProxyJump ? 'Docker SSH ProxyJump' : 'Docker SSH Relay'} E2E ${Date.now()}`,
+ ...(viaProxyJump ? { configHost: 'orca-e2e-destination' } : {}),
+ host: target.host,
+ port: viaProxyJump ? 22 : target.port,
+ username: 'root',
+ identityFile: target.identityFile,
+ identitiesOnly: true,
+ ...(viaProxyJump ? { jumpHost: 'orca-e2e-jump' } : {}),
+ relayGracePeriodSeconds: options.relayGracePeriodSeconds ?? 1
},
{
- target,
remotePath:
options.remotePath ??
- (options.viaProxyJump
- ? DOCKER_SSH_PROXY_JUMP_REMOTE_REPO_PATH
- : DOCKER_SSH_RELAY_REMOTE_REPO_PATH),
- viaProxyJump: options.viaProxyJump ?? false,
- seedInitialTab: options.seedInitialTab ?? true,
- relayGracePeriodSeconds: options.relayGracePeriodSeconds ?? 1
+ (viaProxyJump ? DOCKER_SSH_PROXY_JUMP_REMOTE_REPO_PATH : DOCKER_SSH_RELAY_REMOTE_REPO_PATH),
+ displayName: viaProxyJump ? 'Docker SSH ProxyJump E2E' : 'Docker SSH Relay E2E',
+ seedInitialTab: options.seedInitialTab
}
)
}
diff --git a/tests/e2e/helpers/ssh-test-target-connection.ts b/tests/e2e/helpers/ssh-test-target-connection.ts
new file mode 100644
index 00000000000..2108b2de96b
--- /dev/null
+++ b/tests/e2e/helpers/ssh-test-target-connection.ts
@@ -0,0 +1,155 @@
+import type { Page } from '@stablyai/playwright-test'
+import type { SshTargetCreateInput } from '../../../src/shared/ssh-types'
+
+export type ConnectedSshTestTarget = {
+ targetId: string
+ repoId: string
+ worktreeId: string
+}
+
+type SshTestConnectionOptions = {
+ remotePath: string
+ displayName: string
+ seedInitialTab?: boolean
+}
+
+export async function connectSshTestTarget(
+ page: Page,
+ target: SshTargetCreateInput,
+ options: SshTestConnectionOptions
+): Promise {
+ return page.evaluate(
+ async ({ target, remotePath, displayName, seedInitialTab }) => {
+ const store = window.__store
+ if (!store) {
+ throw new Error('Store unavailable')
+ }
+ const credentialUnsub = window.api.ssh.onCredentialRequest((request) => {
+ void window.api.ssh.submitCredential({ requestId: request.requestId, value: null })
+ })
+ try {
+ const { target: createdTarget, repoReadoptions } = await window.api.ssh.addTarget({
+ target
+ })
+ store.getState().recordSshRepoReadoptions(repoReadoptions)
+ const state = await window.api.ssh.connect({ targetId: createdTarget.id })
+ if (!state || state.status !== 'connected') {
+ throw new Error(`SSH target did not connect: ${JSON.stringify(state)}`)
+ }
+ if (
+ !state.providerEpoch ||
+ !Number.isSafeInteger(state.connectionGeneration) ||
+ state.connectionGeneration === undefined ||
+ state.connectionGeneration < 0
+ ) {
+ throw new Error(`SSH target returned incomplete authority: ${JSON.stringify(state)}`)
+ }
+ store.getState().setSshConnectionState(createdTarget.id, state)
+ const labels = new Map(store.getState().sshTargetLabels)
+ labels.set(createdTarget.id, createdTarget.label)
+ store.getState().setSshTargetLabels(labels)
+ const executionHostId = `ssh:${encodeURIComponent(createdTarget.id)}` as const
+ const authority = {
+ targetId: createdTarget.id,
+ providerEpoch: state.providerEpoch,
+ connectionGeneration: state.connectionGeneration
+ }
+
+ const result = await window.api.repos.addRemote({
+ connectionId: createdTarget.id,
+ remotePath,
+ displayName
+ })
+ if ('error' in result) {
+ throw new Error(result.error)
+ }
+ const hasExpectedRepoOwner = (): boolean =>
+ store
+ .getState()
+ .repos.some(
+ (repo) =>
+ repo.id === result.repo.id &&
+ repo.connectionId === createdTarget.id &&
+ repo.executionHostId === executionHostId
+ )
+ const waitForRepoOwner = async (): Promise => {
+ if (hasExpectedRepoOwner()) {
+ return
+ }
+ await new Promise((resolve, reject) => {
+ const timer = window.setTimeout(() => {
+ unsubscribe()
+ reject(new Error(`Remote repo owner did not hydrate for ${result.repo.path}`))
+ }, 15_000)
+ const unsubscribe = store.subscribe((next) => {
+ if (
+ !next.repos.some(
+ (repo) =>
+ repo.id === result.repo.id &&
+ repo.connectionId === createdTarget.id &&
+ repo.executionHostId === executionHostId
+ )
+ ) {
+ return
+ }
+ window.clearTimeout(timer)
+ unsubscribe()
+ resolve()
+ })
+ })
+ }
+ await store.getState().fetchRepos()
+ await waitForRepoOwner()
+ const currentState = store.getState().sshConnectionStates.get(createdTarget.id)
+ if (
+ currentState?.providerEpoch !== authority.providerEpoch ||
+ currentState.connectionGeneration !== authority.connectionGeneration
+ ) {
+ throw new Error(`SSH authority rotated before worktree hydration for ${result.repo.path}`)
+ }
+ const worktreeResult = await store.getState().fetchWorktrees(result.repo.id, {
+ executionHostId,
+ directSshAuthority: authority,
+ requireAuthoritative: true
+ })
+ if (
+ worktreeResult.status !== 'complete' ||
+ worktreeResult.repoId !== result.repo.id ||
+ worktreeResult.authority.kind !== 'direct-ssh' ||
+ worktreeResult.authority.executionHostId !== executionHostId ||
+ worktreeResult.authority.targetId !== authority.targetId ||
+ worktreeResult.authority.providerEpoch !== authority.providerEpoch ||
+ worktreeResult.authority.connectionGeneration !== authority.connectionGeneration
+ ) {
+ throw new Error(
+ `Remote worktree hydration was not authoritative: ${JSON.stringify(worktreeResult)}`
+ )
+ }
+ const worktree = (store.getState().worktreesByRepo[result.repo.id] ?? []).find(
+ (candidate) => candidate.hostId === executionHostId
+ )
+ if (!worktree) {
+ throw new Error(`No remote worktree found for ${result.repo.path}`)
+ }
+ store.getState().setActiveWorktree(worktree.id)
+ if (seedInitialTab && (store.getState().tabsByWorktree[worktree.id] ?? []).length === 0) {
+ store.getState().createTab(worktree.id)
+ }
+ store.getState().setActiveTabType('terminal')
+ return {
+ targetId: createdTarget.id,
+ repoId: result.repo.id,
+ worktreeId: worktree.id
+ }
+ } finally {
+ credentialUnsub()
+ }
+ },
+ {
+ target,
+ remotePath: options.remotePath,
+ displayName: options.displayName,
+ seedInitialTab: options.seedInitialTab ?? true
+ }
+ )
+}
diff --git a/tests/e2e/ssh-localhost.spec.ts b/tests/e2e/ssh-localhost.spec.ts
index afb3e780b95..00d2d461967 100644
--- a/tests/e2e/ssh-localhost.spec.ts
+++ b/tests/e2e/ssh-localhost.spec.ts
@@ -1,3 +1,4 @@
+import { connectSshTestTarget } from './helpers/ssh-test-target-connection'
import os from 'node:os'
import { createSeededTestRepo } from './helpers/seeded-test-repo'
import { cleanupTestRepository } from './global-teardown'
@@ -163,84 +164,18 @@ test.describe('Localhost SSH', () => {
await waitForActiveWorktree(orcaPage)
const target = readLocalhostSshTarget()
- const remote = await orcaPage.evaluate(
- async ({ remotePath, target }) => {
- const store = window.__store
- if (!store) {
- throw new Error('Store unavailable')
- }
-
- const credentialUnsub = window.api.ssh.onCredentialRequest((request) => {
- void window.api.ssh.submitCredential({ requestId: request.requestId, value: null })
- })
-
- try {
- const { target: createdTarget, repoReadoptions } = await window.api.ssh.addTarget({
- target: {
- ...target,
- // Why: local-only E2E should not leave a long-lived relay process
- // behind if the Electron app is killed between cleanup hooks.
- relayGracePeriodSeconds: 1
- }
- })
- store.getState().recordSshRepoReadoptions(repoReadoptions)
-
- let state
- try {
- state = await window.api.ssh.connect({ targetId: createdTarget.id })
- } catch (err) {
- const message = err instanceof Error ? err.message : String(err)
- throw new Error(
- `Failed to connect to localhost SSH target ${target.username}@${target.host || target.configHost}:${target.port}. ` +
- `Ensure sshd is running and key/agent auth is non-interactive. ${message}`
- )
- }
-
- if (!state || state.status !== 'connected') {
- throw new Error(`SSH target did not reach connected state: ${JSON.stringify(state)}`)
- }
-
- store.getState().setSshConnectionState(createdTarget.id, state)
- const labels = new Map(store.getState().sshTargetLabels)
- labels.set(createdTarget.id, createdTarget.label)
- store.getState().setSshTargetLabels(labels)
-
- const result = await window.api.repos.addRemote({
- connectionId: createdTarget.id,
- remotePath,
- displayName: 'Localhost SSH E2E'
- })
- if ('error' in result) {
- throw new Error(result.error)
- }
-
- await store.getState().fetchRepos()
- await store.getState().fetchWorktrees(result.repo.id)
-
- const worktrees = store.getState().worktreesByRepo[result.repo.id] ?? []
- const worktree =
- worktrees.find((candidate) => candidate.path === result.repo.path) ?? worktrees[0]
- if (!worktree) {
- throw new Error(`No remote worktree found for ${result.repo.path}`)
- }
-
- store.getState().setActiveWorktree(worktree.id)
- if ((store.getState().tabsByWorktree[worktree.id] ?? []).length === 0) {
- store.getState().createTab(worktree.id)
- }
- store.getState().setActiveTabType('terminal')
-
- return {
- targetId: createdTarget.id,
- repoId: result.repo.id,
- worktreeId: worktree.id
- }
- } finally {
- credentialUnsub()
- }
- },
- { remotePath: testRepoPath, target }
- )
+ const remote = await connectSshTestTarget(
+ orcaPage,
+ // Limit orphan relay lifetime if the test app exits before cleanup.
+ { ...target, relayGracePeriodSeconds: 1 },
+ { remotePath: testRepoPath, displayName: 'Localhost SSH E2E' }
+ ).catch((error: unknown) => {
+ throw new Error(
+ `Failed to prepare localhost SSH target ${target.username}@${target.host || target.configHost}:${target.port}. ` +
+ `Ensure sshd is running and key/agent auth is non-interactive. ${String(error)}`,
+ { cause: error }
+ )
+ })
await expect(remote.targetId).toBeTruthy()
await ensureTerminalVisible(orcaPage, 30_000)
From 4cccadcb95adf85db967f4b15687a2594e0def30 Mon Sep 17 00:00:00 2001
From: Neil <4138956+nwparker@users.noreply.github.com>
Date: Sun, 6 Sep 2026 11:23:24 -0700
Subject: [PATCH 14/32] test: keep Activity pane selection in the retained
sidebar (#19107)
---
tests/e2e/activity-agent-pane-isolation.spec.ts | 5 +++--
1 file changed, 3 insertions(+), 2 deletions(-)
diff --git a/tests/e2e/activity-agent-pane-isolation.spec.ts b/tests/e2e/activity-agent-pane-isolation.spec.ts
index 7f9e734b2d0..1a57e897765 100644
--- a/tests/e2e/activity-agent-pane-isolation.spec.ts
+++ b/tests/e2e/activity-agent-pane-isolation.spec.ts
@@ -251,8 +251,9 @@ test.describe('Activity Agent Pane Isolation', () => {
activeLeafId: first.leafId
})
- // Revealing a workspace returns the sidebar to its workspace list.
- await agentsSidebarButton(orcaPage).click()
+ await expect(
+ orcaPage.getByRole('button', { name: 'Turn off activity view', exact: true })
+ ).toHaveAttribute('aria-pressed', 'true')
await orcaPage.getByRole('button').filter({ hasText: second.prompt }).first().click()
await expect
.poll(async () => readActivePaneSelection(orcaPage), {
From 6aa0aaee6bd6d15ef04c5dbba50ef758e1d606e1 Mon Sep 17 00:00:00 2001
From: Neil <4138956+nwparker@users.noreply.github.com>
Date: Sun, 6 Sep 2026 11:25:52 -0700
Subject: [PATCH 15/32] test: isolate source-control generation repositories
per scenario (#19105)
* test: isolate source control generation repositories per scenario
* test: explain scenario repository fixture scope
---
tests/e2e/helpers/orca-app.ts | 9 +++++++--
.../helpers/source-control-generation-app.ts | 19 ++++++-------------
...ce-control-create-pr-intent-switch.spec.ts | 2 +-
3 files changed, 14 insertions(+), 16 deletions(-)
diff --git a/tests/e2e/helpers/orca-app.ts b/tests/e2e/helpers/orca-app.ts
index 5904a8416ff..af0915a6621 100644
--- a/tests/e2e/helpers/orca-app.ts
+++ b/tests/e2e/helpers/orca-app.ts
@@ -48,6 +48,7 @@ type OrcaTestFixtures = {
// Why: most E2E specs need a ready project before assertions start. Golden
// first-run specs opt out so they can prove the zero-project onboarding path.
seedTestRepo: boolean
+ seededRepoPath: string
// Synthetic-list specs need only the primary checkout; switching specs keep the two-row default.
minimumSeededWorktreeCount: number
// Why: spec-scoped launch env. Mutating process.env at spec module scope
@@ -278,6 +279,10 @@ export const test = base.extend({
// Default: dismiss the onboarding overlay so it doesn't intercept clicks.
dismissOnboarding: [true, { option: true }],
seedTestRepo: [true, { option: true }],
+ // Test-scoped so generation scenarios can isolate Git indexes and remotes.
+ seededRepoPath: async ({ testRepoPath }, provideFixture) => {
+ await provideFixture(testRepoPath)
+ },
minimumSeededWorktreeCount: [2, { option: true }],
launchEnv: [{}, { option: true }],
orcaAppExtraEnv: [{}, { option: true }],
@@ -286,7 +291,7 @@ export const test = base.extend({
// Test-scoped: grab the first BrowserWindow, add the test repo, and wait
// until the session is fully ready with a worktree active.
sharedPage: async (
- { electronApp, minimumSeededWorktreeCount, seedTestRepo, testRepoPath },
+ { electronApp, minimumSeededWorktreeCount, seedTestRepo, seededRepoPath },
provideFixture
) => {
// Why: the Electron app may take a while to create the first window,
@@ -308,7 +313,7 @@ export const test = base.extend({
return
}
- const repoPath = isValidGitRepo(testRepoPath) ? testRepoPath : createSeededTestRepo()
+ const repoPath = isValidGitRepo(seededRepoPath) ? seededRepoPath : createSeededTestRepo()
// Add the test repo via the IPC bridge
// Why: calling window.api.repos.add() goes through the same code path as
diff --git a/tests/e2e/helpers/source-control-generation-app.ts b/tests/e2e/helpers/source-control-generation-app.ts
index 2a1d00ca390..220647397d0 100644
--- a/tests/e2e/helpers/source-control-generation-app.ts
+++ b/tests/e2e/helpers/source-control-generation-app.ts
@@ -5,17 +5,10 @@ import { cleanupTestRepository } from '../global-teardown'
export { expect }
export const test = base.extend({
- testRepoPath: [
- // oxlint-disable-next-line no-empty-pattern -- Playwright requires destructured fixture arguments.
- async ({}, provideFixture) => {
- // Generation must not fetch external remotes installed by unrelated specs.
- const repoPath = createSeededTestRepo({ publishPath: false })
- try {
- await provideFixture(repoPath)
- } finally {
- cleanupTestRepository(repoPath)
- }
- },
- { scope: 'worker' }
- ]
+ seededRepoPath: async ({ registerPostElectronShutdownCleanup }, provideFixture) => {
+ // Git indexes and remotes must not survive between generation scenarios.
+ const repoPath = createSeededTestRepo({ publishPath: false })
+ registerPostElectronShutdownCleanup(async () => cleanupTestRepository(repoPath))
+ await provideFixture(repoPath)
+ }
})
diff --git a/tests/e2e/source-control-create-pr-intent-switch.spec.ts b/tests/e2e/source-control-create-pr-intent-switch.spec.ts
index 816cf396d3c..2b6ad8bb9c4 100644
--- a/tests/e2e/source-control-create-pr-intent-switch.spec.ts
+++ b/tests/e2e/source-control-create-pr-intent-switch.spec.ts
@@ -3,7 +3,7 @@ import { execFileSync } from 'node:child_process'
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'
import os from 'node:os'
import path from 'node:path'
-import { test, expect } from './helpers/orca-app'
+import { test, expect } from './helpers/source-control-generation-app'
import { waitForActiveWorktree, waitForSessionReady } from './helpers/store'
import {
createStagedCommitMessageChange,
From 3be526c5e68f6666e88983df63b16d07bb1d0817 Mon Sep 17 00:00:00 2001
From: Neil <4138956+nwparker@users.noreply.github.com>
Date: Sun, 6 Sep 2026 11:25:58 -0700
Subject: [PATCH 16/32] test: cover SSH reattach replay and enable
deterministic Codex CI (#19106)
* test: cover SSH replay replies and run deterministic Codex restore scenarios
* test: register replay probe unit command in reliability gate
---
config/reliability-gates.jsonc | 28 +-
config/scripts/pr-e2e-gate-contract.test.mjs | 9 +-
config/scripts/pr-e2e-source-routing.mjs | 1 +
config/scripts/run-ssh-docker-e2e.mjs | 3 +-
.../ssh-codex-display-artifacts-repro.spec.ts | 271 +++++++++---------
.../e2e/ssh-codex-reconnect-replay-driver.ts | 31 +-
tests/e2e/ssh-codex-replay-reply-probe.ts | 55 ++++
.../ssh-codex-replay-reply-probe.unit.test.ts | 52 ++++
8 files changed, 295 insertions(+), 155 deletions(-)
create mode 100644 tests/e2e/ssh-codex-replay-reply-probe.ts
create mode 100644 tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts
diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc
index 21209935826..70b1092fedf 100644
--- a/config/reliability-gates.jsonc
+++ b/config/reliability-gates.jsonc
@@ -18189,14 +18189,15 @@
"providers": ["ssh"],
"coveredPlatforms": ["macos", "linux"],
"coveredProviders": ["ssh"],
- "coverageNotes": "A macOS Electron client drives a Linux Docker SSH execution host. The six-spec suite passed ten enabled cases with clean worker exit (5.2m). The formerly skipped frozen-host input case now waits for recovered authority before sending input and passed four separate executions (one initial and three repetitions). The existing flooded-shell fixme remains an explicitly reproduced application gap. The bulk-open freeze reproduction runs in Linux headed CI with SwiftShader on Xvfb: headless Linux schedules idle animation frames about 1s apart, invalidating the foreground interaction measurement. Original uninstrumented five-pane workload passed all ten repetitions with zero retries/skips in 6.6m; bulk-open lag 79.3–147.8ms and interaction 127.1–155.9ms, unchanged 2500ms/5000ms budgets. Run 34037669843, head f25eab3fd7d723509ced026633f80b193a139b76, excludes unmerged replay-input application fix #19075.",
+ "coverageNotes": "A macOS Electron client drives a Linux Docker SSH execution host. The six-spec suite passed ten enabled cases with clean worker exit (5.2m). The formerly skipped frozen-host input case now waits for recovered authority before sending input and passed four separate executions (one initial and three repetitions). The existing flooded-shell fixme remains an explicitly reproduced application gap. The bulk-open freeze reproduction runs in Linux headed CI with SwiftShader on Xvfb: headless Linux schedules idle animation frames about 1s apart, invalidating the foreground interaction measurement. Original uninstrumented five-pane workload passed all ten repetitions with zero retries/skips in 6.6m; bulk-open lag 79.3–147.8ms and interaction 127.1–155.9ms, unchanged 2500ms/5000ms budgets. Run 34037669843, head f25eab3fd7d723509ced026633f80b193a139b76, excludes unmerged replay-input application fix #19075. Deterministic remote Codex fixture validation passed three normal restores and three forced reconnects with zero retries on merged main plus the replay probe correction (run 34050117471). The original forced-reconnect probe missed nonempty replay returned in pty:spawn reattach replies. Routine coverage now includes both modes by default; real Codex service execution remains opt-in.",
"motivatingLinks": [
"https://github.com/stablyai/orca/issues/18018",
"https://github.com/stablyai/orca/pull/18546",
"https://github.com/stablyai/orca/issues/12547",
"https://github.com/stablyai/orca/issues/16764",
"https://github.com/stablyai/orca/actions/runs/34037450427",
- "https://github.com/stablyai/orca/actions/runs/34037669843"
+ "https://github.com/stablyai/orca/actions/runs/34037669843",
+ "https://github.com/stablyai/orca/actions/runs/34050117471"
],
"invariant": "Transport loss and frozen-host silence must preserve the remote session; host relay loss may rebind a pane without accumulating reattachable leases. Reconnects must preserve usable terminal content, bounded PTYs/fds/processes, complete large listings, and independently recoverable watcher processes. Electron test shutdown must release inherited pipes after confirmed root exit without closing live-process pipes.",
"oracle": "Poll a changed connected SSH authority after injected faults, then require terminal output and appropriate PTY identity. Read remote process/fd state, listFiles replies, and rendered explorer rows. Resolve Playwright cleanup only after the root process exits and its inherited pipes close; live-process pipes remain untouched.",
@@ -18204,7 +18205,9 @@
"ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-transport-drop-recovery.spec.ts tests/e2e/ssh-docker-half-open-link.spec.ts tests/e2e/ssh-docker-quick-open-large-listing.spec.ts tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts tests/e2e/ssh-docker-resource-accumulation.spec.ts tests/e2e/ssh-docker-watcher-isolation.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1",
"pnpm exec vitest run --config config/vitest.config.ts tests/e2e/helpers/electron-process-shutdown.unit.test.ts",
"ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts --config tests/playwright.config.ts --project=electron-headful --workers=1",
- "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts --config tests/playwright.config.ts --project=electron-headful --workers=1 --repeat-each=10"
+ "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts --config tests/playwright.config.ts --project=electron-headful --workers=1 --repeat-each=10",
+ "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-codex-display-artifacts-repro.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1",
+ "pnpm exec vitest run --config config/vitest.config.ts tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts"
],
"testFiles": [
"tests/e2e/ssh-docker-transport-drop-recovery.spec.ts",
@@ -18214,7 +18217,9 @@
"tests/e2e/ssh-docker-resource-accumulation.spec.ts",
"tests/e2e/ssh-docker-watcher-isolation.spec.ts",
"tests/e2e/helpers/electron-process-shutdown.unit.test.ts",
- "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts"
+ "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts",
+ "tests/e2e/ssh-codex-display-artifacts-repro.spec.ts",
+ "tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts"
],
"assertionRefs": [
{
@@ -18265,6 +18270,18 @@
"assertions": [
"five flooding SSH panes remain below unchanged 2500ms soft and 5000ms hard freeze budgets during bulk reopen and two double-animation-frame view changes"
]
+ },
+ {
+ "file": "tests/e2e/ssh-codex-display-artifacts-repro.spec.ts",
+ "assertions": [
+ "normal restore and forced SSH reconnect leave no stale or duplicate status rows; forced reconnect preserves the original PTY and requires nonempty replay from that PTY through an event or reattach reply"
+ ]
+ },
+ {
+ "file": "tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts",
+ "assertions": [
+ "unrelated, replacement, initial-spawn, empty and non-replay replies do not count; original reattach results and failures pass through unchanged"
+ ]
}
],
"evidenceRuns": [
@@ -18322,7 +18339,8 @@
"Linux headed CI covers the bulk-open freeze reproduction; Windows clients, WSL, folder workspaces, paired runtimes and live agent CLIs are not covered by that result.",
"Some legacy assertions inspect terminal serialization or backing state rather than rendered DOM; no blanket visual coverage claim.",
"No p95 CI history or full product mutation proof.",
- "One headless bulk-open probe reached 6478.6ms in run 34035957303; animation-frame scheduling explains the consistent interaction failures, but does not directly explain that isolated timer-lag outlier. Long-term headed CI soak remains outstanding."
+ "One headless bulk-open probe reached 6478.6ms in run 34035957303; animation-frame scheduling explains the consistent interaction failures, but does not directly explain that isolated timer-lag outlier. Long-term headed CI soak remains outstanding.",
+ "Codex replay artifact evidence uses a deterministic remote TUI on Linux CI; real-service, macOS/Windows clients and cross-version replay remain separate coverage gaps."
],
"demotionRule": "Keep experimental while any recovery reproduction fails or any teardown, identity, resource-count, or rendered oracle flakes; never promote by extending sleeps or retries."
},
diff --git a/config/scripts/pr-e2e-gate-contract.test.mjs b/config/scripts/pr-e2e-gate-contract.test.mjs
index f5295faf1ad..1c926b3622a 100644
--- a/config/scripts/pr-e2e-gate-contract.test.mjs
+++ b/config/scripts/pr-e2e-gate-contract.test.mjs
@@ -376,13 +376,10 @@ describe('PR E2E gate contract', () => {
// that no runner names runs nowhere and still reports green — the silent skip this file
// exists to prevent. Asserting reachability rather than a literal keeps that true when
// the lanes move.
- // Why these two are exempt: each needs something CI cannot give it, recorded in
+ // The remaining exemption needs performance validation before routine CI, recorded in
// run-ssh-docker-e2e.mjs so the gap stays legible rather than looking like coverage.
- const unreachableSpecs = new Set([
- 'tests/e2e/ssh-docker-relay-perf.spec.ts',
- 'tests/e2e/ssh-codex-display-artifacts-repro.spec.ts'
- ])
- // Why comments are stripped: this file's own runner lists the two exempt specs by name in a
+ const unreachableSpecs = new Set(['tests/e2e/ssh-docker-relay-perf.spec.ts'])
+ // Why comments are stripped: the runner documents the exempt spec by name in a
// prose comment. A substring scan over raw text would count any spec merely *discussed* in a
// runner as claimed by it -- the silent skip this assertion exists to catch, re-entering
// through the documentation.
diff --git a/config/scripts/pr-e2e-source-routing.mjs b/config/scripts/pr-e2e-source-routing.mjs
index 18c6b032788..cc326e6caed 100644
--- a/config/scripts/pr-e2e-source-routing.mjs
+++ b/config/scripts/pr-e2e-source-routing.mjs
@@ -57,6 +57,7 @@ export const PR_E2E_SOURCE_ROUTES = [
id: 'ssh-terminal-source',
specs: [
'tests/e2e/pty-input-write-queue-ssh.spec.ts',
+ 'tests/e2e/ssh-codex-display-artifacts-repro.spec.ts',
'tests/e2e/ssh-cold-activation-restore.spec.ts',
'tests/e2e/ssh-docker-half-open-link.spec.ts',
'tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts',
diff --git a/config/scripts/run-ssh-docker-e2e.mjs b/config/scripts/run-ssh-docker-e2e.mjs
index b88bde609bb..435ac2e3b45 100644
--- a/config/scripts/run-ssh-docker-e2e.mjs
+++ b/config/scripts/run-ssh-docker-e2e.mjs
@@ -33,8 +33,6 @@ if (runtime.status !== 0) {
// cost the lane its credibility. NOTE: a runner script test:e2e:ssh-docker-perf exists in
// package.json but NO workflow invokes it, so this spec currently runs in no CI lane at
// all. Recorded as a real gap, not as coverage living somewhere else.
-// ssh-codex-display-artifacts-repro.spec.ts — installs a real remote codex binary that CI
-// runners do not have (observed as `spawn codex ENOENT`). Runs in no CI lane at all.
// The bulk-open frame probe runs headed: headless Linux compositing schedules idle RAFs
// roughly 1s apart, so it cannot measure foreground interaction against the same budget.
//
@@ -62,6 +60,7 @@ const result = spawnSync(
'tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts',
'tests/e2e/pty-input-write-queue-ssh.spec.ts',
'tests/e2e/ssh-ai-vault-session-history.spec.ts',
+ 'tests/e2e/ssh-codex-display-artifacts-repro.spec.ts',
'tests/e2e/ssh-cold-activation-restore.spec.ts',
'tests/e2e/ssh-cold-hydration-gap-tab-seeding.spec.ts',
'tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts',
diff --git a/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts b/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts
index f4c02d04c94..4677047c5d0 100644
--- a/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts
+++ b/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts
@@ -54,146 +54,151 @@ const CAPTURE_WHILE_REMOTE_TUI_RUNNING =
const HIDE_UNTIL_REMOTE_TUI_DONE = process.env.ORCA_E2E_HIDE_UNTIL_REMOTE_TUI_DONE === '1'
const CAPTURE_SCROLLBACK_ARTIFACT_REGION =
process.env.ORCA_E2E_CAPTURE_SCROLLBACK_ARTIFACT_REGION === '1'
-const FORCE_SSH_RECONNECT_DURING_TUI = process.env.ORCA_E2E_FORCE_SSH_RECONNECT_DURING_TUI === '1'
+const reconnectOverride = process.env.ORCA_E2E_FORCE_SSH_RECONNECT_DURING_TUI
+const reconnectModes = reconnectOverride === undefined ? [false, true] : [reconnectOverride === '1']
const KEEP_SSH_REPRO_TARGET = process.env.ORCA_E2E_KEEP_SSH_REPRO_TARGET === '1'
test.describe('Remote SSH Codex display artifacts repro', () => {
test.skip(!RUN_DOCKER_SSH, 'Set ORCA_E2E_SSH_DOCKER=1 to run Docker-backed SSH repro.')
test.skip(process.platform === 'win32', 'Docker SSH repro uses POSIX ssh tooling.')
- test('does not leave duplicated Codex status output after SSH replay', async ({
- orcaPage
- }, testInfo: TestInfo) => {
- test.slow()
- let target: DockerSshRelayTarget | null = null
- try {
- target = startDockerSshRelayTarget(testInfo)
- installRemoteCodexArtifactTui(target)
- if (RUN_REAL_REMOTE_CODEX) {
- installRemoteRealCodex(target)
- } else {
- installRemoteCodexFixture(target)
- }
- await waitForSessionReady(orcaPage)
- await waitForActiveWorktree(orcaPage)
- const remote = await connectDockerRemote(orcaPage, target)
- expect(remote.targetId).toBeTruthy()
- expect(remote.worktreeId).toBeTruthy()
- await ensureTerminalVisible(orcaPage, 45_000)
- await waitForActiveTerminalManager(orcaPage, 60_000)
- await enableRiskyTerminalRendererPath(orcaPage)
- await installPtyReplayProbe(orcaPage)
+ for (const forceReconnect of reconnectModes) {
+ test(`does not leave duplicated Codex status output after SSH replay (${forceReconnect ? 'forced reconnect' : 'normal restore'})`, async ({
+ orcaPage,
+ electronApp
+ }, testInfo: TestInfo) => {
+ test.slow()
+ let target: DockerSshRelayTarget | null = null
+ try {
+ target = startDockerSshRelayTarget(testInfo)
+ installRemoteCodexArtifactTui(target)
+ if (RUN_REAL_REMOTE_CODEX) {
+ installRemoteRealCodex(target)
+ } else {
+ installRemoteCodexFixture(target)
+ }
+ await waitForSessionReady(orcaPage)
+ await waitForActiveWorktree(orcaPage)
+ const remote = await connectDockerRemote(orcaPage, target)
+ expect(remote.targetId).toBeTruthy()
+ expect(remote.worktreeId).toBeTruthy()
+ await ensureTerminalVisible(orcaPage, 45_000)
+ await waitForActiveTerminalManager(orcaPage, 60_000)
+ await enableRiskyTerminalRendererPath(orcaPage)
- const ptyId = await waitForActivePanePtyId(orcaPage, 60_000)
- const doneMarker = RUN_REAL_REMOTE_CODEX
- ? `ORCA_REAL_REMOTE_CODEX_DONE_${Date.now()}`
- : REMOTE_TUI_DONE
- const cleanMarker = RUN_REAL_REMOTE_CODEX
- ? `ORCA_REAL_REMOTE_CODEX_CLEAN_${Date.now()}`
- : doneMarker
- await execInTerminal(
- orcaPage,
- ptyId,
- RUN_REAL_REMOTE_CODEX
- ? realRemoteCodexCommand(doneMarker)
- : `codex --no-alt-screen --dangerously-bypass-approvals-and-sandbox ${shellQuote(
- doneMarker
- )}`
- )
- await orcaPage.waitForTimeout(1_200)
- if (FORCE_SSH_RECONNECT_DURING_TUI) {
- dropDockerSshClientSessions(target)
- await waitForDockerRemoteReconnected(orcaPage, remote.targetId)
- await orcaPage.waitForTimeout(2_000)
- }
- await (RUN_REAL_REMOTE_CODEX
- ? (async () => {
- await stressRestoreRemoteTerminalDuringCodex(orcaPage, remote.worktreeId)
- await waitForRealRemoteCodexCompletion(orcaPage, doneMarker)
- })()
- : (async () => {
- if (CAPTURE_WHILE_REMOTE_TUI_RUNNING) {
- await orcaPage.waitForTimeout(10_000)
- } else {
- await switchToNonRemoteWorktree(orcaPage, remote.worktreeId)
- await (HIDE_UNTIL_REMOTE_TUI_DONE
- ? waitForRemoteFixtureCleanFinalInHiddenPane(orcaPage, remote.worktreeId)
- : orcaPage.waitForTimeout(10_000))
- }
- if (CAPTURE_WHILE_REMOTE_TUI_RUNNING) {
- await orcaPage.waitForTimeout(900)
- return
- }
- await switchToWorktree(orcaPage, remote.worktreeId)
- await ensureTerminalVisible(orcaPage, 45_000)
- await waitForActiveTerminalManager(orcaPage, 60_000)
- await waitForTerminalOutput(
- orcaPage,
- REMOTE_CODEX_FIXTURE_CLEAN_FINAL_TEXT,
- 60_000,
- 120_000
- )
- })())
- await orcaPage.waitForTimeout(600)
- if (CAPTURE_SCROLLBACK_ARTIFACT_REGION) {
- await scrollActiveTerminalToArtifactHistory(orcaPage)
- }
-
- const { analysis, screenshot } = await captureGraySlabAnalysis(orcaPage)
- analysis.replayDebug = await readReplayProbeSnapshot(orcaPage)
- analysis.duplicateStatusRows = await readDuplicateStatusRows(orcaPage)
- const evidenceLabel = RUN_REAL_REMOTE_CODEX
- ? 'real-remote-codex-reconnect-replay'
- : 'fixture-codex-reconnect-replay'
- persistReproEvidence(evidenceLabel, analysis, screenshot)
- const resetEvidence = await resetWebglAndCaptureGraySlabAnalysis(orcaPage)
- resetEvidence.analysis.replayDebug = await readReplayProbeSnapshot(orcaPage)
- resetEvidence.analysis.duplicateStatusRows = await readDuplicateStatusRows(orcaPage)
- persistReproEvidence(
- `${evidenceLabel}-after-webgl-reset`,
- resetEvidence.analysis,
- resetEvidence.screenshot
- )
- await testInfo.attach('remote-codex-artifact-final-screen', {
- body: screenshot,
- contentType: 'image/png'
- })
- await testInfo.attach('remote-codex-artifact-after-webgl-reset', {
- body: resetEvidence.screenshot,
- contentType: 'image/png'
- })
- testInfo.annotations.push({
- type: 'remote-codex-artifact-analysis',
- description: JSON.stringify(analysis)
- })
- testInfo.annotations.push({
- type: 'remote-codex-artifact-after-webgl-reset-analysis',
- description: JSON.stringify(resetEvidence.analysis)
- })
-
- // Why: this spec supports both repro mode and strict regression mode so
- // the same harness can prove a failure and lock the fixed behavior.
- if (EXPECT_NO_ARTIFACTS) {
- expect(analysis.slabCount).toBeLessThanOrEqual(MAX_FINAL_GRAY_SLABS)
- expect(analysis.staleStatusGlyphRowCount).toBe(0)
- expect(analysis.duplicateStatusRows ?? []).toEqual([])
- } else {
- expect(analysis.rawSlabCount + analysis.staleStatusGlyphRowCount).toBeGreaterThan(0)
- }
- if (FORCE_SSH_RECONNECT_DURING_TUI) {
- expect(Number(analysis.replayDebug?.replayCount ?? 0)).toBeGreaterThan(0)
- }
- if (RUN_REAL_REMOTE_CODEX) {
- await clearRemoteTerminalAfterCodex(orcaPage, ptyId, cleanMarker)
- }
- } finally {
- if (KEEP_SSH_REPRO_TARGET && target) {
- console.log(
- `[ssh-codex-repro] keeping Docker SSH target ${target.containerName} on port ${target.port}`
+ const ptyId = await waitForActivePanePtyId(orcaPage, 60_000)
+ await installPtyReplayProbe(orcaPage, electronApp, ptyId)
+ const doneMarker = RUN_REAL_REMOTE_CODEX
+ ? `ORCA_REAL_REMOTE_CODEX_DONE_${Date.now()}`
+ : REMOTE_TUI_DONE
+ const cleanMarker = RUN_REAL_REMOTE_CODEX
+ ? `ORCA_REAL_REMOTE_CODEX_CLEAN_${Date.now()}`
+ : doneMarker
+ await execInTerminal(
+ orcaPage,
+ ptyId,
+ RUN_REAL_REMOTE_CODEX
+ ? realRemoteCodexCommand(doneMarker)
+ : `codex --no-alt-screen --dangerously-bypass-approvals-and-sandbox ${shellQuote(
+ doneMarker
+ )}`
)
- } else {
- cleanupDockerSshRelayTarget(target)
+ await orcaPage.waitForTimeout(1_200)
+ if (forceReconnect) {
+ dropDockerSshClientSessions(target)
+ await waitForDockerRemoteReconnected(orcaPage, remote.targetId)
+ await orcaPage.waitForTimeout(2_000)
+ }
+ await (RUN_REAL_REMOTE_CODEX
+ ? (async () => {
+ await stressRestoreRemoteTerminalDuringCodex(orcaPage, remote.worktreeId)
+ await waitForRealRemoteCodexCompletion(orcaPage, doneMarker)
+ })()
+ : (async () => {
+ if (CAPTURE_WHILE_REMOTE_TUI_RUNNING) {
+ await orcaPage.waitForTimeout(10_000)
+ } else {
+ await switchToNonRemoteWorktree(orcaPage, remote.worktreeId)
+ await (HIDE_UNTIL_REMOTE_TUI_DONE
+ ? waitForRemoteFixtureCleanFinalInHiddenPane(orcaPage, remote.worktreeId)
+ : orcaPage.waitForTimeout(10_000))
+ }
+ if (CAPTURE_WHILE_REMOTE_TUI_RUNNING) {
+ await orcaPage.waitForTimeout(900)
+ return
+ }
+ await switchToWorktree(orcaPage, remote.worktreeId)
+ await ensureTerminalVisible(orcaPage, 45_000)
+ await waitForActiveTerminalManager(orcaPage, 60_000)
+ await waitForTerminalOutput(
+ orcaPage,
+ REMOTE_CODEX_FIXTURE_CLEAN_FINAL_TEXT,
+ 60_000,
+ 120_000
+ )
+ })())
+ await orcaPage.waitForTimeout(600)
+ if (CAPTURE_SCROLLBACK_ARTIFACT_REGION) {
+ await scrollActiveTerminalToArtifactHistory(orcaPage)
+ }
+
+ const { analysis, screenshot } = await captureGraySlabAnalysis(orcaPage)
+ analysis.replayDebug = await readReplayProbeSnapshot(orcaPage, electronApp)
+ analysis.duplicateStatusRows = await readDuplicateStatusRows(orcaPage)
+ const evidenceLabel = RUN_REAL_REMOTE_CODEX
+ ? 'real-remote-codex-reconnect-replay'
+ : 'fixture-codex-reconnect-replay'
+ persistReproEvidence(evidenceLabel, analysis, screenshot)
+ const resetEvidence = await resetWebglAndCaptureGraySlabAnalysis(orcaPage)
+ resetEvidence.analysis.replayDebug = await readReplayProbeSnapshot(orcaPage, electronApp)
+ resetEvidence.analysis.duplicateStatusRows = await readDuplicateStatusRows(orcaPage)
+ persistReproEvidence(
+ `${evidenceLabel}-after-webgl-reset`,
+ resetEvidence.analysis,
+ resetEvidence.screenshot
+ )
+ await testInfo.attach('remote-codex-artifact-final-screen', {
+ body: screenshot,
+ contentType: 'image/png'
+ })
+ await testInfo.attach('remote-codex-artifact-after-webgl-reset', {
+ body: resetEvidence.screenshot,
+ contentType: 'image/png'
+ })
+ testInfo.annotations.push({
+ type: 'remote-codex-artifact-analysis',
+ description: JSON.stringify(analysis)
+ })
+ testInfo.annotations.push({
+ type: 'remote-codex-artifact-after-webgl-reset-analysis',
+ description: JSON.stringify(resetEvidence.analysis)
+ })
+
+ // Why: this spec supports both repro mode and strict regression mode so
+ // the same harness can prove a failure and lock the fixed behavior.
+ if (EXPECT_NO_ARTIFACTS) {
+ expect(analysis.slabCount).toBeLessThanOrEqual(MAX_FINAL_GRAY_SLABS)
+ expect(analysis.staleStatusGlyphRowCount).toBe(0)
+ expect(analysis.duplicateStatusRows ?? []).toEqual([])
+ } else {
+ expect(analysis.rawSlabCount + analysis.staleStatusGlyphRowCount).toBeGreaterThan(0)
+ }
+ if (forceReconnect) {
+ expect(await waitForActivePanePtyId(orcaPage, 60_000)).toBe(ptyId)
+ expect(Number(analysis.replayDebug?.replayCount ?? 0)).toBeGreaterThan(0)
+ }
+ if (RUN_REAL_REMOTE_CODEX) {
+ await clearRemoteTerminalAfterCodex(orcaPage, ptyId, cleanMarker)
+ }
+ } finally {
+ if (KEEP_SSH_REPRO_TARGET && target) {
+ console.log(
+ `[ssh-codex-repro] keeping Docker SSH target ${target.containerName} on port ${target.port}`
+ )
+ } else {
+ cleanupDockerSshRelayTarget(target)
+ }
}
- }
- })
+ })
+ }
})
diff --git a/tests/e2e/ssh-codex-reconnect-replay-driver.ts b/tests/e2e/ssh-codex-reconnect-replay-driver.ts
index 8a6f0029ff2..a54a4d4c308 100644
--- a/tests/e2e/ssh-codex-reconnect-replay-driver.ts
+++ b/tests/e2e/ssh-codex-reconnect-replay-driver.ts
@@ -1,5 +1,6 @@
+import { installSshReplayReplyProbe, readSshReplayReplies } from './ssh-codex-replay-reply-probe'
import { execFileSync } from 'node:child_process'
-import type { Page } from '@stablyai/playwright-test'
+import type { ElectronApplication, Page } from '@stablyai/playwright-test'
import { expect } from './helpers/orca-app'
import {
DOCKER_SSH_RELAY_REMOTE_REPO_PATH,
@@ -135,8 +136,13 @@ export async function switchToNonRemoteWorktree(
return otherWorktreeId
}
-export async function installPtyReplayProbe(page: Page): Promise {
- await page.evaluate(() => {
+export async function installPtyReplayProbe(
+ page: Page,
+ app: ElectronApplication,
+ ptyId: string
+): Promise {
+ await installSshReplayReplyProbe(app, ptyId)
+ await page.evaluate((expectedPtyId) => {
const api = window.api?.pty
if (!api || typeof api.onReplay !== 'function') {
throw new Error('PTY replay API unavailable')
@@ -150,6 +156,9 @@ export async function installPtyReplayProbe(page: Page): Promise {
holder.__orcaSshCodexReplayProbe?.dispose()
const payloads: { id: string; length: number; preview: string }[] = []
const dispose = api.onReplay(({ id, data }) => {
+ if (id !== expectedPtyId) {
+ return
+ }
payloads.push({
id,
length: data.length,
@@ -157,7 +166,7 @@ export async function installPtyReplayProbe(page: Page): Promise {
})
})
holder.__orcaSshCodexReplayProbe = { payloads, dispose }
- })
+ }, ptyId)
}
export async function waitForDockerRemoteReconnected(page: Page, targetId: string): Promise {
@@ -182,8 +191,12 @@ export async function waitForDockerRemoteReconnected(page: Page, targetId: strin
.toBe(true)
}
-export async function readReplayProbeSnapshot(page: Page): Promise> {
- return page.evaluate(() => {
+export async function readReplayProbeSnapshot(
+ page: Page,
+ app: ElectronApplication
+): Promise> {
+ const replies = await readSshReplayReplies(app)
+ return page.evaluate((replies) => {
const probe = (
window as unknown as {
__orcaSshCodexReplayProbe?: {
@@ -192,10 +205,10 @@ export async function readReplayProbeSnapshot(page: Page): Promise {
diff --git a/tests/e2e/ssh-codex-replay-reply-probe.ts b/tests/e2e/ssh-codex-replay-reply-probe.ts
new file mode 100644
index 00000000000..27981628314
--- /dev/null
+++ b/tests/e2e/ssh-codex-replay-reply-probe.ts
@@ -0,0 +1,55 @@
+import type { ElectronApplication } from '@stablyai/playwright-test'
+
+type ReplayPayload = { id: string; length: number; preview: string; source: 'spawn-reply' }
+type SpawnHandler = (event: unknown, args: Record) => Promise
+type ReplayReplyScope = typeof globalThis & {
+ __orcaSshCodexReplayReplies?: ReplayPayload[]
+}
+
+export async function installSshReplayReplyProbe(
+ app: ElectronApplication,
+ ptyId: string
+): Promise {
+ await app.evaluate(({ ipcMain }, expectedPtyId) => {
+ const scope = globalThis as ReplayReplyScope
+ if (scope.__orcaSshCodexReplayReplies) {
+ throw new Error('SSH replay reply probe already installed')
+ }
+ const handlers = (ipcMain as unknown as { _invokeHandlers?: Map })
+ ._invokeHandlers
+ const original = handlers?.get('pty:spawn')
+ if (!handlers || !original) {
+ throw new Error('PTY spawn handler unavailable')
+ }
+ const payloads: ReplayPayload[] = []
+ scope.__orcaSshCodexReplayReplies = payloads
+ // SSH reconnect returns its replay with the reattach reply, without a pty:replay push.
+ handlers.set('pty:spawn', async (event, args) => {
+ const result = await original(event, args)
+ if (
+ args.sessionId === expectedPtyId &&
+ result &&
+ typeof result === 'object' &&
+ 'id' in result &&
+ result.id === expectedPtyId &&
+ 'isReattach' in result &&
+ result.isReattach === true &&
+ 'replay' in result &&
+ typeof result.replay === 'string' &&
+ result.replay.length > 0
+ ) {
+ payloads.push({
+ id: expectedPtyId,
+ length: result.replay.length,
+ preview: result.replay.slice(-400),
+ source: 'spawn-reply'
+ })
+ }
+ return result
+ })
+ }, ptyId)
+}
+
+export async function readSshReplayReplies(app: ElectronApplication): Promise {
+ return app.evaluate(() => (globalThis as ReplayReplyScope).__orcaSshCodexReplayReplies ?? [])
+}
diff --git a/tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts b/tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts
new file mode 100644
index 00000000000..a52f07941b7
--- /dev/null
+++ b/tests/e2e/ssh-codex-replay-reply-probe.unit.test.ts
@@ -0,0 +1,52 @@
+import type { ElectronApplication } from '@stablyai/playwright-test'
+import { afterEach, beforeEach, expect, it, vi } from 'vitest'
+import { installSshReplayReplyProbe, readSshReplayReplies } from './ssh-codex-replay-reply-probe'
+
+beforeEach(() => vi.stubGlobal('__orcaSshCodexReplayReplies', undefined))
+afterEach(() => vi.unstubAllGlobals())
+
+function harness(result: unknown) {
+ const original = vi.fn().mockResolvedValue(result)
+ const handlers = new Map([['pty:spawn', original]])
+ const app = {
+ evaluate: (fn: (electron: unknown, arg: unknown) => unknown, arg: unknown) =>
+ fn({ ipcMain: { _invokeHandlers: handlers } }, arg)
+ } as unknown as ElectronApplication
+ return { app, original, handlers }
+}
+
+it('records the original PTY reattach reply without changing the handler result', async () => {
+ const result = { id: 'ssh:target@@pty-1', isReattach: true, replay: 'restored output' }
+ const { app, handlers, original } = harness(result)
+ await installSshReplayReplyProbe(app, result.id)
+ const event = {}
+ const args = { sessionId: result.id }
+ expect(await handlers.get('pty:spawn')!(event, args)).toBe(result)
+ expect(original).toHaveBeenCalledWith(event, args)
+ expect(await readSshReplayReplies(app)).toEqual([
+ { id: result.id, length: 15, preview: 'restored output', source: 'spawn-reply' }
+ ])
+})
+
+it.each([
+ [{}, { id: 'wanted', isReattach: true, replay: 'initial' }],
+ [{ sessionId: 'other' }, { id: 'other', isReattach: true, replay: 'other PTY' }],
+ [{ sessionId: 'wanted' }, { id: 'replacement', isReattach: true, replay: 'new PTY' }],
+ [{ sessionId: 'wanted' }, { id: 'wanted', replay: 'no reattach proof' }],
+ [{ sessionId: 'wanted' }, { id: 'wanted', isReattach: true, replay: '' }],
+ [{ sessionId: 'wanted' }, { id: 'wanted', isReattach: true, snapshot: 'not replay' }]
+])('does not count unrelated or unproven replay: %j', async (args, result) => {
+ const { app, handlers } = harness(result)
+ await installSshReplayReplyProbe(app, 'wanted')
+ expect(await handlers.get('pty:spawn')!({}, args)).toBe(result)
+ expect(await readSshReplayReplies(app)).toEqual([])
+})
+
+it('preserves a failed reattach without recording replay', async () => {
+ const { app, handlers, original } = harness(null)
+ const error = new Error('unverifiable')
+ original.mockRejectedValue(error)
+ await installSshReplayReplyProbe(app, 'wanted')
+ await expect(handlers.get('pty:spawn')!({}, { sessionId: 'wanted' })).rejects.toBe(error)
+ expect(await readSshReplayReplies(app)).toEqual([])
+})
From c36c23df5f3ebc76c17a3392d7bc0d5355bd006d Mon Sep 17 00:00:00 2001
From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com>
Date: Sun, 6 Sep 2026 11:31:07 -0700
Subject: [PATCH 17/32] fix(chat): stop terminal focus recovery from stealing
Cmd+C in the Chat UI (#18751)
* fix(chat): preserve message copy focus
* fix(chat): scope covered-xterm focus guard to the chat leaf
Chat view mode is a tab flag, but only the chat leaf's xterm is covered. In a
split chat tab with a terminal leaf active, the tab-level guard skipped the
terminal's resume focus and the tab-wide deferred focus then landed on the
covered chat xterm. Decide per pane: resume and window-wake read the active
pane's container, and the surface focus query skips leaves hosting the chat
root.
* fix(chat): close covered terminal focus fallbacks
---------
Co-authored-by: Merge Sim
---
.../terminal-pane/TerminalPaneSurface.tsx | 2 +
.../terminal-pane/native-chat-covered-pane.ts | 21 ++++
.../terminal-visibility-resume.test.ts | 49 ++++++++++
.../terminal-visibility-resume.ts | 14 ++-
...e-global-effects-visibility-resume.test.ts | 57 +++++++++++
.../use-terminal-pane-global-effects.ts | 9 +-
.../use-terminal-pane-global-listeners.ts | 2 +
.../use-terminal-window-wake-recovery.test.ts | 38 +++++++-
.../use-terminal-window-wake-recovery.ts | 7 +-
.../lib/focus-terminal-tab-surface.test.ts | 95 +++++++++++++++++--
.../src/lib/focus-terminal-tab-surface.ts | 17 +++-
11 files changed, 288 insertions(+), 23 deletions(-)
create mode 100644 src/renderer/src/components/terminal-pane/native-chat-covered-pane.ts
diff --git a/src/renderer/src/components/terminal-pane/TerminalPaneSurface.tsx b/src/renderer/src/components/terminal-pane/TerminalPaneSurface.tsx
index 1773aa48d20..dc346ce95b4 100644
--- a/src/renderer/src/components/terminal-pane/TerminalPaneSurface.tsx
+++ b/src/renderer/src/components/terminal-pane/TerminalPaneSurface.tsx
@@ -48,6 +48,7 @@ export function TerminalPaneSurface({
dismissTerminalError,
expectedLayoutLeafIdsAttr,
expandedPaneId,
+ effectiveChatViewMode,
handleCancelClose,
handleConfirmClose,
handleContextMenuToggleNativeChat,
@@ -117,6 +118,7 @@ export function TerminalPaneSurface({
className="absolute inset-0 min-h-0 min-w-0"
data-native-file-drop-target="terminal"
data-terminal-tab-id={tabId}
+ data-terminal-chat-view={effectiveChatViewMode && activePaneIsChatLeaf ? 'true' : undefined}
data-terminal-layout-leaf-ids={expectedLayoutLeafIdsAttr}
data-pane-title-surface={titleUsesLightSurface ? 'light' : 'dark'}
style={terminalContainerStyle}
diff --git a/src/renderer/src/components/terminal-pane/native-chat-covered-pane.ts b/src/renderer/src/components/terminal-pane/native-chat-covered-pane.ts
new file mode 100644
index 00000000000..f7d6534a70a
--- /dev/null
+++ b/src/renderer/src/components/terminal-pane/native-chat-covered-pane.ts
@@ -0,0 +1,21 @@
+import type { PaneManager } from '@/lib/pane-manager/pane-manager'
+
+const NATIVE_CHAT_COVER_SELECTOR = '.native-chat-pane-shell'
+
+/**
+ * Leaf container selector that excludes panes whose xterm sits under the native
+ * chat portal. Chat mode is a tab flag, but only the chat leaf's xterm is
+ * covered — a split terminal leaf in the same tab must still take focus.
+ */
+export const UNCOVERED_TERMINAL_LEAF_SELECTOR = `[data-leaf-id]:not(:has(${NATIVE_CHAT_COVER_SELECTOR}))`
+
+export function paneIsCoveredByNativeChat(
+ pane: { container: Pick } | null | undefined
+): boolean {
+ return pane?.container.querySelector(NATIVE_CHAT_COVER_SELECTOR) != null
+}
+
+/** Mirrors focusActivePane's target so the guard tracks exactly the pane that would take focus. */
+export function activePaneIsCoveredByNativeChat(manager: PaneManager): boolean {
+ return paneIsCoveredByNativeChat(manager.getActivePane() ?? manager.getPanes()[0])
+}
diff --git a/src/renderer/src/components/terminal-pane/terminal-visibility-resume.test.ts b/src/renderer/src/components/terminal-pane/terminal-visibility-resume.test.ts
index 233ac92bd3b..635adf7c36b 100644
--- a/src/renderer/src/components/terminal-pane/terminal-visibility-resume.test.ts
+++ b/src/renderer/src/components/terminal-pane/terminal-visibility-resume.test.ts
@@ -70,6 +70,7 @@ function resumeArgs(manager: FakeManager, shouldUseLightTabResume: boolean) {
return {
manager: manager as never as PaneManager,
isActive: true,
+ isChatViewMode: false,
wasVisible: false,
shouldUseLightTabResume,
captureViewportPositions: vi.fn(() => new Map()),
@@ -202,6 +203,32 @@ describe('resumeTerminalVisibility reveal repaint', () => {
expect(manager.fitAllPanes).not.toHaveBeenCalled()
})
+ it.each([
+ ['light', true],
+ ['heavy', false]
+ ])('does not focus the covered terminal on a %s chat reveal', async (_path, lightResume) => {
+ const manager = createManager()
+ const args = resumeArgs(manager, lightResume)
+ args.isChatViewMode = true
+ const { focusActivePane } = vi.mocked(await import('./pane-helpers'))
+
+ resumeTerminalVisibility(args)
+
+ expect(focusActivePane).not.toHaveBeenCalled()
+ })
+
+ it.each([
+ ['light', true],
+ ['heavy', false]
+ ])('keeps focusing an active terminal on a %s reveal', async (_path, lightResume) => {
+ const manager = createManager()
+ const { focusActivePane } = vi.mocked(await import('./pane-helpers'))
+
+ resumeTerminalVisibility(resumeArgs(manager, lightResume))
+
+ expect(focusActivePane).toHaveBeenCalledWith(manager)
+ })
+
it('checks each pane for a stale WebGL backing on a light tab reveal', () => {
const first = { terminal: { name: 'pane-a' } }
const second = { terminal: { name: 'pane-b' } }
@@ -233,6 +260,7 @@ describe('resumeTerminalVisibility reveal repaint', () => {
recoverVisibleTerminalWindowWake({
manager: manager as never as PaneManager,
isActive: true,
+ isChatViewMode: false,
clearGlyphAtlases: false
})
@@ -240,6 +268,20 @@ describe('resumeTerminalVisibility reveal repaint', () => {
expect(manager.fitAllPanes).not.toHaveBeenCalled()
})
+ it('does not focus the covered terminal during chat window-wake recovery', async () => {
+ const manager = createManager()
+ const { focusActivePane } = vi.mocked(await import('./pane-helpers'))
+
+ recoverVisibleTerminalWindowWake({
+ manager: manager as never as PaneManager,
+ isActive: true,
+ isChatViewMode: true,
+ clearGlyphAtlases: false
+ })
+
+ expect(focusActivePane).not.toHaveBeenCalled()
+ })
+
it('repairs WebGL canvas backing-store dpr on window wake', () => {
// Clamshell undock: dpr changes while the pane stayed "visible" with a
// stale backing store; tab-reveal is not in the path.
@@ -252,6 +294,7 @@ describe('resumeTerminalVisibility reveal repaint', () => {
recoverVisibleTerminalWindowWake({
manager: manager as never as PaneManager,
isActive: true,
+ isChatViewMode: false,
clearGlyphAtlases: false
})
@@ -275,6 +318,7 @@ describe('resumeTerminalVisibility reveal repaint', () => {
recoverVisibleTerminalWindowWake({
manager: manager as never as PaneManager,
isActive: true,
+ isChatViewMode: false,
clearGlyphAtlases: false
})
@@ -314,6 +358,7 @@ describe('resumeTerminalVisibility reveal repaint', () => {
recoverVisibleTerminalWindowWake({
manager: manager as never as PaneManager,
isActive: true,
+ isChatViewMode: false,
clearGlyphAtlases: false
})
@@ -330,6 +375,7 @@ describe('resumeTerminalVisibility reveal repaint', () => {
recoverVisibleTerminalWindowWake({
manager: manager as never as PaneManager,
isActive: true,
+ isChatViewMode: false,
clearGlyphAtlases: false
})
@@ -341,6 +387,7 @@ describe('resumeTerminalVisibility reveal repaint', () => {
recoverVisibleTerminalWindowWake({
manager: manager as never as PaneManager,
isActive: false,
+ isChatViewMode: false,
clearGlyphAtlases: true
})
@@ -356,6 +403,7 @@ describe('resumeTerminalVisibility reveal repaint', () => {
recoverVisibleTerminalWindowWake({
manager: manager as never as PaneManager,
isActive: false,
+ isChatViewMode: false,
clearGlyphAtlases: true
})
@@ -376,6 +424,7 @@ describe('resumeTerminalVisibility reveal repaint', () => {
recoverVisibleTerminalWindowWake({
manager: manager as never as PaneManager,
isActive: false,
+ isChatViewMode: false,
clearGlyphAtlases: false
})
diff --git a/src/renderer/src/components/terminal-pane/terminal-visibility-resume.ts b/src/renderer/src/components/terminal-pane/terminal-visibility-resume.ts
index 6c7935e6fa9..d208fc9ab74 100644
--- a/src/renderer/src/components/terminal-pane/terminal-visibility-resume.ts
+++ b/src/renderer/src/components/terminal-pane/terminal-visibility-resume.ts
@@ -31,6 +31,7 @@ export type TerminalHiddenReason = 'surface' | 'tab'
type ResumeTerminalVisibilityArgs = {
manager: PaneManager
isActive: boolean
+ isChatViewMode: boolean
wasVisible: boolean
shouldUseLightTabResume: boolean
captureViewportPositions: (useRememberedSnapshots: boolean) => Map
@@ -54,12 +55,14 @@ type HideTerminalVisibilityResult = {
type RecoverVisibleTerminalWindowWakeArgs = {
manager: PaneManager
isActive: boolean
+ isChatViewMode: boolean
clearGlyphAtlases: boolean
}
export function resumeTerminalVisibility({
manager,
isActive,
+ isChatViewMode,
wasVisible,
shouldUseLightTabResume,
captureViewportPositions,
@@ -100,13 +103,13 @@ export function resumeTerminalVisibility({
// cell size — refit so cols/rows match before the overlay settles.
manager.fitAllRevealedPanes()
}
- if (isActive) {
+ if (isActive && !isChatViewMode) {
focusActivePane(manager)
}
} else {
// fitAllRevealedPanes flushes after WebGL reattaches, avoiding a redundant
// full refresh in the suspended DOM renderer while preserving first paint.
- repairedDpr = resumeTerminalVisibilityHeavy(manager, isActive)
+ repairedDpr = resumeTerminalVisibilityHeavy(manager, isActive && !isChatViewMode)
}
enforceTerminalViewportIntents(manager)
if (!shouldUseLightTabResume) {
@@ -174,6 +177,7 @@ export function hideTerminalVisibility({
export function recoverVisibleTerminalWindowWake({
manager,
isActive,
+ isChatViewMode,
clearGlyphAtlases
}: RecoverVisibleTerminalWindowWakeArgs): void {
// Why: macOS screensaver/display wake can leave xterm visible but with a
@@ -201,7 +205,7 @@ export function recoverVisibleTerminalWindowWake({
manager.resumeRendering()
// Why: wake re-attaches WebGL — same transient cell-metric wobble guard as the heavy resume.
manager.fitAllRevealedPanes()
- if (isActive) {
+ if (isActive && !isChatViewMode) {
focusActivePane(manager)
}
enforceTerminalViewportIntents(manager)
@@ -226,7 +230,7 @@ function requestLightTabBacklogRecovery(manager: PaneManager): void {
}
}
-function resumeTerminalVisibilityHeavy(manager: PaneManager, isActive: boolean): boolean {
+function resumeTerminalVisibilityHeavy(manager: PaneManager, shouldFocus: boolean): boolean {
// Why: hidden panes can accumulate large PTY bursts while Chromium is
// occluded. Drain a bounded slice before fitting; the scheduler keeps
// ordering and continues the rest asynchronously so return-to-app does
@@ -254,7 +258,7 @@ function resumeTerminalVisibilityHeavy(manager: PaneManager, isActive: boolean):
// from the DOM renderer's; a raw fit here reflows on a transient one-column-off
// grid and garbles diff-painting inline TUIs (grok minimize→restore).
manager.fitAllRevealedPanes()
- if (isActive) {
+ if (shouldFocus) {
focusActivePane(manager)
}
return repairedDpr
diff --git a/src/renderer/src/components/terminal-pane/use-terminal-pane-global-effects-visibility-resume.test.ts b/src/renderer/src/components/terminal-pane/use-terminal-pane-global-effects-visibility-resume.test.ts
index 2b47298889e..cfb4fc1f95d 100644
--- a/src/renderer/src/components/terminal-pane/use-terminal-pane-global-effects-visibility-resume.test.ts
+++ b/src/renderer/src/components/terminal-pane/use-terminal-pane-global-effects-visibility-resume.test.ts
@@ -307,6 +307,63 @@ describe('useTerminalPaneGlobalEffects', () => {
vi.advanceTimersByTime(500)
})
+ it.each([
+ ['skips focus while the chat leaf is active', true],
+ ['keeps focusing an active split terminal leaf', false]
+ ])('chat view mode %s', (_label, covered) => {
+ vi.stubGlobal(
+ 'requestAnimationFrame',
+ vi.fn((callback: FrameRequestCallback) => {
+ callback(0)
+ return 1
+ })
+ )
+ const pane = {
+ id: 1,
+ terminal: { name: 'terminal-a' },
+ container: { querySelector: vi.fn(() => (covered ? {} : null)) }
+ }
+ const manager = {
+ getPanes: vi.fn(() => [pane]),
+ resumeRendering: vi.fn(),
+ resetWebglTextureAtlases: vi.fn(),
+ scheduleRevealRepaint: vi.fn(),
+ scheduleRevealPresent: vi.fn(),
+ refreshAllPanes: vi.fn(),
+ suspendRendering: vi.fn(),
+ fitAllPanes: vi.fn(),
+ fitAllRevealedPanes: vi.fn(),
+ getActivePane: vi.fn(() => pane),
+ setActivePane: vi.fn()
+ }
+ registerManagerForReset(manager)
+
+ beginHookRender()
+ useTerminalPaneGlobalEffects({
+ tabId: 'tab-1',
+ worktreeId: 'wt-1',
+ managerRef: { current: manager as never },
+ containerRef: { current: null },
+ paneTransportsRef: { current: new Map() },
+ isActiveRef: { current: false },
+ isVisibleRef: { current: false },
+ paneCount: 1,
+ isSyncFitEnabled: true,
+ isWorktreeActive: true,
+ toggleExpandPane: vi.fn(),
+ isActive: true,
+ isVisible: true,
+ isChatViewMode: true
+ })
+
+ expect(pane.container.querySelector).toHaveBeenCalledWith('.native-chat-pane-shell')
+ if (covered) {
+ expect(mocks.focusActivePane).not.toHaveBeenCalled()
+ } else {
+ expect(mocks.focusActivePane).toHaveBeenCalledWith(manager)
+ }
+ })
+
it('keeps visible active-state updates on the light resume path', () => {
vi.useFakeTimers()
vi.stubGlobal(
diff --git a/src/renderer/src/components/terminal-pane/use-terminal-pane-global-effects.ts b/src/renderer/src/components/terminal-pane/use-terminal-pane-global-effects.ts
index 9e9a63a94f2..ec693cfe579 100644
--- a/src/renderer/src/components/terminal-pane/use-terminal-pane-global-effects.ts
+++ b/src/renderer/src/components/terminal-pane/use-terminal-pane-global-effects.ts
@@ -26,6 +26,7 @@ import {
releaseRendererPtyVisibilityClaim,
setRendererPtyVisibilityClaim
} from './pty-renderer-delivery-claims'
+import { activePaneIsCoveredByNativeChat } from './native-chat-covered-pane'
type UseTerminalPaneGlobalEffectsArgs = {
tabId: string
@@ -33,6 +34,7 @@ type UseTerminalPaneGlobalEffectsArgs = {
cwd?: string
isActive: boolean
isVisible: boolean
+ isChatViewMode?: boolean
isWorktreeActive?: boolean
isSyncFitEnabled: boolean
paneCount: number
@@ -66,6 +68,7 @@ export function useTerminalPaneGlobalEffects({
cwd,
isActive,
isVisible,
+ isChatViewMode = false,
isWorktreeActive = isVisible,
isSyncFitEnabled,
paneCount,
@@ -121,6 +124,7 @@ export function useTerminalPaneGlobalEffects({
})
useTerminalWindowWakeRecovery({
isVisible: rendererVisible,
+ isChatViewMode,
managerRef,
isActiveRef,
isVisibleRef,
@@ -156,6 +160,9 @@ export function useTerminalPaneGlobalEffects({
resumeTerminalVisibility({
manager,
isActive,
+ // Why: chat mode is tab-wide, but only the chat leaf's xterm is covered;
+ // a split terminal leaf that is active must still regain focus on reveal.
+ isChatViewMode: isChatViewMode && activePaneIsCoveredByNativeChat(manager),
wasVisible,
shouldUseLightTabResume,
captureViewportPositions,
@@ -183,7 +190,7 @@ export function useTerminalPaneGlobalEffects({
wasVisibleRef.current = false
wasWorktreeActiveRef.current = isWorktreeActive
// eslint-disable-next-line react-hooks/exhaustive-deps
- }, [isActive, isWorktreeActive, rendererVisible])
+ }, [isActive, isChatViewMode, isWorktreeActive, rendererVisible])
useEffect(() => {
const ptyId = isActive && isVisible && isWorktreeActive ? activeLeafPtyId : null
diff --git a/src/renderer/src/components/terminal-pane/use-terminal-pane-global-listeners.ts b/src/renderer/src/components/terminal-pane/use-terminal-pane-global-listeners.ts
index ac8cc9b1c30..57ce985002a 100644
--- a/src/renderer/src/components/terminal-pane/use-terminal-pane-global-listeners.ts
+++ b/src/renderer/src/components/terminal-pane/use-terminal-pane-global-listeners.ts
@@ -25,6 +25,7 @@ export function useTerminalPaneGlobalListeners(controller: TerminalPaneCloseCont
handleRequestClosePane,
handleSearchSelectedText,
handleStartRename,
+ effectiveChatViewMode,
isActive,
isActiveRef,
isRendererVisible,
@@ -91,6 +92,7 @@ export function useTerminalPaneGlobalListeners(controller: TerminalPaneCloseCont
cwd,
isActive,
isVisible,
+ isChatViewMode: effectiveChatViewMode,
isWorktreeActive,
isSyncFitEnabled: isRendererVisible || shouldMeasureHiddenStartup,
paneCount,
diff --git a/src/renderer/src/components/terminal-pane/use-terminal-window-wake-recovery.test.ts b/src/renderer/src/components/terminal-pane/use-terminal-window-wake-recovery.test.ts
index c903ec1a2f9..7a5a7a090da 100644
--- a/src/renderer/src/components/terminal-pane/use-terminal-window-wake-recovery.test.ts
+++ b/src/renderer/src/components/terminal-pane/use-terminal-window-wake-recovery.test.ts
@@ -61,11 +61,16 @@ describe('useTerminalWindowWakeRecovery', () => {
delete (window as unknown as { api?: unknown }).api
})
- function renderWakeRecoveryHook(isVisible = true) {
+ function renderWakeRecoveryHook(
+ isVisible = true,
+ isChatViewMode = false,
+ wakeManager: PaneManager = manager
+ ) {
return renderHook(() =>
useTerminalWindowWakeRecovery({
isVisible,
- managerRef: { current: manager },
+ isChatViewMode,
+ managerRef: { current: wakeManager },
isActiveRef: { current: true },
isVisibleRef: { current: true }
})
@@ -83,6 +88,7 @@ describe('useTerminalWindowWakeRecovery', () => {
expect(recoverVisibleTerminalWindowWakeMock).toHaveBeenNthCalledWith(1, {
manager,
isActive: true,
+ isChatViewMode: false,
clearGlyphAtlases: false
})
@@ -93,6 +99,7 @@ describe('useTerminalWindowWakeRecovery', () => {
expect(recoverVisibleTerminalWindowWakeMock).toHaveBeenNthCalledWith(2, {
manager,
isActive: true,
+ isChatViewMode: false,
clearGlyphAtlases: true
})
})
@@ -109,6 +116,27 @@ describe('useTerminalWindowWakeRecovery', () => {
expect(recoverVisibleTerminalWindowWakeMock).toHaveBeenLastCalledWith({
manager,
isActive: true,
+ isChatViewMode: false,
+ clearGlyphAtlases: false
+ })
+ })
+
+ it.each([
+ ['covered chat leaf', true],
+ ['split terminal leaf', false]
+ ])('routes chat coverage into wake recovery only for the %s', (_label, covered) => {
+ const chatManager = {
+ getActivePane: () => ({ container: { querySelector: () => (covered ? {} : null) } }),
+ getPanes: () => []
+ } as unknown as PaneManager
+ renderWakeRecoveryHook(true, true, chatManager)
+
+ window.dispatchEvent(new Event('focus'))
+
+ expect(recoverVisibleTerminalWindowWakeMock).toHaveBeenLastCalledWith({
+ manager: chatManager,
+ isActive: true,
+ isChatViewMode: covered,
clearGlyphAtlases: false
})
})
@@ -136,6 +164,7 @@ describe('useTerminalWindowWakeRecovery', () => {
renderHook(() =>
useTerminalWindowWakeRecovery({
isVisible: true,
+ isChatViewMode: false,
managerRef: { current: manager },
isActiveRef: { current: true },
isVisibleRef: { current: true },
@@ -163,6 +192,7 @@ describe('useTerminalWindowWakeRecovery', () => {
renderHook(() =>
useTerminalWindowWakeRecovery({
isVisible: true,
+ isChatViewMode: false,
managerRef: { current: manager },
isActiveRef: { current: true },
isVisibleRef: { current: true },
@@ -209,6 +239,7 @@ describe('useTerminalWindowWakeRecovery', () => {
const { unmount } = renderHook(() =>
useTerminalWindowWakeRecovery({
isVisible: true,
+ isChatViewMode: false,
managerRef: { current: resizeManager },
isActiveRef: { current: true },
isVisibleRef: { current: true }
@@ -238,6 +269,7 @@ describe('useTerminalWindowWakeRecovery', () => {
renderHook(() =>
useTerminalWindowWakeRecovery({
isVisible: true,
+ isChatViewMode: false,
managerRef,
isActiveRef: { current: true },
isVisibleRef: { current: true }
@@ -265,6 +297,7 @@ describe('useTerminalWindowWakeRecovery', () => {
renderHook(() =>
useTerminalWindowWakeRecovery({
isVisible: true,
+ isChatViewMode: false,
managerRef: { current: { getPanes: () => [pane] } as unknown as PaneManager },
isActiveRef: { current: true },
isVisibleRef: { current: true }
@@ -295,6 +328,7 @@ describe('useTerminalWindowWakeRecovery', () => {
renderHook(() =>
useTerminalWindowWakeRecovery({
isVisible: true,
+ isChatViewMode: false,
managerRef: { current: { getPanes: () => [pane] } as unknown as PaneManager },
isActiveRef: { current: true },
isVisibleRef: { current: true }
diff --git a/src/renderer/src/components/terminal-pane/use-terminal-window-wake-recovery.ts b/src/renderer/src/components/terminal-pane/use-terminal-window-wake-recovery.ts
index 444d0f1f9dd..49f71ed6f30 100644
--- a/src/renderer/src/components/terminal-pane/use-terminal-window-wake-recovery.ts
+++ b/src/renderer/src/components/terminal-pane/use-terminal-window-wake-recovery.ts
@@ -5,9 +5,11 @@ import { repairPaneWebglCanvasDpr } from '@/lib/pane-manager/terminal-canvas-dpr
import { presentPaneViewport } from '@/lib/pane-manager/pane-webgl-renderer'
import { recordTerminalFreezeBreadcrumb } from './terminal-freeze-breadcrumbs'
import type { IDisposable } from '@xterm/xterm'
+import { activePaneIsCoveredByNativeChat } from './native-chat-covered-pane'
type UseTerminalWindowWakeRecoveryArgs = {
isVisible: boolean
+ isChatViewMode: boolean
managerRef: React.RefObject
isActiveRef: React.RefObject
isVisibleRef: React.RefObject
@@ -22,6 +24,7 @@ const DPR_RECOVERY_RETRY_FRAMES = 16
export function useTerminalWindowWakeRecovery({
isVisible,
+ isChatViewMode,
managerRef,
isActiveRef,
isVisibleRef,
@@ -85,6 +88,7 @@ export function useTerminalWindowWakeRecovery({
recoverVisibleTerminalWindowWake({
manager,
isActive: isActiveRef.current,
+ isChatViewMode: isChatViewMode && activePaneIsCoveredByNativeChat(manager),
clearGlyphAtlases
})
if (typeof requestAnimationFrame !== 'function') {
@@ -103,6 +107,7 @@ export function useTerminalWindowWakeRecovery({
recoverVisibleTerminalWindowWake({
manager: settledManager,
isActive: isActiveRef.current,
+ isChatViewMode: isChatViewMode && activePaneIsCoveredByNativeChat(settledManager),
clearGlyphAtlases: clearGlyphAtlasesOnSettle
})
reassertPanePtySizes()
@@ -199,5 +204,5 @@ export function useTerminalWindowWakeRecovery({
}
unsubscribeSystemResumed?.()
}
- }, [isActiveRef, isVisible, isVisibleRef, managerRef, panePtyBindingsRef])
+ }, [isActiveRef, isChatViewMode, isVisible, isVisibleRef, managerRef, panePtyBindingsRef])
}
diff --git a/src/renderer/src/lib/focus-terminal-tab-surface.test.ts b/src/renderer/src/lib/focus-terminal-tab-surface.test.ts
index 9df0ef4132d..20cf491088e 100644
--- a/src/renderer/src/lib/focus-terminal-tab-surface.test.ts
+++ b/src/renderer/src/lib/focus-terminal-tab-surface.test.ts
@@ -9,6 +9,12 @@ vi.mock('@/components/terminal-pane/terminal-ime-input-context-refresh', () => (
refreshTerminalImeInputContext: mocks.refreshTerminalImeInputContext
}))
+// Why: tab-wide queries skip leaves whose xterm sits under the native chat portal.
+const TAB_HELPER_SELECTOR =
+ '[data-terminal-tab-id="tab-1"] [data-leaf-id]:not(:has(.native-chat-pane-shell)) .xterm-helper-textarea'
+const GLOBAL_HELPER_SELECTOR =
+ '[data-leaf-id]:not(:has(.native-chat-pane-shell)) .xterm-helper-textarea'
+
describe('focusTerminalTabSurface', () => {
afterEach(() => {
mocks.refreshTerminalImeInputContext.mockClear()
@@ -28,7 +34,7 @@ describe('focusTerminalTabSurface', () => {
const textarea = { focus: vi.fn() }
vi.stubGlobal('document', {
querySelector: vi.fn((selector: string) =>
- selector === '[data-terminal-tab-id="tab-1"] .xterm-helper-textarea' ? textarea : null
+ selector === TAB_HELPER_SELECTOR ? textarea : null
)
})
@@ -42,7 +48,7 @@ describe('focusTerminalTabSurface', () => {
const textarea = { focus: vi.fn() }
vi.stubGlobal('document', {
querySelector: vi.fn((selector: string) =>
- selector === '[data-terminal-tab-id="tab-1"] .xterm-helper-textarea' ? textarea : null
+ selector === TAB_HELPER_SELECTOR ? textarea : null
)
})
@@ -68,7 +74,7 @@ describe('focusTerminalTabSurface', () => {
activeElement: body as unknown,
body,
querySelector: vi.fn((selector: string) =>
- selector === '[data-terminal-tab-id="tab-1"] .xterm-helper-textarea' ? textarea : null
+ selector === TAB_HELPER_SELECTOR ? textarea : null
)
}
vi.stubGlobal('document', documentState)
@@ -89,9 +95,7 @@ describe('focusTerminalTabSurface', () => {
if (selector === '[data-tab-rename-input="true"]') {
return {}
}
- return selector === '[data-terminal-tab-id="tab-1"] .xterm-helper-textarea'
- ? textarea
- : null
+ return selector === TAB_HELPER_SELECTOR ? textarea : null
})
})
@@ -100,6 +104,77 @@ describe('focusTerminalTabSurface', () => {
expect(textarea.focus).not.toHaveBeenCalled()
})
+ it('does not focus xterm while chat covers the terminal tab', () => {
+ flushAnimationFrames()
+ const textarea = { focus: vi.fn() }
+ vi.stubGlobal('document', {
+ querySelector: vi.fn((selector: string) => {
+ if (selector === '[data-terminal-tab-id="tab-1"]') {
+ return {
+ getAttribute: (name: string) => (name === 'data-terminal-chat-view' ? 'true' : null)
+ }
+ }
+ return selector === TAB_HELPER_SELECTOR ? textarea : null
+ })
+ })
+
+ focusTerminalTabSurface('tab-1')
+
+ expect(textarea.focus).not.toHaveBeenCalled()
+ })
+
+ it('skips the chat leaf helper when a split chat tab has an active terminal leaf', () => {
+ flushAnimationFrames()
+ const coveredTextarea = { focus: vi.fn() }
+ const terminalTextarea = { focus: vi.fn() }
+ vi.stubGlobal('document', {
+ querySelector: vi.fn((selector: string) => {
+ if (selector === '[data-terminal-tab-id="tab-1"]') {
+ return { getAttribute: () => null }
+ }
+ if (selector === TAB_HELPER_SELECTOR) {
+ return terminalTextarea
+ }
+ return selector === '[data-terminal-tab-id="tab-1"] .xterm-helper-textarea'
+ ? coveredTextarea
+ : null
+ })
+ })
+
+ focusTerminalTabSurface('tab-1')
+
+ expect(terminalTextarea.focus).toHaveBeenCalledOnce()
+ expect(coveredTextarea.focus).not.toHaveBeenCalled()
+ })
+
+ it('does not use a covered chat helper as the global mount-race fallback', () => {
+ flushAnimationFrames()
+ const coveredTextarea = { focus: vi.fn() }
+ vi.stubGlobal('document', {
+ querySelector: vi.fn((selector: string) =>
+ selector === '.xterm-helper-textarea' ? coveredTextarea : null
+ )
+ })
+
+ focusTerminalTabSurface('tab-1')
+
+ expect(coveredTextarea.focus).not.toHaveBeenCalled()
+ })
+
+ it('keeps the global mount-race fallback for an uncovered terminal helper', () => {
+ flushAnimationFrames()
+ const textarea = { focus: vi.fn() }
+ vi.stubGlobal('document', {
+ querySelector: vi.fn((selector: string) =>
+ selector === GLOBAL_HELPER_SELECTOR ? textarea : null
+ )
+ })
+
+ focusTerminalTabSurface('tab-1')
+
+ expect(textarea.focus).toHaveBeenCalledOnce()
+ })
+
it('falls back to the single tab helper when an old leaf id was reminted', () => {
flushAnimationFrames()
const textarea = { focus: vi.fn() }
@@ -108,7 +183,7 @@ describe('focusTerminalTabSurface', () => {
selector === '[data-terminal-tab-id="tab-1"]' ? { getAttribute: () => 'new-leaf' } : null
),
querySelectorAll: vi.fn((selector: string) =>
- selector === '[data-terminal-tab-id="tab-1"] .xterm-helper-textarea'
+ selector === TAB_HELPER_SELECTOR
? { length: 1, item: () => textarea }
: { length: 0, item: () => null }
)
@@ -129,7 +204,7 @@ describe('focusTerminalTabSurface', () => {
: null
),
querySelectorAll: vi.fn((selector: string) =>
- selector === '[data-terminal-tab-id="tab-1"] .xterm-helper-textarea'
+ selector === TAB_HELPER_SELECTOR
? { length: 1, item: () => textarea }
: { length: 0, item: () => null }
)
@@ -150,7 +225,7 @@ describe('focusTerminalTabSurface', () => {
: null
),
querySelectorAll: vi.fn((selector: string) =>
- selector === '[data-terminal-tab-id="tab-1"] .xterm-helper-textarea'
+ selector === TAB_HELPER_SELECTOR
? { length: 1, item: () => textarea }
: { length: 0, item: () => null }
)
@@ -172,7 +247,7 @@ describe('focusTerminalTabSurface', () => {
: null
),
querySelectorAll: vi.fn((selector: string) =>
- selector === '[data-terminal-tab-id="tab-1"] .xterm-helper-textarea'
+ selector === TAB_HELPER_SELECTOR
? { length: 2, item: (index: number) => (index === 0 ? first : second) }
: { length: 0, item: () => null }
)
diff --git a/src/renderer/src/lib/focus-terminal-tab-surface.ts b/src/renderer/src/lib/focus-terminal-tab-surface.ts
index 97ceb97ae61..ef23ff051d7 100644
--- a/src/renderer/src/lib/focus-terminal-tab-surface.ts
+++ b/src/renderer/src/lib/focus-terminal-tab-surface.ts
@@ -1,4 +1,5 @@
import { refreshTerminalImeInputContext } from '@/components/terminal-pane/terminal-ime-input-context-refresh'
+import { UNCOVERED_TERMINAL_LEAF_SELECTOR } from '@/components/terminal-pane/native-chat-covered-pane'
/**
* Move keyboard focus into the xterm instance for a freshly-mounted terminal
@@ -70,9 +71,15 @@ export function focusTerminalTabSurface(
return
}
const escapedTabId = cssAttributeString(tabId)
+ const tabElement = document.querySelector(`[data-terminal-tab-id="${escapedTabId}"]`)
+ if (tabElement?.getAttribute('data-terminal-chat-view') === 'true') {
+ return
+ }
+ // Why: a split chat tab keeps a covered xterm under the chat leaf; the
+ // tab-wide query must skip it or the deferred focus lands on it.
const scopedSelector = leafId
- ? `[data-terminal-tab-id="${escapedTabId}"] [data-leaf-id="${cssAttributeString(leafId)}"] .xterm-helper-textarea`
- : `[data-terminal-tab-id="${escapedTabId}"] .xterm-helper-textarea`
+ ? `[data-terminal-tab-id="${escapedTabId}"] [data-leaf-id="${cssAttributeString(leafId)}"]${UNCOVERED_TERMINAL_LEAF_SELECTOR} .xterm-helper-textarea`
+ : `[data-terminal-tab-id="${escapedTabId}"] ${UNCOVERED_TERMINAL_LEAF_SELECTOR} .xterm-helper-textarea`
const scoped = document.querySelector(scopedSelector) as HTMLElement | null
if (scoped) {
focusTerminalHelper(scoped, options)
@@ -87,7 +94,7 @@ export function focusTerminalTabSurface(
// Why: old single-pane remounts could remint the leaf id. Only recover
// after the tab layout no longer expects the requested leaf.
const tabScopedHelpers = document.querySelectorAll(
- `[data-terminal-tab-id="${escapedTabId}"] .xterm-helper-textarea`
+ `[data-terminal-tab-id="${escapedTabId}"] ${UNCOVERED_TERMINAL_LEAF_SELECTOR} .xterm-helper-textarea`
)
if (tabScopedHelpers.length === 1) {
const fallback = tabScopedHelpers.item(0) as HTMLElement | null
@@ -98,7 +105,9 @@ export function focusTerminalTabSurface(
}
return
}
- const fallback = document.querySelector('.xterm-helper-textarea') as HTMLElement | null
+ const fallback = document.querySelector(
+ `${UNCOVERED_TERMINAL_LEAF_SELECTOR} .xterm-helper-textarea`
+ ) as HTMLElement | null
if (fallback) {
focusTerminalHelper(fallback, options)
}
From 7ac194a634b1374a64bca72771fa04c5f5763f0b Mon Sep 17 00:00:00 2001
From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com>
Date: Sun, 6 Sep 2026 11:33:08 -0700
Subject: [PATCH 18/32] Add persistent turn-scoped chat activity indicator
(#19044)
* feat(chat): show turn-scoped activity tail
* fix(chat): keep turn activity broad
---------
Co-authored-by: Merge Sim
---
.../NativeChatMessageList.test.tsx | 178 ++++++++++++++++++
.../native-chat/NativeChatMessageList.tsx | 7 +
.../NativeChatStructuredSession.tsx | 1 +
.../native-chat/NativeChatToolRun.test.tsx | 18 ++
.../native-chat/NativeChatToolRun.tsx | 16 +-
.../NativeChatTurnActivityLine.tsx | 23 +++
.../native-chat-tool-activity-label.ts | 20 ++
.../native-chat-turn-activity.test.ts | 79 ++++++++
.../native-chat/native-chat-turn-activity.ts | 41 ++++
.../use-structured-agent-session.ts | 6 +
10 files changed, 375 insertions(+), 14 deletions(-)
create mode 100644 src/renderer/src/components/native-chat/NativeChatTurnActivityLine.tsx
create mode 100644 src/renderer/src/components/native-chat/native-chat-tool-activity-label.ts
create mode 100644 src/renderer/src/components/native-chat/native-chat-turn-activity.test.ts
create mode 100644 src/renderer/src/components/native-chat/native-chat-turn-activity.ts
diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.test.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.test.tsx
index 860464ac65b..5b71136b86f 100644
--- a/src/renderer/src/components/native-chat/NativeChatMessageList.test.tsx
+++ b/src/renderer/src/components/native-chat/NativeChatMessageList.test.tsx
@@ -84,6 +84,184 @@ describe('NativeChatMessageList assistant messages', () => {
expect(document.querySelector('.text-destructive')).toBeNull()
})
+ it('keeps a reduced-motion-safe spinner activity line at the tail of a no-tool Codex turn', () => {
+ render(
+
+ )
+
+ const activity = screen.getByText('Working…')
+ const row = activity.closest('[data-native-chat-turn-activity]')
+ const spinner = row?.querySelector('svg')
+ expect(activity).not.toHaveClass('animate-pulse', 'animate-spin')
+ expect(spinner).toHaveClass('size-4', 'animate-spin', 'motion-reduce:animate-none')
+ expect(row).toHaveAttribute('aria-live', 'polite')
+ expect(screen.getByText('The answer is still streaming.').compareDocumentPosition(row!)).toBe(
+ Node.DOCUMENT_POSITION_FOLLOWING
+ )
+ })
+
+ it('keeps the broad fallback distinct from the running tool row', () => {
+ render(
+
+ )
+
+ const toolLabel = screen.getByText('Running pnpm test')
+ expect(toolLabel).toHaveClass('animate-pulse')
+ expect(screen.getAllByText('Running pnpm test')).toHaveLength(1)
+ const activity = screen.getByText('Working…')
+ expect(activity.textContent).not.toBe(toolLabel.textContent)
+ expect(activity).not.toHaveTextContent('shell')
+ expect(activity).not.toHaveTextContent('pnpm test')
+ const spinner = activity.closest('[data-native-chat-turn-activity]')?.querySelector('svg')
+ expect(activity).not.toHaveClass('animate-pulse', 'animate-spin')
+ expect(spinner).toHaveClass('animate-spin', 'motion-reduce:animate-none')
+ })
+
+ it('uses the broad fallback after a tool settles', () => {
+ render(
+
+ )
+
+ const settledTool = screen.getByText('shell pnpm test')
+ const activity = screen.getByText('Working…')
+ expect(activity.textContent).not.toBe(settledTool.textContent)
+ expect(activity).not.toHaveTextContent('shell')
+ expect(activity).not.toHaveTextContent('pnpm test')
+ expect(activity).not.toHaveClass('animate-pulse', 'animate-spin')
+ expect(activity.closest('[data-native-chat-turn-activity]')?.querySelector('svg')).toHaveClass(
+ 'animate-spin'
+ )
+ })
+
+ it('keeps a completed tool row static while the turn tail spins, then removes the tail', () => {
+ const workingSession: NativeChatLiveSession = {
+ ...session,
+ status: 'working',
+ messages: [
+ {
+ id: 'assistant-settled-tool',
+ role: 'assistant',
+ blocks: [
+ {
+ type: 'tool-call',
+ name: 'shell',
+ input: { command: 'pnpm test' },
+ state: 'completed'
+ },
+ { type: 'tool-result', output: 'passed' }
+ ],
+ timestamp: 1,
+ source: 'transcript'
+ }
+ ]
+ }
+ const { container, rerender } = render(
+
+ )
+
+ const settledTool = screen.getByText('shell pnpm test')
+ expect(settledTool.closest('button')?.querySelector('.animate-pulse')).toBeNull()
+ expect(settledTool.closest('button')?.querySelector('.lucide-check')).toBeInTheDocument()
+ const activity = screen.getByText('Preparing the answer')
+ expect(activity).not.toHaveClass('animate-pulse', 'animate-spin')
+ expect(activity.closest('[data-native-chat-turn-activity]')?.querySelector('svg')).toHaveClass(
+ 'animate-spin'
+ )
+
+ rerender(
+
+ )
+
+ expect(container.querySelector('[data-native-chat-turn-activity]')).toBeNull()
+ expect(container.querySelector('.animate-pulse')).toBeNull()
+ expect(container.querySelector('.animate-spin')).toBeNull()
+ })
+
it('keeps bridge chats on the legacy activity chrome', () => {
render(
/** Turn timing and disclosure are available on structured agent sessions. */
showTurnStatus?: boolean
+ turnActivity?: NativeChatTurnActivity | null
runtimeContext?: RuntimeFileOperationArgs | null
}): React.JSX.Element {
const scrollRef = useRef(null)
@@ -272,6 +276,9 @@ export function NativeChatMessageList({
workedSeconds={turnStatuses.active.workedSeconds}
/>
) : null}
+ {showTurnStatus && isWorking ? (
+
+ ) : null}
{!showTurnStatus && showTypingIndicator ? : null}
diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx
index 87adb0bda73..64f4c7c1253 100644
--- a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx
+++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx
@@ -180,6 +180,7 @@ export function NativeChatStructuredSession(
fontScale={fontScale.scale}
workingStartedAt={null}
showTurnStatus
+ turnActivity={controller.turnActivity}
onLinkClick={fileLinkClick}
allowFileUriLinks={fileLinkClick !== undefined}
runtimeContext={imageRuntimeContext}
diff --git a/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx b/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx
index e19c203ee5c..da9202254c0 100644
--- a/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx
+++ b/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx
@@ -311,6 +311,24 @@ describe('NativeChatToolRun', () => {
expect(screen.getByText('shell sleep 1')).toBeInTheDocument()
})
+ it('never animates a settled tool row with its completion check', () => {
+ const { container } = render(
+
+ )
+
+ const settledRow = screen.getByText('shell pnpm test').closest('button')
+ expect(settledRow?.querySelector('.lucide-check')).toBeInTheDocument()
+ expect(settledRow?.querySelector('.animate-pulse')).toBeNull()
+ expect(container.querySelector('.animate-pulse')).toBeNull()
+ })
+
it('keeps failed tool runs visually neutral while collapsed', () => {
const blocks: NativeChatBlock[] = [
{ type: 'tool-call', name: 'shell', input: { command: 'false' }, state: 'failed' },
diff --git a/src/renderer/src/components/native-chat/NativeChatToolRun.tsx b/src/renderer/src/components/native-chat/NativeChatToolRun.tsx
index faaab338c66..5c9351eec00 100644
--- a/src/renderer/src/components/native-chat/NativeChatToolRun.tsx
+++ b/src/renderer/src/components/native-chat/NativeChatToolRun.tsx
@@ -22,25 +22,13 @@ import {
truncateToolDetail
} from './native-chat-tool-summary'
import {
- describeActiveToolCall,
NATIVE_CHAT_TOOL_ACTIVITY_COPY,
selectActiveToolCall
} from '../../../../shared/native-chat-tool-activity'
import { nativeChatToolRunIconName } from '../../../../shared/native-chat-tool-icon'
import { NativeChatDiffView } from './NativeChatDiffView'
import { NativeChatToolIcon, NativeChatToolRunIcon } from './NativeChatToolIcon'
-
-function activeToolLabel(call: Extract): string {
- const { key, toolName, preview } = describeActiveToolCall(call)
- const copy = NATIVE_CHAT_TOOL_ACTIVITY_COPY[key]
- return key === 'runningPreview'
- ? translate('components.native-chat.tool.runningPreview', copy, { preview })
- : key === 'runningCommand'
- ? translate('components.native-chat.tool.runningCommand', copy)
- : key === 'runningNamedPreview'
- ? translate('components.native-chat.tool.runningNamedPreview', copy, { toolName, preview })
- : translate('components.native-chat.tool.runningNamed', copy, { toolName })
-}
+import { nativeChatToolActivityLabel } from './native-chat-tool-activity-label'
/** A single inline tool line — `▸ ToolName preview` — that expands in place to
* show the call's diff/input or the result's body. Tool calls read as flat
@@ -267,7 +255,7 @@ export function NativeChatToolRun({
>
- {activeToolLabel(latestActiveCall)}
+ {nativeChatToolActivityLabel(latestActiveCall)}
{open ? : null}
diff --git a/src/renderer/src/components/native-chat/NativeChatTurnActivityLine.tsx b/src/renderer/src/components/native-chat/NativeChatTurnActivityLine.tsx
new file mode 100644
index 00000000000..da11773105d
--- /dev/null
+++ b/src/renderer/src/components/native-chat/NativeChatTurnActivityLine.tsx
@@ -0,0 +1,23 @@
+import { Loader2 } from 'lucide-react'
+import { translate } from '@/i18n/i18n'
+import type { NativeChatTurnActivity } from './native-chat-turn-activity'
+
+export function NativeChatTurnActivityLine({
+ activity
+}: {
+ activity?: NativeChatTurnActivity | null
+}): React.JSX.Element {
+ const label = activity?.text ?? translate('components.native-chat.status.working', 'Working…')
+
+ return (
+
+
+ {label}
+
+ )
+}
diff --git a/src/renderer/src/components/native-chat/native-chat-tool-activity-label.ts b/src/renderer/src/components/native-chat/native-chat-tool-activity-label.ts
new file mode 100644
index 00000000000..2d1f21c22be
--- /dev/null
+++ b/src/renderer/src/components/native-chat/native-chat-tool-activity-label.ts
@@ -0,0 +1,20 @@
+import { translate } from '@/i18n/i18n'
+import {
+ describeActiveToolCall,
+ NATIVE_CHAT_TOOL_ACTIVITY_COPY
+} from '../../../../shared/native-chat-tool-activity'
+import type { NativeChatBlock } from '../../../../shared/native-chat-types'
+
+type ToolCall = Extract
+
+export function nativeChatToolActivityLabel(call: ToolCall): string {
+ const { key, toolName, preview } = describeActiveToolCall(call)
+ const copy = NATIVE_CHAT_TOOL_ACTIVITY_COPY[key]
+ return key === 'runningPreview'
+ ? translate('components.native-chat.tool.runningPreview', copy, { preview })
+ : key === 'runningCommand'
+ ? translate('components.native-chat.tool.runningCommand', copy)
+ : key === 'runningNamedPreview'
+ ? translate('components.native-chat.tool.runningNamedPreview', copy, { toolName, preview })
+ : translate('components.native-chat.tool.runningNamed', copy, { toolName })
+}
diff --git a/src/renderer/src/components/native-chat/native-chat-turn-activity.test.ts b/src/renderer/src/components/native-chat/native-chat-turn-activity.test.ts
new file mode 100644
index 00000000000..f2d30101826
--- /dev/null
+++ b/src/renderer/src/components/native-chat/native-chat-turn-activity.test.ts
@@ -0,0 +1,79 @@
+import { describe, expect, it } from 'vitest'
+import type {
+ AgentJournalItemBody,
+ AgentJournalRenderItem
+} from '../../../../shared/agent-session-journal-types'
+import { selectStructuredAgentTurnActivity } from './native-chat-turn-activity'
+
+function item(sequence: number, body: AgentJournalItemBody): AgentJournalRenderItem {
+ return { itemId: `item-${sequence}`, revision: 1, sequence, observedAt: sequence, body }
+}
+
+const turnStart = item(1, {
+ kind: 'status',
+ text: 'Codex is working…',
+ turnLifecycle: { turnId: 'turn-1', state: 'running' }
+})
+
+describe('selectStructuredAgentTurnActivity', () => {
+ it('prefers the latest provider-authored activity line in the active turn', () => {
+ const activity = selectStructuredAgentTurnActivity(
+ [
+ turnStart,
+ item(2, {
+ kind: 'tool-call',
+ name: 'shell',
+ input: { command: 'pnpm test' },
+ state: 'completed'
+ }),
+ item(3, { kind: 'status', text: 'Checking the results\nPreparing the answer' })
+ ],
+ 'turn-1'
+ )
+
+ expect(activity).toEqual({ kind: 'description', text: 'Preparing the answer' })
+ })
+
+ it('ignores active and settled tools so the tail can use a broad fallback', () => {
+ const activity = selectStructuredAgentTurnActivity(
+ [
+ turnStart,
+ item(2, {
+ kind: 'tool-call',
+ name: 'shell',
+ input: { command: 'pnpm test' },
+ state: 'running'
+ }),
+ item(3, {
+ kind: 'tool-call',
+ name: 'shell',
+ input: { command: 'pnpm lint' },
+ state: 'completed'
+ })
+ ],
+ 'turn-1'
+ )
+
+ expect(activity).toBeNull()
+ })
+
+ it('ignores diagnostic provider frames and returns nothing after the turn settles', () => {
+ const diagnostic = item(2, {
+ kind: 'status',
+ text: 'codex · notification:new/event',
+ providerFrame: {
+ provider: 'codex',
+ kind: 'notification:new/event',
+ payload: {
+ head: '{}',
+ byteLength: 2,
+ digest: 'a'.repeat(64),
+ truncated: false
+ }
+ }
+ })
+
+ expect(selectStructuredAgentTurnActivity([turnStart, diagnostic], 'turn-1')).toBeNull()
+ expect(selectStructuredAgentTurnActivity([turnStart, diagnostic], null)).toBeNull()
+ })
+})
diff --git a/src/renderer/src/components/native-chat/native-chat-turn-activity.ts b/src/renderer/src/components/native-chat/native-chat-turn-activity.ts
new file mode 100644
index 00000000000..37f9fc75015
--- /dev/null
+++ b/src/renderer/src/components/native-chat/native-chat-turn-activity.ts
@@ -0,0 +1,41 @@
+import type { AgentJournalRenderItem } from '../../../../shared/agent-session-journal-types'
+import { normalizePromptField } from '../../../../shared/agent-status-field-normalization'
+
+export type NativeChatTurnActivity = { kind: 'description'; text: string }
+
+function activityLine(text: string): string | null {
+ const lines = text
+ .split('\n')
+ .map((line) => line.trim())
+ .filter(Boolean)
+ const latest = lines.at(-1)
+ return latest ? normalizePromptField(latest) || null : null
+}
+
+/** Prefer provider-authored activity copy; callers provide the broad fallback. */
+export function selectStructuredAgentTurnActivity(
+ items: readonly AgentJournalRenderItem[],
+ turnId: string | null
+): NativeChatTurnActivity | null {
+ if (!turnId) {
+ return null
+ }
+ const turnStartIndex = items.findLastIndex(
+ (item) =>
+ item.body.kind === 'status' &&
+ item.body.turnLifecycle?.turnId === turnId &&
+ item.body.turnLifecycle.state === 'running'
+ )
+ const turnItems = items.slice(Math.max(0, turnStartIndex))
+ for (let index = turnItems.length - 1; index >= 0; index -= 1) {
+ const body = turnItems[index]?.body
+ if (body?.kind !== 'status' || body.turnLifecycle || body.providerFrame) {
+ continue
+ }
+ const text = activityLine(body.text)
+ if (text) {
+ return { kind: 'description', text }
+ }
+ }
+ return null
+}
diff --git a/src/renderer/src/components/native-chat/use-structured-agent-session.ts b/src/renderer/src/components/native-chat/use-structured-agent-session.ts
index 8af9a36e0c4..b0bab73669c 100644
--- a/src/renderer/src/components/native-chat/use-structured-agent-session.ts
+++ b/src/renderer/src/components/native-chat/use-structured-agent-session.ts
@@ -28,6 +28,7 @@ import {
import { useStructuredAgentSessionHold } from './use-structured-agent-session-hold'
import { useStructuredAgentSessionRead } from './use-structured-agent-session-read'
import { projectStructuredAgentSessionMessages } from './structured-agent-session-message-projection'
+import { selectStructuredAgentTurnActivity } from './native-chat-turn-activity'
export type StructuredPromptItem = AgentJournalRenderItem & {
body: Extract
@@ -140,6 +141,10 @@ export function useStructuredAgentSession(args: {
// on the frame that opens each one, so re-read the options as a turn changes
// rather than leaving the last write unconfirmed for the life of the session.
const turnId = activeStructuredAgentSessionTurnId(state.items)
+ const turnActivity = useMemo(
+ () => selectStructuredAgentTurnActivity(state.items, turnId),
+ [state.items, turnId]
+ )
const isMonitoringBackgroundTasks =
turnId === null && state.backgroundTasks?.state === 'monitoring'
@@ -243,6 +248,7 @@ export function useStructuredAgentSession(args: {
send: outboxController.send,
retry: outboxController.retry,
isWorking: turnId !== null,
+ turnActivity,
isMonitoringBackgroundTasks,
backgroundTasks: state.backgroundTasks?.tasks ?? [],
supportsBackgroundTaskStop: state.backgroundTasks?.supportsTaskStop === true,
From 06a607a1d71207e40694641a2244c8a4c64e83ea Mon Sep 17 00:00:00 2001
From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com>
Date: Sun, 6 Sep 2026 14:34:03 -0400
Subject: [PATCH 19/32] feat(orchestration): make multi-agent workflows durable
(#16904)
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
| | Files | Added | Deleted | Net |
| :--- | ---: | ---: | ---: | ---: |
| Test | 225 | $\color{#1a7f37}{\Huge{\mathbf{+}}}$21666 | $\color{#cf222e}{\Huge{\mathbf{−}}}$2820 | $\color{#1a7f37}{\Huge{\mathbf{+}}}$18846 |
| Prod | 348 | $\color{#1a7f37}{\Huge{\mathbf{+}}}$17107 | $\color{#cf222e}{\Huge{\mathbf{−}}}$4706 | $\color{#1a7f37}{\Huge{\mathbf{+}}}$12401 |
## ELI5
Orca now treats orchestration like a durable control plane instead of inferring success from terminal keystrokes. Agents can tell whether a prompt was accepted or a turn started, replay an ambiguous request without sending twice, and recover coordinator mail after a crash. Completed workers can be inspected, released, or retained, and their panes no longer auto-resume as if the work were still running.
## What changed
- **Run receipts** from `run-create/use/current/show/list` are the row without routing plumbing (`home_database`, `coordinator_pane_key`) and without the duplicate `binding` object.
- **`terminal send` receipts are honest and idempotent.** `input_accepted` and `turn_started` are the only stages; `--wait-submit` observes without resending; `--retry-request ` replays the exact request against the same process incarnation. A transport timeout keeps the retry ID; only a different runtime answering strips it. Value-less or non-UUID `--retry-request` is rejected on the CLI and the SSH shim.
- **Mailbox delivery is committed before wakeup.** Pointer writes are staged in the DB before any PTY byte, replayed once after restart, and never emit a naked Enter. The watermark that parks concurrent deliveries is released with the DB reservation. Restart rescans pointer-pending and `dispatch:` mailboxes.
- **Lifecycle is a guarded transition graph** (`lifecycle-transition.ts`) with a table-driven test over every caller edge. Task reopen/overturn stays in the public contract. A PTY exit during `worker-stop` is the stop succeeding, not a failure.
- **Worker lifecycle CLI:** `worker-start` (`--spec` creates Task + attempt in one call), `worker-show`, `worker-read` (provider transcript first, bounded terminal fallback with a typed reason, local/WSL/SSH), `worker-stop`, `worker-abandon`, `worker-release`, `worker-retain`, `worker-list` (rowid-fenced pagination, fleet liveness, `attention`, literal `nextAction`).
- **Release is an explicit ownership table** (`decideWorkerTerminalRelease`): only an `owned` resource can be settled, the archive is mandatory where reachable, and an owner whose process is proven exited can always get out of `retained` via `archive_status: unavailable`. User-taken-over, external, and transferred panes stay retained.
- **Settled-worker resume fence** (folds in #17651): a settled dispatch whose pane is still open is fenced at settlement, on stop/abandon/exit, and at startup; lifted on release, retain, takeover, and pane reuse.
- **Liveness is `live` / `unverifiable` / `exited` only**, from execution-host evidence. Fleet projection reads the evidence clock, not the relay delivery clock. A host-certified exit outranks the worker's settled state. `unverifiable` never authorizes stop, abandon, retry, or release, in code or in the guide.
- **Federation:** structured reads negotiate by `method_not_found` so every shipped host keeps transcript-first output; exited remote workers are closed before being reported closed; epoch fencing holds across peer restart, downgrade, and pairing rotation; no per-second forced capability probe.
- **Schema v35:** repairs databases stamped v34 by the pre-fix branch (mailbox_handle default, index predicates), drops the write-only `lifecycle_transition_receipts` ledger and five never-read v31 identity columns.
- **Schema v36:** `dispatch:` mailboxes get a real consumer generation on `dispatch_contexts` and `remote_dispatch_attachments`, bumped and fenced in the same transaction on every re-attach (manual inject, worker-start, federated attach). A stale worker whose Dispatch moved to another process now gets `consumer_fenced` instead of silently acking the new worker's Delivery. Run mailboxes already worked this way.
- **Schema v37:** `dispatch_contexts` records its creator (`creator_handle`, `creator_pane_key`), so a coordinator's context-only self-dispatch is bookkeeping rather than a nesting parent; before this, one self-dispatch made every later `worker-start` from that coordinator fail the depth cap. Pre-v37 rows keep counting (fails closed).
- **Dispatch-mailbox ownership is checked, not inferred.** A `check` from a process whose pane no longer holds the Dispatch, or whose last Attempt was abandoned/failed and moved to another terminal, gets `consumer_fenced` instead of an empty inbox that reads as "no mail yet". `--peek`/`--all` stay readable. A paneless caller still gets `stable_pane_required` with the rebind recovery.
- **Liveness certification is stricter:** a `process_exited` stage whose termination reason is `unknown` (a stop that was issued but never observed) projects `unverifiable`, not `exited`. Federated `worker-show` carries the execution host's verdict and host kind instead of a local guess. A live, ready worker with nothing pending has `nextAction: none` rather than pointing at the `worker-show` that produced it.
- **Wire:** `workerShow` keeps `dispatch.task_id` next to `taskId` for shipped CLIs. `ask --json` uses the standard `{ok, result}` envelope like every sibling verb.
- **Migration start-version detection** treats the two v32 recovery columns as versioned. Before this, every shipped database stamped below 32 resolved to the v6 floor and replayed the whole chain (the v23 backfill synthesized 68 phantom retained workers on a real v30 profile). Verified on a copy of a real 62 MB v30 profile: starts at 30, no row delta, integrity ok, 11 ms.
- **Skill guide** rewritten as a ≤200-line kernel plus seven references, to the outcome-first standard (Result / Done / Safe failure first, conditions not case lists, one done bar, references loaded at the point of use). The canonical loop uses `worker-start --spec`, names `worker-list` for completion accounting, documents `--retry-request` / `request-show` / `--wait-submit`, and requires positive evidence before any stall action. The other seven guides get the same treatment in #18724, split out so this PR stays orchestration-only.
- **`rpc/methods/orchestration-*`** (126 flat files) regrouped into `orchestration/{worker,federation,messaging,runs,gates}/`.
## Why
User reports showed the same boundary failures: false `agent_prompt_stalled` causing duplicate sends (#15180), coordinators unable to trust screen scrapes, cold-parked terminals receiving a pointer without the submit, settled workers accumulating as live tabs and auto-resuming after restart, and no way to tell a stalled worker from a working one.
## Linked issues
Fixes #15180. Fixes #17935 (orchestration skill description is 866 characters; a guard now caps every bundled skill at 1,024). Supersedes #17651 (fence folded in). Advances #16660, #16522, #14907, #13047.
## Review record
This PR was reviewed adversarially after revival: eight independent lenses (lifecycle, mailbox, send, worker, federation, transcript, complexity, live ergonomics), each required to prove findings with a failing test. That produced 16 proven blockers, all fixed with red-then-green regression tests, followed by two re-review rounds and a third fix wave that caught 3 regressions introduced by the fixes and 7 fixes that missed their target; all closed. A final pass (five lenses incl. a live built-runtime smoke, then a re-review of the fix wave) found and fixed seven more, chiefly the stale-worker mailbox steal, the self-dispatch depth wedge, and the unproven-exit certification. Three independent Codex (gpt-6-astra) passes followed: the first found nothing new, the second found and fixed 3 defects (task-status reachability, WSL-local host classification, peer-capability epoch), the third found and fixed 6 (production PTY controller never installed settled writes, ambiguous in-flight pointer failures allowed duplicate replay, SSH/relay deadlines cut off a valid `--wait-submit`, stop-vs-exit race during inspection, and two release-recovery paths for vanished or exited terminals). The full record (findings, proof tests, triage, declines with reasons) is archived outside the repo.
**Rework after the live smoke.** A first live cross-host run on the shipped adhoc build (this Mac, a paired Windows host on the same build, a paired Mac on 1.4.195, and an SSH host) found a P1: a running local worker read `unverifiable`/`missing_status` because the fleet snapshot rows lacked the terminal handle the matcher keyed on. A 59-row failure table over every bug fixed during review showed the same two classes recurring: a fact dropped in transit through optional fields, and two authorities for one fact. Two blind designs (Opus, Codex) converged on the same mechanisms, and the scoped tranches landed here with red-then-green seam tests from the real producer to the real consumer, faults injected only at the transport or hook-ingest boundary:
- **Settlement (data-loss class):** one three-valued `WriteSettlement` (`accepted | refused{reason} | unverifiable{reason, bytesHandedToTransport}`) from the SSH multiplexer through daemon client, providers, controller, to pointer staging. No boolean, no rejection-as-third-state. The two silent degrades that fabricated a handoff are deleted; a provider that cannot settle refuses before any effect. Pointer text and Enter share the contract; a partial flush is `unverifiable`, never `refused`.
- **Evidence identity (false-liveness class):** fleet agent-status evidence is a tagged union (`binding: worker | pane | unresolved{reason}`, `clock: observed | delivery`) minted once at ingest, so a hook row captured on one process incarnation can never bind to a later dispatch on the same pane. The matcher's `!worker.paneKey ||` defaults are gone. One host-scope parser replaces two.
- **Small pre-merge items:** `capability_unsupported` from an old peer is no longer relabelled `host_unavailable`; a producer census test asserts every agent-status consumer path projects a pane-only hook row as `live`.
Two ergonomics defects the second live run surfaced on a real database are fixed here too: a pre-v3 dispatch already marked `completed` projected as `outcome_unknown` / `requiresAction: true` forever (three copies of the outcome ladder disagreed on legacy rows; now one resolver, legacy `completed` reads `succeeded` with nothing to act on, legacy `failed` stays actionable on the failure), and an unscoped `worker-list` enumerated the entire database (now defaults to the Run bound to the calling terminal, `--run` overrides, and the receipt's additive `scope` field says which).
A third live round on the shipped adhoc build of `b082443e1f` (same four hosts) plus an unscripted run in the user's own prompt style (a plain Claude Code shell, `/orchestration`, three workers, zero errors, bound-Run default confirmed) found two more branch defects, fixed with red-then-green tests: a worker freshly started on a paired server projected `unverifiable`/`host_indeterminate` with `requiresAction` for ~3 minutes, including after its own `worker_done`, because the host's federation observation returned `missing_liveness_verdict` for any PTY the liveness register had not yet swept (the host now reads a connected pane it owns locally as `live`; disconnected or SSH-scoped panes stay `unverifiable`); and six pre-v3 completed rows still carried an `input` category because settling through the task-status path or `failDispatch` never closed the Dispatch's pending question threads (both paths close them now, and schema v38 closes threads already pending on settled rows). The guide's `worker-start` examples now show `--model sonnet`, since an omitted model inherits the launcher's default.
A Codex adversarial pass on the tranche diff found one real design hole (identity minted at read time instead of ingest, now closed) and two daemon settlement paths that threw instead of settling (fixed). Two `@ts-nocheck` runtime mixins on these paths were extracted into checked modules; the repo-wide `@ts-nocheck` count is unchanged at 171.
Deletions during review: ~1,900 lines (write-only ledger, unread columns, dead v1 archive path, test harnesses shipped in prod, duplicated liveness and state-machine copies, self-capability checks that were compile-time true).
## Testing
- `pnpm typecheck:tsc:node|cli|web` clean
- `pnpm run check:code-quality:changed` 0 findings; `check:react-doctor:changed` 0
- `pnpm verify:bundled-skill-guides`, `verify:skill-bundle-manifest`
- full `pnpm test` on the integrated head: 72,332 pass / 292 skipped; the only failures were three non-PR files (two zsh live-shell suites hit a node-pty spawn-helper ENOENT while a concurrent native rebuild ran, 44/44 in isolation; `release-checkout.unit.test.ts` is a known 30 s load timeout that passes in isolation on `origin/main` too).
- CI on 70b4811267 (rerun, pre-Codex): the only reds are five SSH e2e specs plus `terminal-send-agent-prompt-submit:198`, each shown failing identically on main (main's E2E workflow is red on its last 40 runs). The terminal-send spec is root-caused and fixed separately in #18707. The Windows hook-service flake (#17721) and the federation load flake did not recur.
- Skills: `pnpm exec vitest run` over the skill gate files plus `src/cli`, `config/scripts`, `src/main/skills` pass; live smoke on the built CLI of `skills get orchestration` and `--full` (7 references).
- live headless runtime (`orca-dev serve`, isolated profile): canonical loop, stop, release, archive read, retry rejection, stale-handle check, SIGKILL-and-replay all verified with receipts
- Live cross-host smoke on the shipped adhoc build of `0d465e7931` (this Mac and a paired Windows host on the build, a paired Mac left on 1.4.195, an SSH host): local, paired-new, paired-old and SSH loops all settle; running workers read `live` on every host and `exited` after release; the old peer reads `capability_unsupported` and refuses release honestly. Injected 10 s relay stall with a send in flight: delivered exactly once after recovery, zero duplicates. Every liveness field across 104 receipts is only `live` / `unverifiable` / `exited`.
- Final live cross-host smoke on the shipped adhoc build of `b082443e1f` (same hosts): every loop settles; 942 of 948 legacy completed rows read settled with `requiresAction: false` before the question-thread fix and all of them after; `worker-list` scope reads `bound` / `flag` / `all` correctly; 122 JSON receipts carry only `live` / `unverifiable` / `exited`. Unscripted prompt-style run: clean.
- Confirmation smoke on the shipped adhoc build of `2da076d4e9` (this Mac and the paired Windows host, both updated): a freshly started Windows worker reads `live` on the first fleet poll and on all 20 that follow, with no `host_indeterminate` at any point, and `exited` after release; all 948 legacy completed rows read `requiresAction: false` with `nextAction: none` after schema v38; every verdict across 60 receipts is `live` / `unverifiable` / `exited`.
- Not physically exercised: WSL hosts, the renderer notification bell (headless has no renderer), same-session fence via a real pane close (renderer-only state), restart mid-delivery on a real app (covered by e2e only).
## Notes
- Remote-wire additions are optional fields or `method_not_found`-negotiated methods; one new Electron-only IPC channel (`agentStatus:legacyWorkerTerminalResumeFence`) never crosses the wire.
- SSH contact loss remains `unverifiable`; the execution host stays authoritative.
- Intentional wire projection change: an SSH host scope with an empty `targetId` now projects host id `ssh` instead of an empty string (remote-wire-compatibility rule 3, old clients decode the same field). A fleet pane key without a terminal handle is now `unidentifiable` rather than matched by pane key alone.
- Found live but pre-existing on main, filed separately: a relay daemon-start collision during transport loss rewrites the endpoint credential and wedges the surviving relay (host needs a manual kill); `terminal create` on a reconnecting SSH host reports an opaque `No PTY provider for connection`; `terminal list` reports `orphaned:false` and `terminal close` reports `ptyKilled:true` for a pane whose relay is gone (orchestration's own projection reads `unverifiable` correctly at the same moment).
- Downgrade after this PR is not a supported path: main opens a v37 database and early-returns (its inserts still work against the v36/v37 defaulted columns), but its one-outstanding-Delivery-per-Run index is a no-op against the branch's mailbox-scoped index of the same name.
- Known follow-ups (not blockers): `worker-list` materializes every dispatch row per call; a positive "agent absent" signal distinct from PTY liveness is a product decision left open (a headless fake agent never reaches `live`, so its `nextAction` stays `inspect`); a context-only self-dispatch still lists as `role: worker` in `worker-list`; `dispatch` task-not-found / task-not-ready / inject-rejected still surface as `runtime_error`; task and inbox receipts still expose raw row columns. Deferred skill product decisions live on #18724.
---
config/reliability-gates.jsonc | 179 ++--
.../scripts/generate-bundled-skill-guides.mjs | 113 ++-
.../generate-bundled-skill-guides.test.mjs | 73 +-
.../scripts/orca-cli-skill-guidance.test.mjs | 11 +-
...hestration-guide-command-contract.test.mjs | 38 +
.../orchestration-skill-guidance.test.mjs | 787 +++++++++-------
docs/site/content/docs/cli/orchestration.mdx | 2 +-
docs/site/content/docs/cli/reference.mdx | 2 +
docs/site/content/docs/cli/skills.mdx | 4 +
resources/skills/current-manifest.json | 12 +-
resources/skills/snapshot-registry.json | 12 +-
skill-guides/orca-cli.md | 7 +-
skill-guides/orchestration.md | 612 ++++--------
.../references/coordinator-loop.md | 58 ++
.../references/legacy-contract-migration.md | 87 ++
.../references/low-level-topology.md | 25 +
.../references/messaging-and-gates.md | 63 ++
.../references/placement-and-remote.md | 90 ++
.../references/recovery-and-cleanup.md | 159 ++++
.../references/worker-contract.md | 77 ++
skill-stubs/orchestration.md | 13 +-
skills/orchestration/SKILL.md | 39 +-
src/cli/args.test.ts | 20 +
src/cli/bundled-skill-guides.ts | 63 +-
src/cli/cli-error.ts | 74 +-
src/cli/command-suggestion.ts | 12 +-
src/cli/flags.ts | 19 +
src/cli/format-recovery.test.ts | 96 +-
src/cli/format.ts | 5 +-
src/cli/handlers/bundled-skill-guide-table.ts | 57 ++
.../orchestration-check-identity.test.ts | 24 +-
.../orchestration-lifecycle-rejection.test.ts | 36 +
.../orchestration-module-boundaries.test.ts | 16 +-
.../orchestration-task-list-brief.test.ts | 57 ++
.../orchestration-timeout-cli.test.ts | 54 +-
.../handlers/orchestration-worker-cli.test.ts | 414 +++++++-
.../orchestration-worker-settlement.ts | 24 +-
src/cli/handlers/orchestration.test.ts | 80 +-
.../orchestration/mutation-request.ts | 4 +-
.../orchestration/question-handler.ts | 30 +-
.../orchestration/worker-launch-handler.ts | 35 +-
.../worker-list-run-scope.test.ts | 99 ++
.../orchestration/worker-list-run-scope.ts | 41 +
.../worker-observation-handlers.ts | 31 +-
.../orchestration/worker-output.test.ts | 290 ++++++
.../handlers/orchestration/worker-output.ts | 89 +-
.../orchestration/worker-terminal-handlers.ts | 93 +-
src/cli/handlers/skill-guide-get.ts | 108 +++
src/cli/handlers/skills.ts | 74 +-
src/cli/handlers/terminal-close.ts | 105 +++
src/cli/handlers/terminal-send.ts | 113 +++
src/cli/handlers/terminal.test.ts | 344 ++++++-
src/cli/handlers/terminal.ts | 125 +--
src/cli/help.ts | 12 +-
src/cli/index.test.ts | 56 ++
src/cli/index.ts | 6 +-
.../orchestration-mutation-recovery.test.ts | 16 +-
src/cli/orchestration-mutation-recovery.ts | 13 +
src/cli/retry-request-flag.test.ts | 148 +++
src/cli/retry-request-flag.ts | 23 +
src/cli/root-help-text-primary.ts | 4 +-
src/cli/root-help-text-secondary.ts | 4 +-
src/cli/runtime-client-deferral.test.ts | 10 +-
src/cli/runtime/client-recovery.test.ts | 277 +++++-
src/cli/runtime/client.ts | 104 +-
src/cli/runtime/runtime-remote-pairing.ts | 30 +
.../terminal-prompt-mutation-recovery.ts | 108 +++
src/cli/skills-command-flag-help.ts | 15 +
src/cli/skills-reference-selector.test.ts | 186 ++++
src/cli/skills.test.ts | 4 +-
src/cli/specs/core.ts | 9 +-
src/cli/specs/orchestration-worker-specs.ts | 16 +-
src/cli/specs/orchestration.test.ts | 14 +
src/cli/specs/orchestration.ts | 1 +
src/cli/specs/skills.test.ts | 20 +
src/cli/specs/skills.ts | 17 +-
src/cli/specs/terminal-send.ts | 24 +
src/cli/stdout-line.ts | 4 +
src/cli/terminal-format.test.ts | 114 ++-
src/cli/terminal-format.ts | 45 +-
src/cli/worktree-selector-recovery.ts | 55 ++
.../server-replay-evidence-clock.test.ts | 16 +
.../server/server-status-identity.ts | 3 +
src/main/daemon/client.test.ts | 15 +-
src/main/daemon/client.ts | 12 +-
.../daemon-client-notify-settlement.test.ts | 55 ++
.../daemon/daemon-client-notify-settlement.ts | 44 +-
.../daemon/daemon-pty-event-subscriptions.ts | 7 +-
src/main/daemon/daemon-pty-router.test.ts | 9 +-
src/main/daemon/daemon-pty-router.ts | 3 +-
src/main/daemon/daemon-pty-session-control.ts | 89 +-
src/main/daemon/daemon-pty-session-input.ts | 107 +++
...emon-pty-write-settlement-recovery.test.ts | 53 ++
.../degraded-daemon-pty-provider.test.ts | 13 +-
.../daemon/degraded-daemon-pty-provider.ts | 8 +-
src/main/ipc/agent-hooks.test.ts | 3 +-
src/main/ipc/agent-status-ipc-boundary.ts | 91 +-
.../pty-controller-ownership-routing.test.ts | 55 ++
src/main/ipc/pty/runtime/controller.ts | 2 +
src/main/ipc/pty/runtime/operations.ts | 42 +-
.../host-readable-transcript-path.test.ts | 65 ++
.../host-readable-transcript-path.ts | 24 +
...ession-file-resolver-wsl-scan-gate.test.ts | 3 +-
.../session-file-resolver-wsl.test.ts | 78 +-
src/main/native-chat/session-file-resolver.ts | 50 +-
src/main/providers/local-pty-provider.ts | 10 +
src/main/providers/provider-dispatch.test.ts | 2 +
src/main/providers/pty-provider-contract.ts | 6 +-
src/main/providers/settled-pty-write-stub.ts | 21 +
.../settled-pty-writer-census.test.ts | 94 ++
.../ssh-pty-provider-rpc-operations.ts | 3 +-
src/main/providers/ssh-pty-provider.ts | 3 +-
src/main/providers/ssh-pty-write.test.ts | 46 +-
src/main/providers/ssh-pty-write.ts | 30 +-
.../agent-prompt-receipt-correlation.test.ts | 68 ++
.../agent-prompt-request-correlation.test.ts | 85 ++
.../agent-prompt-request-correlation.ts | 219 +++++
.../agent-prompt-submission-runtime.test.ts | 130 ++-
...ent-prompt-submission-verification.test.ts | 45 +-
.../agent-prompt-submission-verification.ts | 68 +-
...gent-session-pty-write-enforcement.test.ts | 21 +-
.../agent-status-observed-pane-identity.ts | 65 ++
...e-adopt-terminal-orphans-from-inventory.ts | 10 +-
...untime-agent-prompt-request-correlation.ts | 139 +++
.../orca-runtime-apply-tracked-pty-title.ts | 6 +
...ca-runtime-controller-knows-pty-is-live.ts | 29 +-
...time-exact-worker-provider-session.test.ts | 57 ++
...me-get-orchestration-dispatch-authority.ts | 16 +-
...rca-runtime-get-pty-record-for-pane-key.ts | 14 +-
.../runtime/orca-runtime-get-runtime-id.ts | 7 +-
...a-runtime-get-terminal-interactive-wait.ts | 7 +
...-runtime-mark-pty-liveness-unverifiable.ts | 11 +-
.../orca-runtime-preserved-branch-cleanup.ts | 8 +-
...ime-record-agent-prompt-lifecycle-state.ts | 1 +
...refresh-floating-workspace-pty-liveness.ts | 2 +-
...ktree-records-with-controller-inventory.ts | 7 +-
...-authoritative-terminal-wait-permission.ts | 4 +-
src/main/runtime/orca-runtime-state-fields.ts | 6 +
.../orca-runtime-stop-requested-pty-ids.ts | 5 +-
...ca-runtime-subscribe-to-terminal-resize.ts | 21 +
.../runtime/orca-runtime-sync-window-graph.ts | 12 +
.../orca-runtime-test-fixtures.spec.ts | 167 +---
...untime-test-orchestration-messages.spec.ts | 343 +++++++
.../lineage-and-scan-cache-part-05.spec.ts | 20 +-
...creation-and-orchestration-part-02.spec.ts | 28 +-
...creation-and-orchestration-part-03.spec.ts | 24 +-
.../orchestration-attention-batching.spec.ts | 81 ++
.../terminal-handles-and-agent-status.spec.ts | 10 +
...erminal-output-and-worker-recovery.spec.ts | 15 +-
...runtime-write-orchestration-pointer-pty.ts | 79 +-
...rca-runtime-write-terminal-agent-prompt.ts | 102 +-
src/main/runtime/orca-runtime.test.ts | 1 +
...stration-dispatch-mailbox-delivery.test.ts | 217 +++++
...chestration-fleet-agent-status-snapshot.ts | 33 +
...chestration-mailbox-cold-park-idle.test.ts | 141 +++
...chestration-mailbox-crash-recovery.test.ts | 117 +++
...estration-mailbox-detached-routing.test.ts | 10 +-
...estration-mailbox-filtered-waiters.test.ts | 138 +++
...n-mailbox-notification-consistency.test.ts | 305 +++---
...ation-mailbox-notification-test-harness.ts | 39 +-
...ration-mailbox-pointer-cli-command.test.ts | 47 +
...chestration-mailbox-pty-write-gate.test.ts | 116 +++
...ation-mailbox-transport-settlement.test.ts | 157 +++-
...stration-message-delivery-identity.test.ts | 16 +-
...orchestration-messages-fake-parity.test.ts | 65 ++
...rchestration-structured-chat-lease.test.ts | 59 +-
.../__snapshots__/preamble.test.ts.snap | 34 +-
.../runtime/orchestration/cli-command.test.ts | 19 +
src/main/runtime/orchestration/cli-command.ts | 6 +-
.../context-only-dispatch-release.ts | 39 +-
.../coordinator-runtime-contract.ts | 12 +-
.../coordinator-task-dispatch.ts | 6 +-
.../db-task-dispatch-invariant.test.ts | 37 +
.../db-task-dispatch-lifecycle-guards.test.ts | 209 +++++
.../db-task-dispatch-races.test.ts | 94 +-
.../db-undelivered-mailboxes.test.ts | 34 +
src/main/runtime/orchestration/db.ts | 14 +
.../db/attach-orchestration-db-methods.ts | 12 +
.../db/attempt-observation-store.ts | 186 ++++
.../db/attempt-observation-types.ts | 109 +++
.../db/attempt-outcome-projection.test.ts | 442 +++++++++
.../db/attempt-outcome-projection.ts | 159 ++++
.../orchestration/db/contract-constants.ts | 4 +-
.../db/decision-gate-lifecycle.test.ts | 39 +
.../db/decision-gates/decision-gate-store.ts | 16 +-
.../dispatch-context/dispatch-capability.ts | 37 +-
.../dispatch-context/dispatch-completion.ts | 193 ++--
.../dispatch-context-store.ts | 13 +-
.../task-dispatch-reconciliation.ts | 27 +-
.../worker-report-settlement.ts | 219 +++--
.../orchestration/db/dispatch-depth.ts | 70 +-
.../dispatch-mailbox-consumer-fencing.test.ts | 210 +++++
.../orchestration/db/dispatch-row-writer.ts | 22 +-
...derated-dispatch-observation-fence.test.ts | 86 ++
.../federated-dispatch-observation-fence.ts | 108 +++
.../db/federation/federated-dispatch-store.ts | 36 +-
.../federation/remote-attachment-liveness.ts | 16 +
.../remote-dispatch-attachment-authority.ts | 136 ++-
...remote-dispatch-attachment-release.test.ts | 68 ++
.../remote-dispatch-attachment-release.ts | 88 ++
.../remote-dispatch-attachment-stop.ts | 5 +-
.../db/hot-path-statement-compilation.test.ts | 3 +-
.../db/lifecycle-transition-boundary.test.ts | 25 +
.../db/lifecycle-transition.test.ts | 57 ++
.../orchestration/db/lifecycle-transition.ts | 204 ++++
.../db/lifecycle-write-transaction-runner.ts | 22 +
.../messages/mailbox-pointer-enter-state.ts | 228 +++++
.../db/messages/message-inbox.ts | 23 +-
.../db/messages/message-insert.ts | 10 +-
.../db/messages/role-mailbox-delivery.ts | 219 +++++
.../mutation-receipt-store.ts | 36 +
.../db/orchestration-db-methods.ts | 14 +-
.../db/reset/orchestration-reset.ts | 2 +
.../orchestration/db/row-column-lists.test.ts | 4 +-
.../orchestration/db/row-column-lists.ts | 32 +-
.../orchestration/db/runs/run-delivery.ts | 163 +---
.../orchestration/db/runs/run-lookup.ts | 4 +-
.../db/schema/create-core-tables-sql.ts | 34 +-
.../db/schema/create-graph-tables-sql.ts | 35 +-
.../migrate-mailbox-pointer-enter-v33.ts | 22 +
.../migrate-role-mailbox-delivery-v34.ts | 53 ++
.../db/schema/migrate-v13-v30.ts | 33 +-
.../orchestration/db/schema/migrate-v35.ts | 121 +++
.../orchestration/db/schema/migrate-v36.ts | 18 +
.../orchestration/db/schema/migrate-v37.ts | 18 +
.../orchestration/db/schema/migrate-v38.ts | 21 +
.../orchestration/db/schema/migrate.ts | 12 +
.../db/schema/schema-column-probes.ts | 15 +
.../db/tasks/task-status-transition.ts | 167 ++--
.../orchestration/db/tasks/task-store.ts | 8 +-
.../federated-worker-start-reconcile.ts | 183 ++--
.../worker-dispatch-abandon.ts | 40 +-
.../worker-dispatch-authority.ts | 13 +-
.../worker-dispatch-outcome.ts | 141 ++-
.../worker-dispatch/worker-dispatch-stage.ts | 74 +-
.../worker-dispatch/worker-dispatch-start.ts | 68 +-
.../worker-dispatch/worker-dispatch-stop.ts | 173 ++--
.../worker-terminal-recovery.ts | 82 +-
.../failed-start-terminal-adoption.ts | 68 ++
.../worker-terminal-attention-query.ts | 137 +++
.../worker-terminal-inventory-counts.ts | 111 +++
.../worker-terminal-listing.ts | 287 ++++--
.../worker-terminal-release.ts | 51 +-
.../worker-terminal-resource-store.ts | 31 +-
.../worker-terminal-transfer.ts | 11 +-
.../worker-terminal-user-takeover.ts | 63 ++
...atch-consumer-generation-migration.test.ts | 99 ++
...ispatch-creator-identity-migration.test.ts | 76 ++
.../orchestration/environment-transport.ts | 10 +-
.../failed-start-terminal-adoption.test.ts | 157 ++++
.../federation-ack-checkpoints.test.ts | 61 ++
.../federation-sync-capability.ts | 32 +
.../orchestration/federation-sync-message.ts | 104 ++
.../federation-sync-test-harness.ts | 109 +++
.../orchestration/federation-sync.test.ts | 394 +++++---
.../runtime/orchestration/federation-sync.ts | 245 +++--
.../runtime/orchestration/formatter.test.ts | 9 +
src/main/runtime/orchestration/formatter.ts | 9 +-
.../lifecycle-caller-edges.test.ts | 143 +++
.../lifecycle-reconciliation.test.ts | 73 ++
.../orchestration/lifecycle-reconciliation.ts | 4 +-
.../runtime/orchestration/mailbox-owner.ts | 8 +-
.../mailbox-pointer-delivery-contract.ts | 36 +
.../orchestration/mailbox-pointer-delivery.ts | 240 ++---
.../mailbox-pointer-eligibility.ts | 5 +-
.../mailbox-pointer-pty-write.ts | 86 ++
.../orchestration/mailbox-pointer-resume.ts | 100 ++
.../mailbox-pointer-stage.test.ts | 182 ++++
.../orchestration/mailbox-pointer-stage.ts | 200 ++++
.../orchestration/mailbox-pointer-state.ts | 41 +-
.../mailbox-pointer-submit.test.ts | 491 ++++++++++
.../orchestration/mailbox-pointer-submit.ts | 102 +-
.../message-batch-atomicity.test.ts | 28 +
...ation-all-start-versions-migration.test.ts | 43 +
.../orchestration-legacy-storage-db.test.ts | 6 +-
...chestration-legacy-storage-test-fixture.ts | 11 +-
...on-legacy-worker-terminal-recovery.test.ts | 18 +-
...tration-legacy-worker-terminal-recovery.ts | 30 +-
...rchestration-peer-capability-cache.test.ts | 373 ++++++++
.../orchestration-peer-capability-cache.ts | 285 ++++++
...chestration-run-list-compatibility.test.ts | 2 +-
.../orchestration-schema-version-skew.ts | 70 +-
...ion-settled-worker-resume-fence-db.test.ts | 124 +++
...chestration-version-skew-migration.test.ts | 389 ++++++++
.../orchestration-worker-dispatch-db.test.ts | 92 +-
.../runtime/orchestration/preamble.test.ts | 52 +-
src/main/runtime/orchestration/preamble.ts | 37 +-
.../r1-identity-migration.test.ts | 129 +++
...settled-question-threads-migration.test.ts | 67 ++
src/main/runtime/orchestration/types.ts | 15 +
.../worker-attention-context.test.ts | 122 +++
.../orchestration/worker-attention-context.ts | 60 ++
.../worker-output-archive.test.ts | 202 ++++
.../orchestration/worker-output-archive.ts | 98 +-
.../worker-output-cursor.test.ts | 19 +-
.../orchestration/worker-output-cursor.ts | 35 +-
.../worker-provider-session.test.ts | 47 +
.../orchestration/worker-provider-session.ts | 41 +-
.../worker-report-observation.ts | 13 +
...start-unobserved-prompt-settlement.test.ts | 34 +
.../worker-terminal-ownership.ts | 38 +-
.../worker-terminal-process-liveness.ts | 39 +-
.../worker-terminal-release-reconciliation.ts | 26 +-
.../worker-transcript-local-checkpoint.ts | 70 ++
.../worker-transcript-local-read.ts | 284 ++++++
.../worker-transcript-payload.test.ts | 40 +
.../worker-transcript-payload.ts | 91 +-
.../worker-transcript-read.test.ts | 53 +-
.../orchestration/worker-transcript-read.ts | 250 ++---
.../worker-transcript-remote-range-read.ts | 129 +++
.../worker-transcript-remote-read.test.ts | 370 ++++++++
.../worker-transcript-remote-read.ts | 269 ++++++
.../worker-transcript-source-identity.ts | 90 ++
.../pty-inventory-liveness-verdict.test.ts | 28 +-
src/main/runtime/rpc/core.ts | 6 +
.../rpc/dispatcher-caller-fingerprint.ts | 4 +-
.../rpc/dispatcher-unary-method-invocation.ts | 89 ++
src/main/runtime/rpc/dispatcher.ts | 79 +-
src/main/runtime/rpc/errors.test.ts | 26 +
src/main/runtime/rpc/errors.ts | 4 +
...ration-federation-liveness-verdict.test.ts | 183 ----
.../orchestration-federation-methods.ts | 10 -
.../orchestration-federation-output.test.ts | 312 ------
.../orchestration-send-point-to-point.ts | 188 ----
.../methods/orchestration-worker-methods.ts | 12 -
.../orchestration-worker-observation.ts | 156 ---
.../orchestration-worker-release.test.ts | 886 ------------------
.../orchestration-worker-start-schema.ts | 32 -
.../rpc/methods/orchestration-worker-stop.ts | 221 -----
.../rpc/methods/orchestration-workers.ts | 302 ------
src/main/runtime/rpc/methods/orchestration.ts | 25 +-
.../cli-runtime-boundary.test.ts} | 12 +-
.../federated-attach-receipt.test.ts} | 2 +-
.../federation/federated-attach-receipt.ts} | 2 +-
.../federation/federated-fleet-host-groups.ts | 47 +
.../federated-fleet-snapshot.test.ts | 474 ++++++++++
.../federation/federated-fleet-snapshot.ts | 267 ++++++
.../federated-message-targeting.test.ts} | 12 +-
.../federated-release-safety.test.ts | 202 ++++
.../federated-transport-safety.test.ts | 328 +++++++
.../federation/federated-worker-read.ts | 113 +++
.../federated-worker-release-host.ts | 312 ++++++
.../federation/federated-worker-release.ts | 198 ++++
.../federation/federated-worker-show.ts | 158 ++++
.../federated-worker-start-receipt.test.ts} | 25 +-
.../federated-worker-start-receipts.ts} | 27 +-
.../federation/federated-worker-start.ts} | 76 +-
.../federation-agent-launch.test.ts} | 6 +-
.../federation-attachment-observation.ts | 88 ++
.../federation-control-mail.test.ts} | 51 +-
.../federation/federation-control.ts} | 152 +--
.../federation/federation-effects.test.ts} | 2 +-
.../federation/federation-effects.ts} | 0
.../federation-folder-placement.test.ts} | 6 +-
.../federation-lifecycle-settlement.test.ts} | 18 +-
.../federation-liveness-verdict.test.ts | 415 ++++++++
.../federation/federation-methods.ts | 10 +
.../federation/federation-output.test.ts | 825 ++++++++++++++++
.../federation/federation-relay.ts} | 12 +-
...release-recovery-scenarios.test-support.ts | 266 ++++++
.../federation-request.test-support.ts} | 4 +-
.../federation-runtime.test-support.ts | 63 ++
.../federation/federation-setup.test.ts} | 10 +-
.../federation/federation-setup.ts} | 11 +-
.../federation-start-prompt-budget.test.ts | 62 ++
.../federation/federation-start-receipt.ts} | 8 +-
.../federation/federation-start-schema.ts} | 4 +-
.../federation/federation.test.ts} | 122 +--
.../federation/federation.ts} | 46 +-
.../gates/gate-run-authorization.test.ts} | 2 +-
.../gates/gates.test.ts} | 6 +-
.../gates/gates.ts} | 20 +-
.../messaging/ask-methods.ts} | 14 +-
.../messaging/ask-remote.ts} | 8 +-
.../messaging/ask.test.ts} | 12 +-
.../messaging/check-direct.ts} | 19 +-
.../messaging/check-methods.ts} | 41 +-
.../messaging/check-run.ts} | 27 +-
.../check-superseded-terminal.test.ts | 135 +++
.../check-worker-consumer-fencing.test.ts | 230 +++++
.../messaging/check-worker.ts} | 146 ++-
.../messaging/check.test.ts} | 88 +-
.../messaging/dispatch-mailbox-fence.ts | 41 +
.../messaging/mailbox-message-receipt.ts | 28 +
.../messaging/message-methods.ts} | 63 +-
.../messaging/mutation-replay-nudge.ts | 65 ++
.../messaging/recipient-routing.test.ts} | 18 +-
.../messaging/recipient-routing.ts} | 8 +-
.../messaging/send-control-mail.ts} | 26 +-
.../send-dispatch-authority.test.ts} | 14 +-
.../messaging/send-group.ts} | 30 +-
.../messaging/send-invalid-type.test.ts} | 10 +-
.../messaging/send-methods.ts} | 51 +-
.../messaging/send-point-to-point.ts | 238 +++++
.../messaging/send-receipt-plumbing.test.ts | 109 +++
.../messaging/send-remote.ts} | 18 +-
.../messaging/send.test.ts} | 18 +-
.../messaging/settled-dispatch-mail.test.ts} | 8 +-
.../routing.ts} | 12 +-
.../rpc-test-harness.ts} | 8 +-
.../runs/dispatch-creator.ts} | 4 +-
.../runs/dispatch-methods.ts} | 69 +-
.../runs/migration-behavior.test.ts} | 20 +-
.../runs/mutation-request-show.ts} | 6 +-
.../runs/reset-methods.ts} | 4 +-
.../orchestration/runs/run-receipt.test.ts | 61 ++
.../methods/orchestration/runs/run-receipt.ts | 14 +
.../runs/run-scope.ts} | 10 +-
.../runs/runs.test.ts} | 31 +-
.../runs/runs.ts} | 28 +-
.../runs/tasks-dispatch.test.ts} | 26 +-
.../schemas.ts} | 15 +-
.../agent-status-producer-census.test.ts | 390 ++++++++
.../worker/composed-workers.test.ts} | 17 +-
.../context-only-dispatch-retry.test.ts | 58 ++
.../failed-start-residual-terminal.test.ts | 185 ++++
.../worker/failed-start-residual-terminal.ts | 53 ++
.../fleet-status-observed-identity.test.ts | 285 ++++++
.../fleet-status-terminal-identity.test.ts | 250 +++++
.../worker/folder-worktree-placement.ts} | 6 +-
.../worker/legacy-dispatch-projection.test.ts | 122 +++
.../worker/local-worker-start.ts | 293 ++++++
.../manual-dispatch-observation.test.ts} | 38 +-
.../worker/manual-dispatch-release.test.ts} | 8 +-
.../self-dispatch-nesting-depth.test.ts | 71 ++
.../worker/task-deps-argument.ts | 20 +
.../worker/worker-archive-read.ts} | 146 ++-
.../worker/worker-control.ts} | 183 +---
.../worker/worker-interactive-wait.test.ts} | 10 +-
.../worker/worker-launch-preferences.test.ts} | 39 +-
.../worker/worker-launch-preferences.ts} | 12 +-
.../worker/worker-legacy-federated-read.ts} | 17 +-
.../worker/worker-list-cursor.ts | 80 ++
.../worker/worker-list-method.ts | 305 ++++++
.../worker/worker-list-pagination.test.ts | 643 +++++++++++++
.../worker/worker-list-projection.ts | 79 ++
.../worker/worker-list-run-scope-rpc.test.ts | 62 ++
.../worker/worker-list-snapshot-store.ts | 157 ++++
.../orchestration/worker/worker-methods.ts | 12 +
.../worker/worker-observation.test.ts | 149 +++
.../worker/worker-observation.ts | 281 ++++++
.../worker/worker-output.test.ts} | 116 ++-
.../worker/worker-output.ts} | 60 +-
.../worker/worker-read-projection.test.ts | 38 +
.../worker/worker-release-archive.test.ts | 247 +++++
.../worker/worker-release-close-error.ts | 33 +
.../worker/worker-release-completion.ts} | 221 ++---
.../worker/worker-release-inventory.test.ts | 194 ++++
.../worker-release-liveness-verdict.test.ts} | 62 +-
.../worker-release-ownership-guard.test.ts | 104 ++
.../worker/worker-release-recovery.test.ts} | 144 ++-
.../worker/worker-release-schemas.ts | 24 +
.../worker/worker-release.test-support.ts | 202 ++++
.../worker/worker-release.test.ts | 430 +++++++++
.../worker/worker-release.ts} | 98 +-
.../worker/worker-setup-gate.ts} | 4 +-
.../worker/worker-start-budgets.test.ts} | 6 +-
.../worker/worker-start-budgets.ts} | 2 +-
...rker-start-outcome-classification.test.ts} | 2 +-
.../worker/worker-start-prompt-budget.test.ts | 45 +
.../worker/worker-start-prompt-budget.ts | 20 +
.../worker-start-prompt-contract.test.ts} | 100 +-
.../worker/worker-start-receipt.ts} | 31 +-
.../worker/worker-start-schema.ts | 63 ++
.../worker-start-terminal-target.test.ts | 159 ++++
.../worker/worker-start-validation.ts} | 14 +-
.../worker/worker-stop-capability.test.ts} | 8 +-
.../worker/worker-stop-exit-race.test.ts | 105 +++
.../worker-stop-liveness-verdict.test.ts} | 6 +-
.../orchestration/worker/worker-stop.ts | 271 ++++++
.../worker/worker-terminal-release-lease.ts | 26 +
.../worker-terminal-resource-presentation.ts | 51 +
.../worker/worker-topology.ts} | 8 +-
.../worker/workers-new-worktree.test.ts} | 12 +-
.../worker/workers-recovery.test.ts} | 99 +-
.../methods/orchestration/worker/workers.ts | 69 ++
.../settled-worker-resume-fence-sweep.ts | 45 +
.../terminal/terminal-prompt-receipt.ts | 68 ++
.../methods/terminal/terminal-send-method.ts | 60 +-
.../rpc/methods/terminal/unary-schemas.ts | 2 +
...ion-commit-notify-characterization.test.ts | 481 ++++++++++
...ation-current-authority-precedence.test.ts | 32 +
...on-legacy-compatibility-dispatcher.test.ts | 8 +-
...-legacy-takeover-current-authority.test.ts | 9 +-
...tration-legacy-takeover-dispatcher.test.ts | 69 +-
.../orchestration-mutation-executor.test.ts | 289 ++++++
.../rpc/orchestration-mutation-executor.ts | 282 ++++--
.../rpc/orchestration-mutation-receipt.ts | 225 +++++
...stration-runtime-update-settlement.test.ts | 9 +-
.../terminal-prompt-delivery-receipt.test.ts | 431 +++++++++
.../runtime-agent-orchestration-projection.ts | 61 +-
...cy-worker-terminal-recovery-persistence.ts | 80 +-
...egacy-worker-terminal-resume-fence.test.ts | 286 ++++++
src/main/runtime/runtime-notifier-contract.ts | 2 +
.../runtime-orchestration-federation.ts | 15 +-
.../runtime-pty-controller-contract.ts | 4 +-
.../runtime-rpc-long-poll-transport.test.ts | 14 +
...ntime-rpc-websocket-long-poll-caps.test.ts | 4 +
.../runtime/runtime-terminal-contracts.ts | 11 +
.../terminal-send-stale-leaf-liveness.test.ts | 188 +++-
src/main/sqlite/sync-database.test.ts | 10 +
src/main/sqlite/sync-database.ts | 4 +
...ssh-channel-multiplexer-settlement.test.ts | 15 +-
src/main/ssh/ssh-channel-multiplexer.test.ts | 3 +-
src/main/ssh/ssh-channel-multiplexer.ts | 8 +-
src/main/ssh/ssh-host-cli-deadline.ts | 43 +
.../ssh-multiplexer-transport-writer.test.ts | 40 +-
.../ssh/ssh-multiplexer-transport-writer.ts | 84 +-
.../ssh-relay-session-data-delivery.test.ts | 6 +-
src/main/ssh/ssh-relay-session.ts | 9 +-
src/main/ssh/ssh-remote-cli-args.ts | 47 +-
.../ssh-remote-cli-host-passthrough.test.ts | 21 +
.../ssh/ssh-remote-cli-host-passthrough.ts | 53 +-
src/main/ssh/ssh-remote-orca-cli.ts | 3 +-
...remote-orchestration-compatibility.test.ts | 113 ++-
.../startup/main-process-runtime-service.ts | 20 +-
src/main/window/runtime-window-lifecycle.ts | 2 +
src/preload/api/agent-status-api.ts | 4 +
src/preload/api/agent-status-bridge.ts | 10 +
src/relay/remote-cli-timeout.ts | 43 +-
.../ipc-events/agent-status-listeners.ts | 8 +
.../src/hooks/useIpcEvents-lifecycle.test.ts | 2 +
...ctivation-emptied-workspace-reseed.test.ts | 52 +
src/renderer/src/lib/worktree-activation.ts | 22 +-
.../worktree-agent-activation-seam.test.ts | 19 +-
.../lib/worktree-initial-terminal-seeding.ts | 23 +
.../sync-runtime-graph-parked-leaf.test.ts | 2 +-
.../sync-runtime-graph/graph-publication.ts | 1 +
...agent-status-open-tab-resume-fence.test.ts | 60 ++
.../agent-status-orchestration-context.ts | 4 +-
.../slices/agent-status-recovery-actions.ts | 17 +-
...agent-status-runtime-orchestration.test.ts | 47 +
.../slices/agent-status-sleeping-records.ts | 6 +-
.../slices/agent-status-slice-contract.ts | 4 +
src/renderer/src/store/slices/agent-status.ts | 1 +
.../web/preload-api/web-agent-status-api.ts | 1 +
src/shared/agent-prompt-injection.test.ts | 7 +
src/shared/agent-prompt-injection.ts | 13 +
src/shared/agent-status-types.ts | 3 +
src/shared/cli-argument-boundary.ts | 2 +
...chestration-fleet-agent-status-evidence.ts | 118 +++
.../orchestration-fleet-attention.test.ts | 77 ++
src/shared/orchestration-fleet-attention.ts | 102 ++
...orchestration-fleet-evidence-clock.test.ts | 111 +++
.../orchestration-fleet-outcome-resolution.ts | 62 ++
.../orchestration-fleet-projection.test.ts | 605 ++++++++++++
src/shared/orchestration-fleet-projection.ts | 176 ++++
.../orchestration-fleet-status-index.ts | 165 ++++
.../orchestration-fleet-worker-projection.ts | 266 ++++++
src/shared/orchestration-retry-request-id.ts | 12 +
src/shared/orchestration-rpc-contract.ts | 23 +-
src/shared/orchestration-worker-output.ts | 18 +
...rchestration-worker-start-prompt-budget.ts | 28 +
.../pane-agent-identity-inventory.test.ts | 2 +-
src/shared/protocol-version.ts | 12 +
src/shared/pty-liveness-verdict.test.ts | 28 +
src/shared/pty-liveness-verdict.ts | 10 +-
src/shared/pty-write-settlement.ts | 59 ++
src/shared/runtime-session-contracts.ts | 2 +
src/shared/runtime-terminal-contracts.ts | 17 +
src/shared/runtime-types.ts | 2 +
src/shared/worker-terminal-host-scope.test.ts | 203 ++++
src/shared/worker-terminal-host-scope.ts | 82 ++
...completed-worker-retirement-resume.spec.ts | 2 +-
.../cross-version-terminal-wire.unit.test.ts | 27 -
.../helpers/orchestration-mail-pane-agent.ts | 30 +-
tests/e2e/helpers/orchestration-mail-store.ts | 11 +-
.../orchestration-idle-mail-delivery.spec.ts | 231 ++++-
.../orchestration-idle-mail-restore.spec.ts | 8 +-
...tration-worker-terminal-visibility.spec.ts | 27 +-
...ration-worker-transcript-providers.spec.ts | 427 +++++++++
.../terminal-send-agent-prompt-submit.spec.ts | 7 +-
tests/tools/repro-terminal-send-submit.mjs | 22 +-
573 files changed, 38802 insertions(+), 7555 deletions(-)
create mode 100644 config/scripts/orchestration-guide-command-contract.test.mjs
create mode 100644 skill-guides/orchestration/references/coordinator-loop.md
create mode 100644 skill-guides/orchestration/references/legacy-contract-migration.md
create mode 100644 skill-guides/orchestration/references/low-level-topology.md
create mode 100644 skill-guides/orchestration/references/messaging-and-gates.md
create mode 100644 skill-guides/orchestration/references/placement-and-remote.md
create mode 100644 skill-guides/orchestration/references/recovery-and-cleanup.md
create mode 100644 skill-guides/orchestration/references/worker-contract.md
create mode 100644 src/cli/handlers/bundled-skill-guide-table.ts
create mode 100644 src/cli/handlers/orchestration-task-list-brief.test.ts
create mode 100644 src/cli/handlers/orchestration/worker-list-run-scope.test.ts
create mode 100644 src/cli/handlers/orchestration/worker-list-run-scope.ts
create mode 100644 src/cli/handlers/orchestration/worker-output.test.ts
create mode 100644 src/cli/handlers/skill-guide-get.ts
create mode 100644 src/cli/handlers/terminal-close.ts
create mode 100644 src/cli/handlers/terminal-send.ts
create mode 100644 src/cli/retry-request-flag.test.ts
create mode 100644 src/cli/retry-request-flag.ts
create mode 100644 src/cli/runtime/runtime-remote-pairing.ts
create mode 100644 src/cli/runtime/terminal-prompt-mutation-recovery.ts
create mode 100644 src/cli/skills-command-flag-help.ts
create mode 100644 src/cli/skills-reference-selector.test.ts
create mode 100644 src/cli/specs/terminal-send.ts
create mode 100644 src/cli/stdout-line.ts
create mode 100644 src/cli/worktree-selector-recovery.ts
create mode 100644 src/main/daemon/daemon-client-notify-settlement.test.ts
create mode 100644 src/main/daemon/daemon-pty-session-input.ts
create mode 100644 src/main/daemon/daemon-pty-write-settlement-recovery.test.ts
create mode 100644 src/main/providers/settled-pty-write-stub.ts
create mode 100644 src/main/providers/settled-pty-writer-census.test.ts
create mode 100644 src/main/runtime/agent-prompt-receipt-correlation.test.ts
create mode 100644 src/main/runtime/agent-prompt-request-correlation.test.ts
create mode 100644 src/main/runtime/agent-prompt-request-correlation.ts
create mode 100644 src/main/runtime/agent-status-observed-pane-identity.ts
create mode 100644 src/main/runtime/orca-runtime-agent-prompt-request-correlation.ts
create mode 100644 src/main/runtime/orca-runtime-exact-worker-provider-session.test.ts
create mode 100644 src/main/runtime/orca-runtime-test-orchestration-messages.spec.ts
create mode 100644 src/main/runtime/orca-runtime-tests/orchestration-attention-batching.spec.ts
create mode 100644 src/main/runtime/orchestration-dispatch-mailbox-delivery.test.ts
create mode 100644 src/main/runtime/orchestration-fleet-agent-status-snapshot.ts
create mode 100644 src/main/runtime/orchestration-mailbox-cold-park-idle.test.ts
create mode 100644 src/main/runtime/orchestration-mailbox-crash-recovery.test.ts
create mode 100644 src/main/runtime/orchestration-mailbox-filtered-waiters.test.ts
create mode 100644 src/main/runtime/orchestration-mailbox-pointer-cli-command.test.ts
create mode 100644 src/main/runtime/orchestration-mailbox-pty-write-gate.test.ts
create mode 100644 src/main/runtime/orchestration-messages-fake-parity.test.ts
create mode 100644 src/main/runtime/orchestration/db/attempt-observation-store.ts
create mode 100644 src/main/runtime/orchestration/db/attempt-observation-types.ts
create mode 100644 src/main/runtime/orchestration/db/attempt-outcome-projection.test.ts
create mode 100644 src/main/runtime/orchestration/db/attempt-outcome-projection.ts
create mode 100644 src/main/runtime/orchestration/db/decision-gate-lifecycle.test.ts
create mode 100644 src/main/runtime/orchestration/db/dispatch-mailbox-consumer-fencing.test.ts
create mode 100644 src/main/runtime/orchestration/db/federation/federated-dispatch-observation-fence.test.ts
create mode 100644 src/main/runtime/orchestration/db/federation/federated-dispatch-observation-fence.ts
create mode 100644 src/main/runtime/orchestration/db/federation/remote-attachment-liveness.ts
create mode 100644 src/main/runtime/orchestration/db/federation/remote-dispatch-attachment-release.test.ts
create mode 100644 src/main/runtime/orchestration/db/federation/remote-dispatch-attachment-release.ts
create mode 100644 src/main/runtime/orchestration/db/lifecycle-transition-boundary.test.ts
create mode 100644 src/main/runtime/orchestration/db/lifecycle-transition.test.ts
create mode 100644 src/main/runtime/orchestration/db/lifecycle-transition.ts
create mode 100644 src/main/runtime/orchestration/db/lifecycle-write-transaction-runner.ts
create mode 100644 src/main/runtime/orchestration/db/messages/mailbox-pointer-enter-state.ts
create mode 100644 src/main/runtime/orchestration/db/messages/role-mailbox-delivery.ts
create mode 100644 src/main/runtime/orchestration/db/schema/migrate-mailbox-pointer-enter-v33.ts
create mode 100644 src/main/runtime/orchestration/db/schema/migrate-role-mailbox-delivery-v34.ts
create mode 100644 src/main/runtime/orchestration/db/schema/migrate-v35.ts
create mode 100644 src/main/runtime/orchestration/db/schema/migrate-v36.ts
create mode 100644 src/main/runtime/orchestration/db/schema/migrate-v37.ts
create mode 100644 src/main/runtime/orchestration/db/schema/migrate-v38.ts
create mode 100644 src/main/runtime/orchestration/db/worker-terminal/failed-start-terminal-adoption.ts
create mode 100644 src/main/runtime/orchestration/db/worker-terminal/worker-terminal-attention-query.ts
create mode 100644 src/main/runtime/orchestration/db/worker-terminal/worker-terminal-inventory-counts.ts
create mode 100644 src/main/runtime/orchestration/db/worker-terminal/worker-terminal-user-takeover.ts
create mode 100644 src/main/runtime/orchestration/dispatch-consumer-generation-migration.test.ts
create mode 100644 src/main/runtime/orchestration/dispatch-creator-identity-migration.test.ts
create mode 100644 src/main/runtime/orchestration/failed-start-terminal-adoption.test.ts
create mode 100644 src/main/runtime/orchestration/federation-ack-checkpoints.test.ts
create mode 100644 src/main/runtime/orchestration/federation-sync-capability.ts
create mode 100644 src/main/runtime/orchestration/federation-sync-message.ts
create mode 100644 src/main/runtime/orchestration/federation-sync-test-harness.ts
create mode 100644 src/main/runtime/orchestration/lifecycle-caller-edges.test.ts
create mode 100644 src/main/runtime/orchestration/mailbox-pointer-delivery-contract.ts
create mode 100644 src/main/runtime/orchestration/mailbox-pointer-pty-write.ts
create mode 100644 src/main/runtime/orchestration/mailbox-pointer-resume.ts
create mode 100644 src/main/runtime/orchestration/mailbox-pointer-stage.test.ts
create mode 100644 src/main/runtime/orchestration/mailbox-pointer-stage.ts
create mode 100644 src/main/runtime/orchestration/mailbox-pointer-submit.test.ts
create mode 100644 src/main/runtime/orchestration/orchestration-all-start-versions-migration.test.ts
create mode 100644 src/main/runtime/orchestration/orchestration-peer-capability-cache.test.ts
create mode 100644 src/main/runtime/orchestration/orchestration-peer-capability-cache.ts
create mode 100644 src/main/runtime/orchestration/orchestration-settled-worker-resume-fence-db.test.ts
create mode 100644 src/main/runtime/orchestration/r1-identity-migration.test.ts
create mode 100644 src/main/runtime/orchestration/settled-question-threads-migration.test.ts
create mode 100644 src/main/runtime/orchestration/worker-attention-context.test.ts
create mode 100644 src/main/runtime/orchestration/worker-attention-context.ts
create mode 100644 src/main/runtime/orchestration/worker-output-archive.test.ts
create mode 100644 src/main/runtime/orchestration/worker-report-observation.ts
create mode 100644 src/main/runtime/orchestration/worker-transcript-local-checkpoint.ts
create mode 100644 src/main/runtime/orchestration/worker-transcript-local-read.ts
create mode 100644 src/main/runtime/orchestration/worker-transcript-remote-range-read.ts
create mode 100644 src/main/runtime/orchestration/worker-transcript-remote-read.test.ts
create mode 100644 src/main/runtime/orchestration/worker-transcript-remote-read.ts
create mode 100644 src/main/runtime/orchestration/worker-transcript-source-identity.ts
create mode 100644 src/main/runtime/rpc/dispatcher-unary-method-invocation.ts
delete mode 100644 src/main/runtime/rpc/methods/orchestration-federation-liveness-verdict.test.ts
delete mode 100644 src/main/runtime/rpc/methods/orchestration-federation-methods.ts
delete mode 100644 src/main/runtime/rpc/methods/orchestration-federation-output.test.ts
delete mode 100644 src/main/runtime/rpc/methods/orchestration-send-point-to-point.ts
delete mode 100644 src/main/runtime/rpc/methods/orchestration-worker-methods.ts
delete mode 100644 src/main/runtime/rpc/methods/orchestration-worker-observation.ts
delete mode 100644 src/main/runtime/rpc/methods/orchestration-worker-release.test.ts
delete mode 100644 src/main/runtime/rpc/methods/orchestration-worker-start-schema.ts
delete mode 100644 src/main/runtime/rpc/methods/orchestration-worker-stop.ts
delete mode 100644 src/main/runtime/rpc/methods/orchestration-workers.ts
rename src/main/runtime/rpc/methods/{orchestration-cli-runtime-boundary.test.ts => orchestration/cli-runtime-boundary.test.ts} (92%)
rename src/main/runtime/rpc/methods/{orchestration-federated-attach-receipt.test.ts => orchestration/federation/federated-attach-receipt.test.ts} (91%)
rename src/main/runtime/rpc/methods/{orchestration-federated-attach-receipt.ts => orchestration/federation/federated-attach-receipt.ts} (94%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federated-fleet-host-groups.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federated-fleet-snapshot.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federated-fleet-snapshot.ts
rename src/main/runtime/rpc/methods/{orchestration-federated-message-targeting.test.ts => orchestration/federation/federated-message-targeting.test.ts} (89%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federated-release-safety.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federated-transport-safety.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federated-worker-read.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federated-worker-release-host.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federated-worker-release.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federated-worker-show.ts
rename src/main/runtime/rpc/methods/{orchestration-federated-worker-start-receipt.test.ts => orchestration/federation/federated-worker-start-receipt.test.ts} (76%)
rename src/main/runtime/rpc/methods/{orchestration-federated-worker-start-unknown-receipt.ts => orchestration/federation/federated-worker-start-receipts.ts} (50%)
rename src/main/runtime/rpc/methods/{orchestration-federated-worker-start.ts => orchestration/federation/federated-worker-start.ts} (82%)
rename src/main/runtime/rpc/methods/{orchestration-federation-agent-launch.test.ts => orchestration/federation/federation-agent-launch.test.ts} (95%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federation-attachment-observation.ts
rename src/main/runtime/rpc/methods/{orchestration-federation-control-mail.test.ts => orchestration/federation/federation-control-mail.test.ts} (85%)
rename src/main/runtime/rpc/methods/{orchestration-federation-control.ts => orchestration/federation/federation-control.ts} (67%)
rename src/main/runtime/rpc/methods/{orchestration-federation-effects.test.ts => orchestration/federation/federation-effects.test.ts} (96%)
rename src/main/runtime/rpc/methods/{orchestration-federation-effects.ts => orchestration/federation/federation-effects.ts} (100%)
rename src/main/runtime/rpc/methods/{orchestration-federation-folder-placement.test.ts => orchestration/federation/federation-folder-placement.test.ts} (90%)
rename src/main/runtime/rpc/methods/{orchestration-federation-lifecycle-settlement.test.ts => orchestration/federation/federation-lifecycle-settlement.test.ts} (97%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federation-liveness-verdict.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federation-methods.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federation-output.test.ts
rename src/main/runtime/rpc/methods/{orchestration-federation-relay.ts => orchestration/federation/federation-relay.ts} (95%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federation-release-recovery-scenarios.test-support.ts
rename src/main/runtime/rpc/methods/{orchestration-federation-test-request.ts => orchestration/federation/federation-request.test-support.ts} (80%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federation-runtime.test-support.ts
rename src/main/runtime/rpc/methods/{orchestration-federation-setup.test.ts => orchestration/federation/federation-setup.test.ts} (95%)
rename src/main/runtime/rpc/methods/{orchestration-federation-setup.ts => orchestration/federation/federation-setup.ts} (91%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/federation/federation-start-prompt-budget.test.ts
rename src/main/runtime/rpc/methods/{orchestration-federation-start-receipt.ts => orchestration/federation/federation-start-receipt.ts} (75%)
rename src/main/runtime/rpc/methods/{orchestration-federation-start-schema.ts => orchestration/federation/federation-start-schema.ts} (91%)
rename src/main/runtime/rpc/methods/{orchestration-federation.test.ts => orchestration/federation/federation.test.ts} (90%)
rename src/main/runtime/rpc/methods/{orchestration-federation.ts => orchestration/federation/federation.ts} (85%)
rename src/main/runtime/rpc/methods/{orchestration-gate-run-authorization.test.ts => orchestration/gates/gate-run-authorization.test.ts} (99%)
rename src/main/runtime/rpc/methods/{orchestration-gates.test.ts => orchestration/gates/gates.test.ts} (95%)
rename src/main/runtime/rpc/methods/{orchestration-gates.ts => orchestration/gates/gates.ts} (92%)
rename src/main/runtime/rpc/methods/{orchestration-ask-methods.ts => orchestration/messaging/ask-methods.ts} (92%)
rename src/main/runtime/rpc/methods/{orchestration-ask-remote.ts => orchestration/messaging/ask-remote.ts} (92%)
rename src/main/runtime/rpc/methods/{orchestration-ask.test.ts => orchestration/messaging/ask.test.ts} (96%)
rename src/main/runtime/rpc/methods/{orchestration-check-direct.ts => orchestration/messaging/check-direct.ts} (76%)
rename src/main/runtime/rpc/methods/{orchestration-check-methods.ts => orchestration/messaging/check-methods.ts} (52%)
rename src/main/runtime/rpc/methods/{orchestration-check-run.ts => orchestration/messaging/check-run.ts} (90%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/messaging/check-superseded-terminal.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/messaging/check-worker-consumer-fencing.test.ts
rename src/main/runtime/rpc/methods/{orchestration-check-worker.ts => orchestration/messaging/check-worker.ts} (52%)
rename src/main/runtime/rpc/methods/{orchestration-check.test.ts => orchestration/messaging/check.test.ts} (89%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/messaging/dispatch-mailbox-fence.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/messaging/mailbox-message-receipt.ts
rename src/main/runtime/rpc/methods/{orchestration-message-methods.ts => orchestration/messaging/message-methods.ts} (79%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/messaging/mutation-replay-nudge.ts
rename src/main/runtime/rpc/methods/{orchestration-recipient-routing.test.ts => orchestration/messaging/recipient-routing.test.ts} (96%)
rename src/main/runtime/rpc/methods/{orchestration-recipient-routing.ts => orchestration/messaging/recipient-routing.ts} (95%)
rename src/main/runtime/rpc/methods/{orchestration-send-control-mail.ts => orchestration/messaging/send-control-mail.ts} (74%)
rename src/main/runtime/rpc/methods/{orchestration-send-dispatch-authority.test.ts => orchestration/messaging/send-dispatch-authority.test.ts} (90%)
rename src/main/runtime/rpc/methods/{orchestration-send-group.ts => orchestration/messaging/send-group.ts} (81%)
rename src/main/runtime/rpc/methods/{orchestration-send-invalid-type.test.ts => orchestration/messaging/send-invalid-type.test.ts} (77%)
rename src/main/runtime/rpc/methods/{orchestration-send-methods.ts => orchestration/messaging/send-methods.ts} (77%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/messaging/send-point-to-point.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/messaging/send-receipt-plumbing.test.ts
rename src/main/runtime/rpc/methods/{orchestration-send-remote.ts => orchestration/messaging/send-remote.ts} (83%)
rename src/main/runtime/rpc/methods/{orchestration-send.test.ts => orchestration/messaging/send.test.ts} (98%)
rename src/main/runtime/rpc/methods/{orchestration-settled-dispatch-mail.test.ts => orchestration/messaging/settled-dispatch-mail.test.ts} (90%)
rename src/main/runtime/rpc/methods/{orchestration-routing.ts => orchestration/routing.ts} (91%)
rename src/main/runtime/rpc/methods/{orchestration-rpc-test-harness.ts => orchestration/rpc-test-harness.ts} (94%)
rename src/main/runtime/rpc/methods/{orchestration-dispatch-creator.ts => orchestration/runs/dispatch-creator.ts} (85%)
rename src/main/runtime/rpc/methods/{orchestration-dispatch-methods.ts => orchestration/runs/dispatch-methods.ts} (74%)
rename src/main/runtime/rpc/methods/{orchestration-migration-behavior.test.ts => orchestration/runs/migration-behavior.test.ts} (92%)
rename src/main/runtime/rpc/methods/{orchestration-mutation-request-show.ts => orchestration/runs/mutation-request-show.ts} (91%)
rename src/main/runtime/rpc/methods/{orchestration-reset-methods.ts => orchestration/runs/reset-methods.ts} (85%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/runs/run-receipt.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/runs/run-receipt.ts
rename src/main/runtime/rpc/methods/{orchestration-run-scope.ts => orchestration/runs/run-scope.ts} (93%)
rename src/main/runtime/rpc/methods/{orchestration-runs.test.ts => orchestration/runs/runs.test.ts} (90%)
rename src/main/runtime/rpc/methods/{orchestration-runs.ts => orchestration/runs/runs.ts} (83%)
rename src/main/runtime/rpc/methods/{orchestration-tasks-dispatch.test.ts => orchestration/runs/tasks-dispatch.test.ts} (95%)
rename src/main/runtime/rpc/methods/{orchestration-schemas.ts => orchestration/schemas.ts} (95%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/agent-status-producer-census.test.ts
rename src/main/runtime/rpc/methods/{orchestration-composed-workers.test.ts => orchestration/worker/composed-workers.test.ts} (97%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/context-only-dispatch-retry.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/failed-start-residual-terminal.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/failed-start-residual-terminal.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/fleet-status-observed-identity.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/fleet-status-terminal-identity.test.ts
rename src/main/runtime/rpc/methods/{orchestration-folder-worktree-placement.ts => orchestration/worker/folder-worktree-placement.ts} (65%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/legacy-dispatch-projection.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/local-worker-start.ts
rename src/main/runtime/rpc/methods/{orchestration-manual-dispatch-observation.test.ts => orchestration/worker/manual-dispatch-observation.test.ts} (86%)
rename src/main/runtime/rpc/methods/{orchestration-manual-dispatch-release.test.ts => orchestration/worker/manual-dispatch-release.test.ts} (96%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/self-dispatch-nesting-depth.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/task-deps-argument.ts
rename src/main/runtime/rpc/methods/{orchestration-worker-archive-read.ts => orchestration/worker/worker-archive-read.ts} (57%)
rename src/main/runtime/rpc/methods/{orchestration-worker-control.ts => orchestration/worker/worker-control.ts} (52%)
rename src/main/runtime/rpc/methods/{orchestration-worker-interactive-wait.test.ts => orchestration/worker/worker-interactive-wait.test.ts} (94%)
rename src/main/runtime/rpc/methods/{orchestration-worker-launch-preferences.test.ts => orchestration/worker/worker-launch-preferences.test.ts} (83%)
rename src/main/runtime/rpc/methods/{orchestration-worker-launch-preferences.ts => orchestration/worker/worker-launch-preferences.ts} (89%)
rename src/main/runtime/rpc/methods/{orchestration-worker-legacy-federated-read.ts => orchestration/worker/worker-legacy-federated-read.ts} (83%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-list-cursor.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-list-method.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-list-pagination.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-list-projection.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-list-run-scope-rpc.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-list-snapshot-store.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-methods.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-observation.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-observation.ts
rename src/main/runtime/rpc/methods/{orchestration-worker-output.test.ts => orchestration/worker/worker-output.test.ts} (65%)
rename src/main/runtime/rpc/methods/{orchestration-worker-output.ts => orchestration/worker/worker-output.ts} (75%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-read-projection.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-release-archive.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-release-close-error.ts
rename src/main/runtime/rpc/methods/{orchestration-worker-release-completion.ts => orchestration/worker/worker-release-completion.ts} (58%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-release-inventory.test.ts
rename src/main/runtime/rpc/methods/{orchestration-worker-release-liveness-verdict.test.ts => orchestration/worker/worker-release-liveness-verdict.test.ts} (53%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-release-ownership-guard.test.ts
rename src/main/runtime/rpc/methods/{orchestration-worker-release-recovery.test.ts => orchestration/worker/worker-release-recovery.test.ts} (68%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-release-schemas.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-release.test-support.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts
rename src/main/runtime/rpc/methods/{orchestration-worker-release.ts => orchestration/worker/worker-release.ts} (63%)
rename src/main/runtime/rpc/methods/{orchestration-worker-setup-gate.ts => orchestration/worker/worker-setup-gate.ts} (95%)
rename src/main/runtime/rpc/methods/{orchestration-worker-start-budgets.test.ts => orchestration/worker/worker-start-budgets.test.ts} (86%)
rename src/main/runtime/rpc/methods/{orchestration-worker-start-budgets.ts => orchestration/worker/worker-start-budgets.ts} (94%)
rename src/main/runtime/rpc/methods/{orchestration-worker-start-outcome-classification.test.ts => orchestration/worker/worker-start-outcome-classification.test.ts} (94%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-start-prompt-budget.test.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-start-prompt-budget.ts
rename src/main/runtime/rpc/methods/{orchestration-worker-start-prompt-contract.test.ts => orchestration/worker/worker-start-prompt-contract.test.ts} (74%)
rename src/main/runtime/rpc/methods/{orchestration-worker-start-receipt.ts => orchestration/worker/worker-start-receipt.ts} (57%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-start-schema.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-start-terminal-target.test.ts
rename src/main/runtime/rpc/methods/{orchestration-worker-start-validation.ts => orchestration/worker/worker-start-validation.ts} (91%)
rename src/main/runtime/rpc/methods/{orchestration-worker-stop-capability.test.ts => orchestration/worker/worker-stop-capability.test.ts} (89%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-stop-exit-race.test.ts
rename src/main/runtime/rpc/methods/{orchestration-worker-stop-liveness-verdict.test.ts => orchestration/worker/worker-stop-liveness-verdict.test.ts} (97%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-stop.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-terminal-release-lease.ts
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/worker-terminal-resource-presentation.ts
rename src/main/runtime/rpc/methods/{orchestration-worker-topology.ts => orchestration/worker/worker-topology.ts} (96%)
rename src/main/runtime/rpc/methods/{orchestration-workers-new-worktree.test.ts => orchestration/worker/workers-new-worktree.test.ts} (98%)
rename src/main/runtime/rpc/methods/{orchestration-workers-recovery.test.ts => orchestration/worker/workers-recovery.test.ts} (74%)
create mode 100644 src/main/runtime/rpc/methods/orchestration/worker/workers.ts
create mode 100644 src/main/runtime/rpc/methods/settled-worker-resume-fence-sweep.ts
create mode 100644 src/main/runtime/rpc/methods/terminal/terminal-prompt-receipt.ts
create mode 100644 src/main/runtime/rpc/orchestration-commit-notify-characterization.test.ts
create mode 100644 src/main/runtime/rpc/orchestration-mutation-executor.test.ts
create mode 100644 src/main/runtime/rpc/orchestration-mutation-receipt.ts
create mode 100644 src/main/runtime/rpc/terminal-prompt-delivery-receipt.test.ts
create mode 100644 src/main/runtime/runtime-legacy-worker-terminal-resume-fence.test.ts
create mode 100644 src/main/ssh/ssh-host-cli-deadline.ts
create mode 100644 src/renderer/src/store/slices/agent-status-open-tab-resume-fence.test.ts
create mode 100644 src/shared/orchestration-fleet-agent-status-evidence.ts
create mode 100644 src/shared/orchestration-fleet-attention.test.ts
create mode 100644 src/shared/orchestration-fleet-attention.ts
create mode 100644 src/shared/orchestration-fleet-evidence-clock.test.ts
create mode 100644 src/shared/orchestration-fleet-outcome-resolution.ts
create mode 100644 src/shared/orchestration-fleet-projection.test.ts
create mode 100644 src/shared/orchestration-fleet-projection.ts
create mode 100644 src/shared/orchestration-fleet-status-index.ts
create mode 100644 src/shared/orchestration-fleet-worker-projection.ts
create mode 100644 src/shared/orchestration-retry-request-id.ts
create mode 100644 src/shared/orchestration-worker-start-prompt-budget.ts
create mode 100644 src/shared/pty-liveness-verdict.test.ts
create mode 100644 src/shared/pty-write-settlement.ts
create mode 100644 src/shared/worker-terminal-host-scope.test.ts
create mode 100644 src/shared/worker-terminal-host-scope.ts
create mode 100644 tests/e2e/orchestration-worker-transcript-providers.spec.ts
diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc
index 70b1092fedf..a6ac6b1fe23 100644
--- a/config/reliability-gates.jsonc
+++ b/config/reliability-gates.jsonc
@@ -12959,15 +12959,15 @@
"invariant": "Injected orchestration task prompts for recognized agent CLIs must send the prompt body inside one bracketed-paste frame, sanitize embedded ESC bytes, preserve chunk boundaries without losing the frame, and submit exactly once only after the agent can accept Enter. A successful orchestration.workerStart must durably record exactly one accepted and started turn; a swallowed Enter must fail with agent_prompt_stalled and never trigger a blind rescue Enter. Claude and Codex must emit a post-paste composer marker and then settle, or reach the bounded fallback first; every other agent retains the platform delay.",
"oracle": "Runtime tests assert the exact PTY write sequence, failure cleanup, Claude/Codex marker-gated multi-frame renders, and the legacy platform delay for every other configured agent. The candidate resets settlement on later frames, gives a late marker a fresh bounded window, and still submits once at the hard deadline if output never settles. The worker-start contract drives the production RPC through a delayed fake Codex composer and independently checks exact turn/Enter counts plus reopened SQLite Task, Dispatch, worker receipt, and mutation receipt state for accepted and swallowed outcomes. Other orchestration tests assert dispatch/coordinator use the agent prompt path; the live CLI harness covers long Codex-like framing.",
"commands": [
- "pnpm exec vitest run --config config/vitest.config.ts src/shared/agent-prompt-injection.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts src/main/runtime/orchestration/coordinator.test.ts",
- "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration-worker-start-prompt-contract.test.ts --reporter=dot",
+ "pnpm exec vitest run --config config/vitest.config.ts src/shared/agent-prompt-injection.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/rpc/methods/orchestration/runs/tasks-dispatch.test.ts src/main/runtime/orchestration/coordinator.test.ts",
+ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/worker/worker-start-prompt-contract.test.ts --reporter=dot",
"node tests/tools/repro-orchestration-long-prompt.mjs --cli out/bin/orca-dev --mode codex-like --size-kb 32 --timeout-ms 20000"
],
"testFiles": [
"src/shared/agent-prompt-injection.test.ts",
"src/main/runtime/orca-runtime.test.ts",
- "src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts",
- "src/main/runtime/rpc/methods/orchestration-worker-start-prompt-contract.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/runs/tasks-dispatch.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/worker/worker-start-prompt-contract.test.ts",
"src/main/runtime/orchestration/coordinator.test.ts",
"tests/tools/repro-orchestration-long-prompt.mjs"
],
@@ -12995,7 +12995,7 @@
]
},
{
- "file": "src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts",
+ "file": "src/main/runtime/rpc/methods/orchestration/runs/tasks-dispatch.test.ts",
"assertions": [
"orchestration.dispatch uses the agent prompt path for injected preambles",
"raw terminal.send is not called for injected task prompts",
@@ -13003,7 +13003,7 @@
]
},
{
- "file": "src/main/runtime/rpc/methods/orchestration-worker-start-prompt-contract.test.ts",
+ "file": "src/main/runtime/rpc/methods/orchestration/worker/worker-start-prompt-contract.test.ts",
"assertions": [
"delayed composer readiness produces exactly one submitted and started turn with no premature Enter and durable ready receipts",
"a swallowed Enter records agent_prompt_stalled across Task, Dispatch, worker, and mutation receipts without a rescue Enter"
@@ -13030,7 +13030,7 @@
"date": "2026-08-23",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration-worker-start-prompt-contract.test.ts --reporter=dot",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/worker/worker-start-prompt-contract.test.ts --reporter=dot",
"result": "passed",
"durationSeconds": 21.84,
"summary": "Two deterministic worker-start RPC contracts passed with fake clocks and reopened SQLite receipts for one accepted turn and one swallowed-Enter stalled outcome."
@@ -13039,7 +13039,7 @@
"date": "2026-08-14",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/shared/agent-prompt-injection.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts src/main/runtime/orchestration/coordinator.test.ts",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/shared/agent-prompt-injection.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/rpc/methods/orchestration/runs/tasks-dispatch.test.ts src/main/runtime/orchestration/coordinator.test.ts",
"result": "passed",
"durationSeconds": 11.32,
"summary": "4 files and 1,303 tests passed with one skipped. Claude and Codex both wait for post-marker quiescence, and a Codex marker arriving at 7.9 seconds receives a fresh window through its final slow frame. Exact-build live Codex workers accepted injected prompts without manual Enter, replied, called worker_done, and settled successfully in the rendered Electron UI."
@@ -13048,7 +13048,7 @@
"date": "2026-08-13",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/shared/agent-prompt-injection.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts src/main/runtime/orchestration/coordinator.test.ts",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/shared/agent-prompt-injection.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/rpc/methods/orchestration/runs/tasks-dispatch.test.ts src/main/runtime/orchestration/coordinator.test.ts",
"result": "passed",
"durationSeconds": 13.3,
"summary": "4 files and 1,283 tests passed. The hardened multi-frame oracle failed on the first-marker candidate because it submitted at 751 ms during an intermediate Claude frame; the quiescence candidate waited through the final 1,000 ms frame and submitted once at 2,500 ms. Continuous render output remained bounded to one fallback submit at 8 seconds. An isolated Claude Code 2.1.231 Haiku probe saw the first marker at 400 ms, continued output through 1,500 ms, sent one Enter at 3,000 ms after 1.5 seconds quiet, and created the expected marker; no Fable or Opus probe was used."
@@ -13057,7 +13057,7 @@
"date": "2026-08-13",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/shared/agent-prompt-injection.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts src/main/runtime/orchestration/coordinator.test.ts",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/shared/agent-prompt-injection.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/rpc/methods/orchestration/runs/tasks-dispatch.test.ts src/main/runtime/orchestration/coordinator.test.ts",
"result": "passed",
"durationSeconds": 16.9,
"summary": "4 files and 1,282 tests passed. Unmodified main wrote Enter at 500 ms before the deterministic Claude composer rendered at 750 ms; the candidate waited for the split show-cursor marker and wrote one Enter. A live Claude Code 2.1.231 Haiku trace rendered the pasted marker and show-cursor in one 523-byte frame without submitting a model request."
@@ -13066,7 +13066,7 @@
"date": "2026-07-07",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/shared/agent-prompt-injection.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts src/main/runtime/orchestration/coordinator.test.ts",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/shared/agent-prompt-injection.test.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/rpc/methods/orchestration/runs/tasks-dispatch.test.ts src/main/runtime/orchestration/coordinator.test.ts",
"result": "passed",
"durationSeconds": 7.4,
"summary": "4 test files passed, 697 tests passed; covers framing, runtime PTY writes, orchestration RPC dispatch, and coordinator dispatch behavior."
@@ -13136,12 +13136,12 @@
"internal incident evidence: improve-vps-setup, 2026-08-10"
],
"invariant": "Each message has one stable row ID and authoritative recipient; coordinator-addressed current-delivery inserts are atomically owned by run:. Pointer staging may set delivered_at but never consumes mail. Each Run consumer generation has at most one outstanding Delivery with a fixed ID and fixed message IDs; ordinary checks replay it until an explicit matching acknowledgment marks exactly those rows read. Rebinding fences the old generation, notification types/counts correspond to unread rows retrievable under the same authority, and federation replay imports each stable message identity once without re-waking an already-read duplicate.",
- "oracle": "Seed status, dispatch, and worker_done rows across direct-handle and canonical Run recipients in an isolated DB. Compare pointer count, RPC and built-CLI check output, direct SQLite rows, unread/peek/all/type filters, concurrent pollers, fixed Delivery IDs, explicit acknowledgment, restart, filtered check --wait, and coordinator remint. Route a 125-row old-handle backlog, inject a commit without notification, and require startup repair. Exercise duplicate Run/Dispatch owners, stale panes, 50-row pages, cancellation, lifecycle fencing, and absent PTYs. Drop a federation ACK, reconnect/restart v1/v2 peers, and require stable import plus no duplicate read-row wake. Hold a healthy SSH write past five seconds but below the 60-second settlement deadline, then separately exceed the bound and require retryable undelivered state.",
+ "oracle": "Seed status, dispatch, and worker_done rows across direct-handle and canonical Run recipients in an isolated DB. Compare pointer count, RPC and built-CLI check output, direct SQLite rows, unread/peek/all/type filters, concurrent pollers, fixed Delivery IDs, explicit acknowledgment, restart, filtered check --wait, and coordinator remint. Route a 125-row old-handle backlog, inject a commit without notification, and require startup repair. Exercise duplicate Run/Dispatch owners, stale panes, 50-row pages, cancellation, lifecycle fencing, and absent PTYs. Drop a federation ACK, reconnect/restart v1/v2 peers, and require stable import plus no duplicate read-row wake. Hold a healthy SSH write past five seconds but below the 60-second settlement deadline, then distinguish the three settlement outcomes end to end: only a proven refusal releases the reservation and drains a delivery parked behind the watermark; a dropped in-flight settlement must surface as unverifiable with bytes handed to the transport, preserve the durable write-attempted reservation, and emit no duplicate pointer after restart; a settled write that throws mid-pointer is unverifiable, not a refusal; and an Enter whose settlement is lost stays at enter-attempted so restart emits no second Enter. Install the production PTY controller and verify that it routes settled writes through the owning provider and refuses before any byte when the routed provider cannot settle. Census every production PTY provider class and reject a settlement synthesized from the fire-and-forget write.",
"commands": [
"pnpm run build:cli && pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration-message-delivery-identity.test.ts --reporter=dot --testTimeout=5000",
- "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration-mailbox-routing-races.test.ts src/main/runtime/orchestration-mailbox-notification-consistency.test.ts src/main/runtime/orchestration-mailbox-detached-routing.test.ts src/main/runtime/orchestration-mailbox-transport-settlement.test.ts src/main/runtime/orchestration/run-coordinator-handle-migration.test.ts src/main/runtime/orchestration/orchestration-run-delivery-db.test.ts src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts src/main/runtime/orchestration/formatter.test.ts src/main/providers/ssh-pty-provider.test.ts src/main/providers/ssh-pty-write.test.ts src/main/daemon/client.test.ts src/main/daemon/daemon-pty-router.test.ts src/main/daemon/degraded-daemon-pty-provider.test.ts",
- "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/terminal-send-stale-leaf-liveness.test.ts src/main/runtime/rpc/methods/orchestration-runs.test.ts src/main/runtime/rpc/methods/orchestration-send.test.ts src/main/runtime/rpc/methods/orchestration-check.test.ts",
- "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration/federation-sync.test.ts src/main/runtime/rpc/methods/orchestration-federation.test.ts src/main/runtime/rpc/methods/orchestration-federation-lifecycle-settlement.test.ts --reporter=dot"
+ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration-mailbox-routing-races.test.ts src/main/runtime/orchestration-mailbox-notification-consistency.test.ts src/main/runtime/orchestration-mailbox-detached-routing.test.ts src/main/runtime/orchestration-mailbox-transport-settlement.test.ts src/main/ipc/pty-controller-ownership-routing.test.ts src/main/runtime/orchestration/run-coordinator-handle-migration.test.ts src/main/runtime/orchestration/orchestration-run-delivery-db.test.ts src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts src/main/runtime/orchestration/formatter.test.ts src/main/providers/ssh-pty-provider.test.ts src/main/providers/ssh-pty-write.test.ts src/main/providers/settled-pty-writer-census.test.ts src/main/runtime/orchestration/mailbox-pointer-stage.test.ts src/main/daemon/client.test.ts src/main/daemon/daemon-pty-router.test.ts src/main/daemon/degraded-daemon-pty-provider.test.ts",
+ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/terminal-send-stale-leaf-liveness.test.ts src/main/runtime/rpc/methods/orchestration/runs/runs.test.ts src/main/runtime/rpc/methods/orchestration/messaging/send.test.ts src/main/runtime/rpc/methods/orchestration/messaging/check.test.ts",
+ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration/federation-sync.test.ts src/main/runtime/rpc/methods/orchestration/federation/federation.test.ts src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts --reporter=dot"
],
"testFiles": [
"src/main/runtime/orchestration-message-delivery-identity.test.ts",
@@ -13149,23 +13149,26 @@
"src/main/runtime/orchestration-mailbox-detached-routing.test.ts",
"src/main/runtime/orchestration-mailbox-routing-races.test.ts",
"src/main/runtime/orchestration-mailbox-transport-settlement.test.ts",
+ "src/main/ipc/pty-controller-ownership-routing.test.ts",
"src/main/runtime/orchestration/run-coordinator-handle-migration.test.ts",
"src/main/runtime/orchestration/orchestration-run-delivery-db.test.ts",
"src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts",
"src/main/runtime/orchestration/formatter.test.ts",
"src/main/providers/ssh-pty-provider.test.ts",
"src/main/providers/ssh-pty-write.test.ts",
+ "src/main/providers/settled-pty-writer-census.test.ts",
+ "src/main/runtime/orchestration/mailbox-pointer-stage.test.ts",
"src/main/daemon/client.test.ts",
"src/main/daemon/daemon-pty-router.test.ts",
"src/main/daemon/degraded-daemon-pty-provider.test.ts",
"src/main/runtime/orca-runtime.test.ts",
"src/main/runtime/terminal-send-stale-leaf-liveness.test.ts",
- "src/main/runtime/rpc/methods/orchestration-runs.test.ts",
- "src/main/runtime/rpc/methods/orchestration-send.test.ts",
- "src/main/runtime/rpc/methods/orchestration-check.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/runs/runs.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/messaging/send.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/messaging/check.test.ts",
"src/main/runtime/orchestration/federation-sync.test.ts",
- "src/main/runtime/rpc/methods/orchestration-federation.test.ts",
- "src/main/runtime/rpc/methods/orchestration-federation-lifecycle-settlement.test.ts"
+ "src/main/runtime/rpc/methods/orchestration/federation/federation.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts"
],
"assertionRefs": [
{
@@ -13229,14 +13232,14 @@
]
},
{
- "file": "src/main/runtime/rpc/methods/orchestration-federation.test.ts",
+ "file": "src/main/runtime/rpc/methods/orchestration/federation/federation.test.ts",
"assertions": [
"a lost relay acknowledgment retries without duplicating the home message",
"a reordered relay gap converges without loss or duplication"
]
},
{
- "file": "src/main/runtime/rpc/methods/orchestration-federation-lifecycle-settlement.test.ts",
+ "file": "src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts",
"assertions": [
"protocol v1 and v2 completion acknowledgments replay after Run-home restart",
"terminal settlement remains replayable until the worker durably acknowledges it"
@@ -13245,13 +13248,36 @@
{
"file": "src/main/runtime/orchestration-mailbox-transport-settlement.test.ts",
"assertions": [
- "a rejected pointer transport stays undelivered and becomes restart-retryable"
+ "a refused pointer transport releases its reservation, stays undelivered, and becomes restart-retryable",
+ "a dropped in-flight SSH settlement reaches the stager as unverifiable with bytes handed to the transport and emits no duplicate pointer after restart",
+ "a settled write that throws mid-pointer preserves the write-attempted reservation",
+ "an Enter whose settlement is lost stays at enter-attempted and restart emits no second Enter"
+ ]
+ },
+ {
+ "file": "src/main/runtime/orchestration/mailbox-pointer-stage.test.ts",
+ "assertions": [
+ "a refused pointer write drains a delivery parked behind its watermark"
+ ]
+ },
+ {
+ "file": "src/main/providers/settled-pty-writer-census.test.ts",
+ "assertions": [
+ "every production IPtyProvider class exposes a settled writer",
+ "no settled writer synthesizes its settlement from the fire-and-forget write"
+ ]
+ },
+ {
+ "file": "src/main/ipc/pty-controller-ownership-routing.test.ts",
+ "assertions": [
+ "the installed controller preserves provider uncertainty instead of flattening it",
+ "a routed provider that cannot settle is refused before any byte reaches its write"
]
},
{
"file": "src/main/daemon/client.test.ts",
"assertions": [
- "an asynchronous daemon socket write failure settles as rejected",
+ "an asynchronous daemon socket write failure settles as unverifiable, never as a proven refusal",
"a wedged daemon socket write disconnects at its bounded settlement deadline"
]
},
@@ -13276,11 +13302,20 @@
}
],
"evidenceRuns": [
+ {
+ "date": "2026-09-05",
+ "runner": "local",
+ "platform": "macos",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration-mailbox-routing-races.test.ts src/main/runtime/orchestration-mailbox-notification-consistency.test.ts src/main/runtime/orchestration-mailbox-detached-routing.test.ts src/main/runtime/orchestration-mailbox-transport-settlement.test.ts src/main/ipc/pty-controller-ownership-routing.test.ts src/main/runtime/orchestration/run-coordinator-handle-migration.test.ts src/main/runtime/orchestration/orchestration-run-delivery-db.test.ts src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts src/main/runtime/orchestration/formatter.test.ts src/main/providers/ssh-pty-provider.test.ts src/main/providers/ssh-pty-write.test.ts src/main/providers/settled-pty-writer-census.test.ts src/main/runtime/orchestration/mailbox-pointer-stage.test.ts src/main/daemon/client.test.ts src/main/daemon/daemon-pty-router.test.ts src/main/daemon/degraded-daemon-pty-provider.test.ts",
+ "result": "passed",
+ "durationSeconds": 4.73,
+ "summary": "267 tests passed after the pointer-write path moved to the three-valued WriteSettlement union. New coverage: a dropped in-flight SSH settlement reaches the stager as unverifiable with bytes handed to the transport, a settled write that throws mid-pointer preserves the write-attempted reservation, an Enter whose settlement is lost stays at enter-attempted with no second Enter after restart, a refusal releases the reservation and drains a delivery parked behind its watermark, the production controller refuses before any byte when the routed provider cannot settle, and a census pins the five production IPtyProvider classes and rejects a settlement synthesized from the fire-and-forget write. Each new assertion was verified red against the pre-fix shape."
+ },
{
"date": "2026-08-13",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration-mailbox-routing-races.test.ts src/main/runtime/orchestration-mailbox-notification-consistency.test.ts src/main/runtime/orchestration-mailbox-detached-routing.test.ts src/main/runtime/orchestration-mailbox-transport-settlement.test.ts src/main/runtime/orchestration/run-coordinator-handle-migration.test.ts src/main/runtime/orchestration/orchestration-run-delivery-db.test.ts src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts src/main/runtime/orchestration/formatter.test.ts src/main/providers/ssh-pty-provider.test.ts src/main/providers/ssh-pty-write.test.ts src/main/daemon/client.test.ts src/main/daemon/daemon-pty-router.test.ts src/main/daemon/degraded-daemon-pty-provider.test.ts",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration-mailbox-routing-races.test.ts src/main/runtime/orchestration-mailbox-notification-consistency.test.ts src/main/runtime/orchestration-mailbox-detached-routing.test.ts src/main/runtime/orchestration-mailbox-transport-settlement.test.ts src/main/ipc/pty-controller-ownership-routing.test.ts src/main/runtime/orchestration/run-coordinator-handle-migration.test.ts src/main/runtime/orchestration/orchestration-run-delivery-db.test.ts src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts src/main/runtime/orchestration/formatter.test.ts src/main/providers/ssh-pty-provider.test.ts src/main/providers/ssh-pty-write.test.ts src/main/providers/settled-pty-writer-census.test.ts src/main/runtime/orchestration/mailbox-pointer-stage.test.ts src/main/daemon/client.test.ts src/main/daemon/daemon-pty-router.test.ts src/main/daemon/degraded-daemon-pty-provider.test.ts",
"result": "passed",
"durationSeconds": 8.22,
"summary": "245 tests passed across mailbox identity, durable coordinator-handle migration, insertion-time canonicalization, duplicate-free 51-row ownership branch caps, unrestricted reservation merging, direct and Dispatch pointer suppression, persisted reconciliation, 50-row paging and filtered waits, cross-PTY serialization, lifecycle fencing, bounded daemon and SSH transport settlement, outstanding Deliveries, reminted Dispatch ownership, acknowledgment, cancellation, and bounded pane lookup."
@@ -13289,7 +13324,7 @@
"date": "2026-08-14",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration/federation-sync.test.ts src/main/runtime/rpc/methods/orchestration-federation.test.ts src/main/runtime/rpc/methods/orchestration-federation-lifecycle-settlement.test.ts --reporter=dot",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration/federation-sync.test.ts src/main/runtime/rpc/methods/orchestration/federation/federation.test.ts src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts --reporter=dot",
"result": "passed",
"durationSeconds": 8.99,
"summary": "52 tests passed with real OrchestrationDb rows, a deliberately dropped federation acknowledgment, reconnect/restart, forward-only checkpoints, duplicate read-row wake suppression, and protocol v1/v2 lifecycle settlement replay. The broader final federation/cross-version set passed 77/77."
@@ -13307,7 +13342,7 @@
"date": "2026-08-14",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/terminal-send-stale-leaf-liveness.test.ts src/main/runtime/rpc/methods/orchestration-runs.test.ts src/main/runtime/rpc/methods/orchestration-send.test.ts src/main/runtime/rpc/methods/orchestration-check.test.ts",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime.test.ts src/main/runtime/terminal-send-stale-leaf-liveness.test.ts src/main/runtime/rpc/methods/orchestration/runs/runs.test.ts src/main/runtime/rpc/methods/orchestration/messaging/send.test.ts src/main/runtime/rpc/methods/orchestration/messaging/check.test.ts",
"result": "passed",
"durationSeconds": 14.15,
"summary": "1,293 tests passed and 1 was skipped across Run-bound pointer delivery, PTY retirement and respawn, stale-leaf liveness, direct-mail routing, filtered waiter ownership, canonical stored-recipient notification, and orchestration RPC behavior."
@@ -13375,10 +13410,10 @@
"oracle": "Drive Run create, Task create, and worker-start through production Electron runtimes with a deterministic Codex fixture. Require append-only ledgers with one still-live PID and no interruption, a visible inactive worker tab while the coordinator stays active, Run delivery through stable pane identity, and stable PTY/incarnation, tab, leaf, worktree, Task, and Dispatch across workspace re-entry. In a restart journey, retain the original daemon PTY and PID, remove renderer ownership, retain sleeping-session evidence, mark the Dispatch legacy, relaunch, and require exact inactive tab adoption, readable ACK output, cleared resume state, one spawn, and no resume argv or Conversation interrupted text after another workspace round trip. The service oracle removes renderer lookup identity from current-contract callers while retaining real restored-PTY and hook commitments, replays authenticated completion and takeover across fresh runtimes, and requires one Task, Dispatch, terminal authority, message, mutation, ordinary-mail delivery, remote process fencing, and unchanged fixture marker bytes while foreign pane evidence remains rejected. Unit tests separately remint a creator pane and process from Run A into Run B, require the nested Run A worker to fall back to its current coordinator, require indexed query plans, and bound 300 Task reads with 50,000 retained Runs. They also assert authority-specific legacy affordances, exact identity and owner matching, retained-output fallback, pane-stable routing, federated non-activation, and SSH fallback parity.",
"commands": [
"pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/orchestration-runtime-update-settlement.test.ts --reporter=dot",
- "pnpm exec vitest run --config config/vitest.config.ts src/cli/handlers/orchestration.test.ts src/cli/handlers/orchestration-check-identity.test.ts src/cli/handlers/orchestration-worker-cli.test.ts src/main/runtime/rpc/methods/orchestration-composed-workers.test.ts src/main/runtime/rpc/methods/orchestration-check.test.ts src/main/runtime/rpc/methods/orchestration-send.test.ts src/main/ssh/ssh-remote-orca-cli.test.ts",
+ "pnpm exec vitest run --config config/vitest.config.ts src/cli/handlers/orchestration.test.ts src/cli/handlers/orchestration-check-identity.test.ts src/cli/handlers/orchestration-worker-cli.test.ts src/main/runtime/rpc/methods/orchestration/worker/composed-workers.test.ts src/main/runtime/rpc/methods/orchestration/messaging/check.test.ts src/main/runtime/rpc/methods/orchestration/messaging/send.test.ts src/main/ssh/ssh-remote-orca-cli.test.ts",
"pnpm exec vitest run --config config/vitest.config.ts src/cli/handlers/orchestration-lifecycle-rejection.test.ts src/cli/handlers/orchestration-lifecycle-json-rejection.test.ts src/cli/handlers/orchestration-migration.test.ts",
- "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration/formatter.test.ts src/main/runtime/rpc/methods/orchestration-federation.test.ts",
- "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration-federation-lifecycle-settlement.test.ts --reporter=dot",
+ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration/formatter.test.ts src/main/runtime/rpc/methods/orchestration/federation/federation.test.ts",
+ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts --reporter=dot",
"pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration/federation-acknowledgment-migration.test.ts --reporter=dot",
"pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration/orchestration-legacy-worker-terminal-recovery.test.ts",
"pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration/orchestration-creator-authority-performance.test.ts",
@@ -13399,11 +13434,11 @@
"src/cli/handlers/orchestration-migration.test.ts",
"src/cli/handlers/orchestration-check-identity.test.ts",
"src/cli/handlers/orchestration-worker-cli.test.ts",
- "src/main/runtime/rpc/methods/orchestration-composed-workers.test.ts",
- "src/main/runtime/rpc/methods/orchestration-check.test.ts",
- "src/main/runtime/rpc/methods/orchestration-send.test.ts",
- "src/main/runtime/rpc/methods/orchestration-federation.test.ts",
- "src/main/runtime/rpc/methods/orchestration-federation-lifecycle-settlement.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/worker/composed-workers.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/messaging/check.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/messaging/send.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/federation/federation.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts",
"src/main/runtime/orchestration/federation-acknowledgment-migration.test.ts",
"src/main/ssh/ssh-remote-orca-cli.test.ts",
"tests/e2e/orchestration-worker-terminal-visibility.spec.ts",
@@ -13486,27 +13521,27 @@
]
},
{
- "file": "src/main/runtime/rpc/methods/orchestration-composed-workers.test.ts",
+ "file": "src/main/runtime/rpc/methods/orchestration/worker/composed-workers.test.ts",
"assertions": [
"same-workspace worker creation uses visible inactive presentation",
"worker-start preserves and reports renderer reveal failures"
]
},
{
- "file": "src/main/runtime/rpc/methods/orchestration-check.test.ts",
+ "file": "src/main/runtime/rpc/methods/orchestration/messaging/check.test.ts",
"assertions": [
"Run delivery resolves through a stable coordinator pane after handle remint",
"a live handle cannot be retargeted by mismatched pane metadata"
]
},
{
- "file": "src/main/runtime/rpc/methods/orchestration-send.test.ts",
+ "file": "src/main/runtime/rpc/methods/orchestration/messaging/send.test.ts",
"assertions": [
"Dispatch delivery resolves through a stable worker pane after handle remint"
]
},
{
- "file": "src/main/runtime/rpc/methods/orchestration-federation-lifecycle-settlement.test.ts",
+ "file": "src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts",
"assertions": [
"a remote worker_done waits for Run-home settlement even when an older CLI omits the wait hint",
"protocol v1/v2 clients can start fresh workers and complete success or failure on a current worker server",
@@ -13533,7 +13568,7 @@
]
},
{
- "file": "src/main/runtime/rpc/methods/orchestration-federation.test.ts",
+ "file": "src/main/runtime/rpc/methods/orchestration/federation/federation.test.ts",
"assertions": ["federated worker placement explicitly sets activate=false"]
},
{
@@ -13578,7 +13613,7 @@
"date": "2026-08-13",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration-federation-lifecycle-settlement.test.ts --reporter=dot",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts --reporter=dot",
"result": "passed",
"durationSeconds": 6.02,
"summary": "The 70f1d52f mixed-version oracle passed all 21 cases. Protocol v1/v2 clients started fresh workers on a current server, completed success and failure with explicit legacy authority, and automatically retried a lost ACK after Run-home restart; current-protocol settlement and duplicate-report controls stayed green."
@@ -13587,7 +13622,7 @@
"date": "2026-08-13",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration-federation-lifecycle-settlement.test.ts --reporter=dot",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts --reporter=dot",
"result": "failed",
"durationSeconds": 4.21,
"summary": "The byte-identical 70f1d52f oracle failed 6 mixed-version cases while 15 controls passed when the fresh v1/v2 refusal was restored: success and failure through both negotiated versions plus both lost-ACK restart cases."
@@ -13596,7 +13631,7 @@
"date": "2026-08-12",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration-federation-lifecycle-settlement.test.ts --reporter=dot",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts --reporter=dot",
"result": "failed",
"durationSeconds": 5.05,
"summary": "The byte-identical ac7bdf4e federation oracle failed 7 of 17 tests on affected 09ec516ae5: fresh v1/v2 work started before completion rejection, persisted v1/v2 work could not finish after update, same-outcome ACKs rejected, duplicate reports remained pending, and a dropped ACK was not replayed."
@@ -13605,7 +13640,7 @@
"date": "2026-08-12",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration-federation-lifecycle-settlement.test.ts --reporter=dot",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts --reporter=dot",
"result": "failed",
"durationSeconds": 5.86,
"summary": "The same byte-identical oracle failed the same 7 of 17 tests on latest main 1136503c6a."
@@ -13614,7 +13649,7 @@
"date": "2026-08-12",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration-federation-lifecycle-settlement.test.ts --reporter=dot",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts --reporter=dot",
"result": "passed",
"durationSeconds": 4.28,
"summary": "The same byte-identical oracle passed all 17 tests on candidate 008f740161, including restart replay and both directions of v1/v2 update compatibility."
@@ -13623,7 +13658,7 @@
"date": "2026-08-12",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration-federation-lifecycle-settlement.test.ts --reporter=dot",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts --reporter=dot",
"result": "failed",
"durationSeconds": 19.84,
"summary": "With the claimed production files restored to latest main in 3a15d3ed5d, the same byte-identical oracle returned to the same 7 failures while 10 unaffected cases still passed."
@@ -13686,7 +13721,7 @@
"date": "2026-07-28",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/cli/handlers/orchestration.test.ts src/cli/handlers/orchestration-check-identity.test.ts src/cli/handlers/orchestration-worker-cli.test.ts src/main/runtime/rpc/methods/orchestration-composed-workers.test.ts src/main/runtime/rpc/methods/orchestration-check.test.ts src/main/runtime/rpc/methods/orchestration-send.test.ts src/main/ssh/ssh-remote-orca-cli.test.ts",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/cli/handlers/orchestration.test.ts src/cli/handlers/orchestration-check-identity.test.ts src/cli/handlers/orchestration-worker-cli.test.ts src/main/runtime/rpc/methods/orchestration/worker/composed-workers.test.ts src/main/runtime/rpc/methods/orchestration/messaging/check.test.ts src/main/runtime/rpc/methods/orchestration/messaging/send.test.ts src/main/ssh/ssh-remote-orca-cli.test.ts",
"result": "passed",
"durationSeconds": 5.27,
"summary": "Five focused files passed with 216 tests, covering visible inactive local worker creation, reveal-failure warnings, stable-pane mailbox routing, live-handle precedence, and SSH fallback parity."
@@ -13704,7 +13739,7 @@
"date": "2026-08-12",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration-federation-lifecycle-settlement.test.ts --reporter=dot",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/federation/federation-lifecycle-settlement.test.ts --reporter=dot",
"result": "passed",
"durationSeconds": 4.58,
"summary": "Nine deterministic tests passed for protocol negotiation, Run-home completion and rejection, already-aborted waits, authoritative remote-attachment settlement bound to the exact queued worker_done outcome, and exact verdict replay after lost acknowledgments without mutating durable rejection mail twice."
@@ -13713,7 +13748,7 @@
"date": "2026-07-28",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration/formatter.test.ts src/main/runtime/rpc/methods/orchestration-federation.test.ts",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orchestration/formatter.test.ts src/main/runtime/rpc/methods/orchestration/federation/federation.test.ts",
"result": "passed",
"durationSeconds": 2.72,
"summary": "Two focused files passed with 34 tests, covering authority-aware legacy affordances and federated non-reveal."
@@ -13795,21 +13830,21 @@
"invariant": "A live Dispatch created by orchestration dispatch can be stopped or abandoned even though it has no supervised worker row. Release must durably record the requested outcome, revoke lifecycle authority, close questions, free the exact assignee identity, and block only the Task whose current Dispatch was released. It must never close the unsupervised terminal process, disturb unrelated or supervised workers, or let a repeat or opposite verb rewrite the persisted outcome.",
"oracle": "Create manual, unrelated, and supervised Dispatches through production runtime methods. Require dispatch-show to return the manual id while no worker row exists, then release it and require failed status with exact stopped or abandoned provenance, completion and revocation timestamps, one status notification, zero terminal closes, and immediate redispatch to the same terminal. Repeat through the opposite verb and require the first durable outcome. Create two active contexts for one Task through an explicit ready override, release the older context, and require only its identity to unlock while the newer context and Task remain dispatched. In an isolated Electron runtime, repeat both verbs against one real pane and require the same PTY/incarnation to survive before a third dispatch succeeds.",
"commands": [
- "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration-manual-dispatch-release.test.ts src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts src/main/runtime/rpc/methods/orchestration-workers-recovery.test.ts src/main/runtime/rpc/methods/orchestration-worker-release.test.ts src/cli/handlers/orchestration-worker-cli.test.ts --reporter=dot",
+ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-release.test.ts src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts src/main/runtime/rpc/methods/orchestration/worker/workers-recovery.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/cli/handlers/orchestration-worker-cli.test.ts --reporter=dot",
"pnpm run ensure:electron-runtime && pnpm exec playwright test tests/e2e/orchestration-low-level-dispatch-release.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1",
"SKIP_BUILD=1 pnpm exec playwright test tests/e2e/orchestration-low-level-dispatch-release.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1"
],
"testFiles": [
- "src/main/runtime/rpc/methods/orchestration-manual-dispatch-release.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-release.test.ts",
"src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts",
- "src/main/runtime/rpc/methods/orchestration-workers-recovery.test.ts",
- "src/main/runtime/rpc/methods/orchestration-worker-release.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/worker/workers-recovery.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts",
"src/cli/handlers/orchestration-worker-cli.test.ts",
"tests/e2e/orchestration-low-level-dispatch-release.spec.ts"
],
"assertionRefs": [
{
- "file": "src/main/runtime/rpc/methods/orchestration-manual-dispatch-release.test.ts",
+ "file": "src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-release.test.ts",
"assertions": [
"worker-abandon and worker-stop durably release context-only Dispatches without closing terminals",
"repeat and cross-verb calls preserve the first stored outcome",
@@ -13847,7 +13882,7 @@
"date": "2026-08-09",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration-manual-dispatch-release.test.ts src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts src/main/runtime/rpc/methods/orchestration-workers-recovery.test.ts src/main/runtime/rpc/methods/orchestration-worker-release.test.ts src/cli/handlers/orchestration-worker-cli.test.ts --reporter=dot",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-release.test.ts src/main/runtime/orchestration/orchestration-worker-dispatch-db.test.ts src/main/runtime/rpc/methods/orchestration/worker/workers-recovery.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/cli/handlers/orchestration-worker-cli.test.ts --reporter=dot",
"result": "passed",
"durationSeconds": 3.38,
"summary": "Five focused files passed 60 tests, including both context-only release verbs, stale/current ownership, question closure, repeat and cross-verb idempotency, supervised controls, terminal-close negative assertions, and text-mode retained-process guidance."
@@ -13920,17 +13955,19 @@
"invariant": "A settled Dispatch may close only its one coordinator-created terminal lease. Explicit reuse, real user input, retain, identity or host change, ambiguity, and another resource for the same exact host/pane/process must fence closure. Once the authoritative owning provider positively excludes the resource's exact immutable process incarnation, even an external, user-owned, or transferred dead resource must converge to released without any process close. Unknown host scope, missing incarnation metadata, or unavailable inventory must remain retained. Exact terminal-close persistence must settle when a host partition omits renderer-owned layout state. Output preservation and the requested-to-releasing transition are atomic, archives remain readable without the provider file, retries resume idempotently, and orchestration reset removes archive and authority state.",
"oracle": "Record release intent for a settled owner, attempt exact reuse before close, and require worker-start to fail with terminal_release_in_progress while the terminal stays open; then release the original owner exactly once. Race retain and real user input against a controlled archive promise and require no committed archive or close. Rebase a closed web-terminal host partition without terminalLayoutsByTabId and require the persistence write to complete while preserving host-authoritative membership; replay a valid legacy retirement under the same omission and require exact membership removal plus revision advancement. For retained external, user-owned, transferred, stopped, and abandoned resources, run one fresh inventory against the exact local/WSL or SSH provider: an exact live incarnation and every unknown inventory shape stay retained, while positive absence atomically sets ownership_state and release_state to released with processAction none and zero closeTerminal calls. Change host or process identity and inject duplicate resource evidence to require retention. Freeze a structured transcript, delete its source file, and require archived worker-read to return the same bounded redacted messages. Restart a pending mutation, reset orchestration state, and create 50 resources while asserting replay convergence, zero orphan rows, two-query worker listing, and no unrelated close.",
"commands": [
- "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/mobile-session-terminal-persistence-retirement.test.ts src/main/runtime/rpc/methods/orchestration-worker-release.test.ts src/main/runtime/rpc/methods/orchestration-worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
- "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/rpc/methods/orchestration-worker-release.test.ts src/main/runtime/rpc/methods/orchestration-worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
- "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration-worker-release.test.ts src/main/runtime/rpc/methods/orchestration-worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
+ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/pty-inventory-liveness-verdict.test.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/mobile-session-terminal-persistence-retirement.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
+ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/mobile-session-terminal-persistence-retirement.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
+ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
+ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
"pnpm exec vitest run --config config/vitest.config.ts tests/e2e/completed-worker-retirement-resume.unit.test.ts --reporter=verbose",
"pnpm run build:cli && SKIP_BUILD=1 pnpm exec playwright test tests/e2e/orchestration-worker-settlement-release-cli.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1"
],
"testFiles": [
+ "src/main/runtime/pty-inventory-liveness-verdict.test.ts",
"src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts",
"src/main/runtime/mobile-session-terminal-persistence-retirement.test.ts",
- "src/main/runtime/rpc/methods/orchestration-worker-release.test.ts",
- "src/main/runtime/rpc/methods/orchestration-worker-release-recovery.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts",
+ "src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts",
"src/main/runtime/rpc/orchestration-mutation-ledger.test.ts",
"src/main/runtime/orchestration/worker-transcript-read.test.ts",
"src/renderer/src/lib/worker-terminal-takeover-report.test.ts",
@@ -13938,6 +13975,14 @@
"tests/e2e/orchestration-worker-settlement-release-cli.spec.ts"
],
"assertionRefs": [
+ {
+ "file": "src/main/runtime/pty-inventory-liveness-verdict.test.ts",
+ "assertions": [
+ "320 simultaneously live PTYs retain truthful verdicts with linear identity checks and no detached history",
+ "400 unresolved PTY retirements preserve active doubt while bounding history at 256 entries",
+ "a replacement lifecycle clears the retained historical verdict for the reused PTY id"
+ ]
+ },
{
"file": "src/main/runtime/mobile-session-terminal-persistence-retirement.test.ts",
"assertions": [
@@ -13961,7 +14006,7 @@
]
},
{
- "file": "src/main/runtime/rpc/methods/orchestration-worker-release.test.ts",
+ "file": "src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts",
"assertions": [
"reconciles a dead external terminal without closing a process",
"reconciles a dead user-taken-over terminal without closing a process",
@@ -13980,7 +14025,7 @@
"assertions": ["resumes a pending idempotent worker release after restart"]
},
{
- "file": "src/main/runtime/rpc/methods/orchestration-worker-release-recovery.test.ts",
+ "file": "src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts",
"assertions": [
"finishes a requested release after restart-style interruption",
"coalesces overlapping reconciliation passes and closes each resource once",
@@ -14002,7 +14047,7 @@
"date": "2026-08-27",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/mobile-session-terminal-persistence-retirement.test.ts src/main/runtime/rpc/methods/orchestration-worker-release.test.ts src/main/runtime/rpc/methods/orchestration-worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/mobile-session-terminal-persistence-retirement.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
"result": "passed",
"durationSeconds": 8.78,
"summary": "Seven deterministic files passed 78 tests, including red-green host-partition rebase and legacy-retirement regressions with an absent web-terminal layout map plus exact lease, reuse, takeover, recovery, restart, archive, and accounting contracts."
@@ -14020,7 +14065,7 @@
"date": "2026-08-11",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/rpc/methods/orchestration-worker-release.test.ts src/main/runtime/rpc/methods/orchestration-worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-process-incarnation-liveness.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
"result": "passed",
"durationSeconds": 4.98,
"summary": "Six focused files passed 67 tests on the rebased candidate, covering dead external, user-owned, stopped, abandoned, and transferred reconciliation; exact local/WSL/SSH provider routing; malformed, missing, and unavailable inventory retention; zero process closes; existing lease, archive, recovery, mutation, and renderer-input contracts."
@@ -14029,7 +14074,7 @@
"date": "2026-08-03",
"runner": "local",
"platform": "macos",
- "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration-worker-release.test.ts src/main/runtime/rpc/methods/orchestration-worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
+ "command": "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release.test.ts src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts src/main/runtime/rpc/orchestration-mutation-ledger.test.ts src/main/runtime/orchestration/worker-transcript-read.test.ts src/renderer/src/lib/worker-terminal-takeover-report.test.ts --reporter=dot",
"result": "passed",
"durationSeconds": 3.48,
"summary": "Five focused files passed 56 tests covering lease serialization, reminted-handle transfer, duplicate-identity fencing, retain and takeover races, immutable archives, conservative legacy migration, mutation restart, reset cleanup, bounded accounting, and renderer input reporting."
@@ -14045,11 +14090,11 @@
},
"redGreenEvidence": {
"status": "complete",
- "evidence": "The version-skew legacy-retirement test deterministically threw at mobile-session-terminal-persistence-retirement.ts:75 before the null-safe layout read and passed with exact tab removal, tombstone cleanup, and topology-revision advancement after the fix. The byte-identical compiled-CLI Electron oracle left the dead resource external/retained on latest main 5ea7df1a5b, passed on combined candidate d697666ce8 with released/released SQLite state and processAction none, and reproduced external/retained after disabling the claimed production files at merge-base 64aec94cb2. The earlier unchanged three-case dead external/user-owned/transferred service oracle likewise failed 3/3 on main, passed 3/3 on candidate, and failed 3/3 with production restored; every run asserted durable state and zero terminal close calls."
+ "evidence": "The version-skew legacy-retirement test deterministically threw at mobile-session-terminal-persistence-retirement.ts:75 before the null-safe layout read and passed with exact tab removal, tombstone cleanup, and topology-revision advancement after the fix. The byte-identical compiled-CLI Electron oracle left the dead resource external/retained on latest main 5ea7df1a5b, passed on combined candidate d697666ce8 with released/released SQLite state and processAction none, and reproduced external/retained after disabling the claimed production files at merge-base 64aec94cb2. The earlier unchanged three-case dead external/user-owned/transferred service oracle likewise failed 3/3 on main, passed 3/3 on candidate, and failed 3/3 with production restored; every run asserted durable state and zero terminal close calls. The 320-live-PTY oracle failed on the prior single-map implementation and passes with complete active evidence, zero detached history, and a linear identity-check bound after the cache split."
},
"performanceBudget": {
"required": true,
- "evidence": "Normal owned release performs constant-count indexed resource and identity queries plus one bounded archive capture. Missing layout maps use constant-time empty-record fallbacks inside the existing explicit persistence pass, with no added scan or allocation proportional to terminal history. A retained release performs exactly one bounded inventory against its authoritative local/WSL or specific SSH provider, with no retry, polling, timer, subprocess, renderer subscription, or per-session follow-up fanout. Worker-list uses two set queries rather than one resource lookup per worker."
+ "evidence": "Normal owned release performs constant-count indexed resource and identity queries plus one bounded archive capture. Missing layout maps use constant-time empty-record fallbacks inside the existing explicit persistence pass, with no added scan or allocation proportional to terminal history. A retained release performs exactly one bounded inventory against its authoritative local/WSL or specific SSH provider, with no retry, polling, timer, subprocess, renderer subscription, or per-session follow-up fanout. Each liveness observation performs constant-time active-identity classification; retirement performs one historical insertion and at most one oldest-entry eviction, while active evidence scales only with supported PTYs and detached history is capped at 256. Worker-list uses two set queries rather than one resource lookup per worker."
},
"promotionCriteria": [
"Collect 100 consecutive focused CI passes or 14 days of soak history.",
diff --git a/config/scripts/generate-bundled-skill-guides.mjs b/config/scripts/generate-bundled-skill-guides.mjs
index bc44f5e72d6..abc172eb100 100644
--- a/config/scripts/generate-bundled-skill-guides.mjs
+++ b/config/scripts/generate-bundled-skill-guides.mjs
@@ -101,29 +101,112 @@ function constantName(name) {
return `${name.replace(/-/g, '_').toUpperCase()}_MARKDOWN`
}
-function serializeEmbeddedModule(guides) {
- const markdownConstants = guides
+function fullConstantName(name) {
+ return `${name.replace(/-/g, '_').toUpperCase()}_FULL_MARKDOWN`
+}
+
+function referenceConstantName(guideName, referenceName) {
+ return `${`${guideName}_${referenceName}`.replace(/-/g, '_').toUpperCase()}_REFERENCE_MARKDOWN`
+}
+
+function composeFullMarkdown(markdown, references) {
+ if (references.length === 0) {
+ return markdown
+ }
+ const packageHeader =
+ '\n\n---\n\n# Bundled references\n\n' +
+ 'These references belong to the version-matched guide above. Read only the documents ' +
+ 'named by its action gates.\n'
+ const documents = references
.map(
- (guide) =>
- `// oxfmt-ignore\nconst ${constantName(guide.name)} = ${JSON.stringify(guide.markdown)}`
+ ({ relativePath, markdown: referenceMarkdown }) =>
+ `\n\n\n${referenceMarkdown.trimEnd()}\n`
)
+ .join('')
+ return `${markdown.trimEnd()}${packageHeader}${documents}`
+}
+
+function serializeEmbeddedModule(guides) {
+ const referenceConstants = guides.flatMap((guide) =>
+ guide.references.map((reference) => referenceConstantName(guide.name, reference.name))
+ )
+ // Why: the constant name flattens guide and reference names, so two topics could otherwise
+ // produce one identifier and silently serve the wrong reference.
+ if (new Set(referenceConstants).size !== referenceConstants.length) {
+ throw new Error(`Guide reference constant names collide: ${referenceConstants.join(', ')}`)
+ }
+ const markdownConstants = guides
+ .flatMap((guide) => {
+ const constants = [
+ `// oxfmt-ignore\nconst ${constantName(guide.name)} = ${JSON.stringify(guide.markdown)}`
+ ]
+ if (guide.fullMarkdown !== guide.markdown) {
+ constants.push(
+ `// oxfmt-ignore\nconst ${fullConstantName(guide.name)} = ${JSON.stringify(guide.fullMarkdown)}`
+ )
+ }
+ for (const reference of guide.references) {
+ constants.push(
+ `// oxfmt-ignore\nconst ${referenceConstantName(guide.name, reference.name)} = ${JSON.stringify(reference.markdown)}`
+ )
+ }
+ return constants
+ })
.join('\n\n')
const guideEntries = guides
.map((guide) => {
const markdownConstant = constantName(guide.name)
+ const referenceEntries = guide.references
+ .map(
+ (reference) =>
+ `{ name: ${JSON.stringify(reference.name)}, markdown: ${referenceConstantName(guide.name, reference.name)} }`
+ )
+ .join(', ')
return [
' {',
` name: ${JSON.stringify(guide.name)},`,
` description: ${JSON.stringify(guide.description)},`,
` markdown: ${markdownConstant},`,
- ` fullMarkdown: ${markdownConstant},`,
- ` aliases: ${JSON.stringify(guide.aliases)}`,
+ ` fullMarkdown: ${guide.fullMarkdown === guide.markdown ? markdownConstant : fullConstantName(guide.name)},`,
+ ` aliases: ${JSON.stringify(guide.aliases)},`,
+ ` references: [${referenceEntries}]`,
' }'
].join('\n')
})
.join(',\n')
- return `// Generated by config/scripts/generate-bundled-skill-guides.mjs. Do not edit.\n\nexport type BundledSkillGuide = {\n readonly name: string\n readonly description: string\n readonly markdown: string\n readonly fullMarkdown: string\n readonly aliases: readonly string[]\n}\n\n${markdownConstants}\n\n// Why: no current guide has bundled reference documents, so --full is byte-identical for now.\n// oxfmt-ignore\nexport const BUNDLED_SKILL_GUIDES = [\n${guideEntries}\n] as const satisfies readonly BundledSkillGuide[]\n`
+ return `// Generated by config/scripts/generate-bundled-skill-guides.mjs. Do not edit.\n\nexport type BundledSkillGuideReference = {\n readonly name: string\n readonly markdown: string\n}\n\nexport type BundledSkillGuide = {\n readonly name: string\n readonly description: string\n readonly markdown: string\n readonly fullMarkdown: string\n readonly aliases: readonly string[]\n readonly references: readonly BundledSkillGuideReference[]\n}\n\n${markdownConstants}\n\n// oxfmt-ignore\nexport const BUNDLED_SKILL_GUIDES = [\n${guideEntries}\n] as const satisfies readonly BundledSkillGuide[]\n`
+}
+
+async function readGuideReferences(repoRoot, guideName) {
+ const referenceRoot = path.join(repoRoot, 'skill-guides', guideName, 'references')
+ let entries
+ try {
+ entries = await readdir(referenceRoot, { withFileTypes: true })
+ } catch (error) {
+ if (error.code === 'ENOENT') {
+ return []
+ }
+ throw error
+ }
+ const unsupported = entries.find((entry) => !entry.isFile() || !entry.name.endsWith('.md'))
+ if (unsupported) {
+ throw new Error(
+ `Guide references must be Markdown files: skill-guides/${guideName}/references/${unsupported.name}`
+ )
+ }
+ return Promise.all(
+ entries
+ .sort((left, right) => left.name.localeCompare(right.name, 'en'))
+ .map(async (entry) => {
+ const sourcePath = path.join(referenceRoot, entry.name)
+ const markdown = normalizeMarkdown(await readFile(sourcePath, 'utf8'))
+ if (!markdown.trim()) {
+ throw new Error(`Guide reference is empty: ${toPosixRelativePath(repoRoot, sourcePath)}`)
+ }
+ return { name: entry.name.slice(0, -3), relativePath: `references/${entry.name}`, markdown }
+ })
+ )
}
function assertAliasContract(guides) {
@@ -204,9 +287,22 @@ async function buildArtifacts(repoRoot = REPO_ROOT) {
throw new Error(`Guide source ${name}.md declares mismatched name ${frontmatter.name}`)
}
const aliases = GUIDE_ALIASES[name]
+ const references = await readGuideReferences(repoRoot, name)
// Why: the embedded table always carries the full guide (served by `skills get`);
// only the installable projection thins to a stub once a topic is in STUB_TOPICS.
- guides.push({ name, description: frontmatter.description, markdown, aliases })
+ guides.push({
+ name,
+ description: frontmatter.description,
+ markdown,
+ fullMarkdown: composeFullMarkdown(markdown, references),
+ aliases,
+ // Why: `skills get --reference` serves one of these alone, so it keeps the
+ // per-file identity that fullMarkdown's concatenation erases.
+ references: references.map(({ name: referenceName, markdown: referenceMarkdown }) => ({
+ name: referenceName,
+ markdown: referenceMarkdown
+ }))
+ })
const stubPath = path.join(repoRoot, 'skill-stubs', `${name}.md`)
const content = stubTopics.has(name)
? composeStubProjection(markdown, await readFile(stubPath, 'utf8'), `skill-stubs/${name}.md`)
@@ -273,6 +369,7 @@ export {
STUB_TOPICS,
assertAliasContract,
buildArtifacts,
+ composeFullMarkdown,
composeStubProjection,
frontmatterBlock,
normalizeMarkdown,
diff --git a/config/scripts/generate-bundled-skill-guides.test.mjs b/config/scripts/generate-bundled-skill-guides.test.mjs
index 6b90a499d90..24fe63de873 100644
--- a/config/scripts/generate-bundled-skill-guides.test.mjs
+++ b/config/scripts/generate-bundled-skill-guides.test.mjs
@@ -22,6 +22,15 @@ import {
const projectDir = path.resolve(import.meta.dirname, '..', '..')
const temporaryDirectories = []
const execFileAsync = promisify(execFile)
+const ORCHESTRATION_REFERENCES = [
+ 'coordinator-loop.md',
+ 'legacy-contract-migration.md',
+ 'low-level-topology.md',
+ 'messaging-and-gates.md',
+ 'placement-and-remote.md',
+ 'recovery-and-cleanup.md',
+ 'worker-contract.md'
+]
async function createFixture() {
const root = await mkdtemp(path.join(tmpdir(), 'orca-bundled-skill-guides-'))
@@ -181,7 +190,7 @@ describe('bundled skill guide generator', () => {
}
)
- it('embeds canonical names, discovery descriptions, Markdown, and append-only aliases', async () => {
+ it('embeds compact guides, version-matched reference packages, and append-only aliases', async () => {
expect(BUNDLED_SKILL_GUIDES.map((guide) => guide.name)).toEqual(
[...CANONICAL_GUIDE_NAMES].sort((left, right) => left.localeCompare(right, 'en'))
)
@@ -194,8 +203,46 @@ describe('bundled skill guide generator', () => {
const frontmatter = parseFrontmatter(source, `${guide.name}.md`)
expect(guide.description).toBe(frontmatter.description)
expect(guide.markdown).toBe(source)
- expect(guide.fullMarkdown).toBe(source)
expect(guide.aliases).toEqual(GUIDE_ALIASES[guide.name])
+ if (guide.name !== 'orchestration') {
+ expect(guide.fullMarkdown).toBe(source)
+ expect(guide.references).toEqual([])
+ continue
+ }
+ // Why: the per-reference selector serves these verbatim, so an entry that
+ // drifts from the file on disk ships a stale reference to every agent.
+ expect(guide.references.map((reference) => reference.name)).toEqual(
+ ORCHESTRATION_REFERENCES.map((reference) => reference.replace(/\.md$/u, ''))
+ )
+ for (const reference of guide.references) {
+ expect(reference.markdown).toBe(
+ normalizeMarkdown(
+ await readFile(
+ path.join(
+ projectDir,
+ 'skill-guides',
+ 'orchestration',
+ 'references',
+ `${reference.name}.md`
+ ),
+ 'utf8'
+ )
+ )
+ )
+ }
+ expect(guide.fullMarkdown).not.toBe(guide.markdown)
+ expect(guide.fullMarkdown.length).toBeGreaterThan(guide.markdown.length)
+ expect(guide.fullMarkdown.startsWith(source.trimEnd())).toBe(true)
+ for (const reference of ORCHESTRATION_REFERENCES) {
+ const marker = ``
+ expect(guide.fullMarkdown.split(marker)).toHaveLength(2)
+ expect(guide.fullMarkdown).toContain(
+ await readFile(
+ path.join(projectDir, 'skill-guides', 'orchestration', 'references', reference),
+ 'utf8'
+ )
+ )
+ }
}
})
@@ -237,6 +284,17 @@ describe('bundled skill guide generator', () => {
const stubSource = await readFile(stubPath, 'utf8')
await writeFile(stubPath, stubSource.replaceAll('\n', '\r\n'))
}
+ for (const reference of ORCHESTRATION_REFERENCES) {
+ const referencePath = path.join(
+ root,
+ 'skill-guides',
+ 'orchestration',
+ 'references',
+ reference
+ )
+ const source = await readFile(referencePath, 'utf8')
+ await writeFile(referencePath, source.replaceAll('\n', '\r\n'))
+ }
const actual = await buildArtifacts(root)
expect(actual.map((artifact) => artifact.content)).toEqual(
@@ -303,4 +361,15 @@ describe('bundled skill guide generator', () => {
])
).toThrow('collides with canonical name')
})
+
+ it('rejects non-Markdown and empty bundled references', async () => {
+ const root = await createFixture()
+ const referenceRoot = path.join(root, 'skill-guides', 'orchestration', 'references')
+
+ await writeFile(path.join(referenceRoot, 'notes.txt'), 'not a reference\n')
+ await expect(buildArtifacts(root)).rejects.toThrow('Guide references must be Markdown files')
+ await rm(path.join(referenceRoot, 'notes.txt'))
+ await writeFile(path.join(referenceRoot, 'empty.md'), '\n')
+ await expect(buildArtifacts(root)).rejects.toThrow('Guide reference is empty')
+ })
})
diff --git a/config/scripts/orca-cli-skill-guidance.test.mjs b/config/scripts/orca-cli-skill-guidance.test.mjs
index 28c50c2daf3..d8c48e8b77c 100644
--- a/config/scripts/orca-cli-skill-guidance.test.mjs
+++ b/config/scripts/orca-cli-skill-guidance.test.mjs
@@ -10,7 +10,14 @@ const guidePath = join(projectDir, 'skill-guides', 'orca-cli.md')
const stubPath = join(projectDir, 'skills', 'orca-cli', 'SKILL.md')
// Why: orchestration and orca-emulator also ship hybrid stubs now, so their version-sensitive
// command guidance lives in the guide sources — read the cross-guide worktree-id contract there.
-const orchestrationSkillPath = join(projectDir, 'skill-guides', 'orchestration.md')
+// Why: the worktree-selector rule lives in the orchestration placement reference, not the kernel.
+const orchestrationPlacementPath = join(
+ projectDir,
+ 'skill-guides',
+ 'orchestration',
+ 'references',
+ 'placement-and-remote.md'
+)
const emulatorSkillPath = join(projectDir, 'skill-guides', 'orca-emulator.md')
function readSkill(path = guidePath) {
@@ -95,7 +102,7 @@ describe('orca CLI skill guidance', () => {
it('requires full worktree ids across bundled agent guidance', () => {
const cliSkill = readSkill()
- const orchestrationSkill = readSkill(orchestrationSkillPath)
+ const orchestrationSkill = readSkill(orchestrationPlacementPath)
const emulatorSkill = readSkill(emulatorSkillPath)
for (const skill of [cliSkill, orchestrationSkill, emulatorSkill]) {
diff --git a/config/scripts/orchestration-guide-command-contract.test.mjs b/config/scripts/orchestration-guide-command-contract.test.mjs
new file mode 100644
index 00000000000..89a3b99097f
--- /dev/null
+++ b/config/scripts/orchestration-guide-command-contract.test.mjs
@@ -0,0 +1,38 @@
+import { readFileSync, readdirSync } from 'node:fs'
+import { join, resolve } from 'node:path'
+import { describe, expect, it } from 'vitest'
+import { ORCHESTRATION_COMMAND_SPECS } from '../../src/cli/specs/orchestration'
+
+const projectDir = resolve(import.meta.dirname, '../..')
+const guideRoot = join(projectDir, 'skill-guides', 'orchestration')
+const guidePaths = [
+ join(projectDir, 'skill-guides', 'orchestration.md'),
+ ...readdirSync(join(guideRoot, 'references')).map((name) => join(guideRoot, 'references', name))
+]
+
+function documentedInvocations() {
+ return guidePaths.flatMap((path) => {
+ const text = readFileSync(path, 'utf8')
+ return [...text.matchAll(/ORCA orchestration ([a-z-]+)([^`\n]*)/gu)].map((match) => ({
+ path,
+ verb: match[1],
+ flags: [...match[2].matchAll(/(?:^|\s)--([a-z][a-z-]*)/gu)].map((flag) => flag[1])
+ }))
+ })
+}
+
+describe('orchestration guide command contract', () => {
+ it('documents only orchestration verbs and flags accepted by the CLI specs', () => {
+ const specs = new Map(
+ ORCHESTRATION_COMMAND_SPECS.map((spec) => [spec.path[1], new Set(spec.allowedFlags)])
+ )
+
+ for (const invocation of documentedInvocations()) {
+ const allowed = specs.get(invocation.verb)
+ expect(allowed, `${invocation.path}: ${invocation.verb}`).toBeDefined()
+ for (const flag of invocation.flags) {
+ expect(allowed, `${invocation.path}: ${invocation.verb} --${flag}`).toContain(flag)
+ }
+ }
+ })
+})
diff --git a/config/scripts/orchestration-skill-guidance.test.mjs b/config/scripts/orchestration-skill-guidance.test.mjs
index 9d86471bc00..e84697255a5 100644
--- a/config/scripts/orchestration-skill-guidance.test.mjs
+++ b/config/scripts/orchestration-skill-guidance.test.mjs
@@ -1,32 +1,58 @@
-import { readFileSync } from 'node:fs'
+import { readFileSync, readdirSync } from 'node:fs'
import { join, resolve } from 'node:path'
import { describe, expect, it } from 'vitest'
const projectDir = resolve(import.meta.dirname, '../..')
-// Why: orchestration now ships a hybrid discovery stub, so its version-sensitive command
-// guidance lives in the authoritative guide source — assert that content there. The
-// installable stub projection is checked separately below.
const guidePath = join(projectDir, 'skill-guides', 'orchestration.md')
+const referenceRoot = join(projectDir, 'skill-guides', 'orchestration', 'references')
const stubPath = join(projectDir, 'skills', 'orchestration', 'SKILL.md')
-function readSkill() {
+function readKernel() {
return readFileSync(guidePath, 'utf8')
}
-function getSection(markdown, heading) {
- const escapedHeading = heading.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
- const match = markdown.match(
- new RegExp(`## ${escapedHeading}\\r?\\n([\\s\\S]*?)(?=\\r?\\n## |$)`)
- )
-
- expect(match).not.toBeNull()
-
- return match?.[1] ?? ''
+function readReference(name) {
+ return readFileSync(join(referenceRoot, name), 'utf8')
}
-describe('orchestration skill guidance', () => {
+function frontmatter(text) {
+ return /^---\n[\s\S]*?\n---\n/u.exec(text)?.[0]
+}
+
+function squash(text) {
+ return text.replace(/\s+/gu, ' ').trim()
+}
+
+// Routing lives in the frontmatter description alone; the body must not satisfy these.
+function readDescription() {
+ return squash(frontmatter(readKernel()))
+}
+
+describe('orchestration skill routing', () => {
+ it('keeps the verbatim routing triggers a model matches the skill on', () => {
+ const description = readDescription()
+
+ for (const trigger of [
+ 'threaded messages',
+ 'worker_done/escalation waits',
+ 'decision gates',
+ 'decomposing work across agents',
+ '"hand off"',
+ '"handoff"',
+ '"handover"',
+ '"give this to another agent"',
+ '"another worktree"',
+ 'lightweight terminal prompts',
+ 'shell commands',
+ 'Orca worktree management',
+ 'reading or waiting on terminals'
+ ]) {
+ expect(description).toContain(trigger)
+ }
+ })
+
it('keeps external browser routing at the OS/page boundary', () => {
- const description = readFileSync(guidePath, 'utf8').replace(/\s+/gu, ' ')
+ const description = readDescription()
expect(description).toContain(
"Use Computer Use for external browser windows, webviews, Orca app UI, or desktop UI outside Orca's embedded browser only when the task requires OS/window-level control such as focus, menus, dialogs, coordinates, or screenshots."
@@ -35,383 +61,444 @@ describe('orchestration skill guidance', () => {
"`orca-cli` for Orca's embedded pages and a page-automation tool such as Playwright or CDP for external pages."
)
})
+})
- it('requires Orca runtime state before claiming a worker was orchestrated', () => {
- const skill = readSkill()
- const toolBoundary = getSection(skill, 'Tool Boundary')
+describe('orchestration kernel', () => {
+ it('keeps the always-loaded guide compact and ordered around the normal protocol', () => {
+ const kernel = readKernel()
+ const headings = [
+ '## Outcome',
+ '## Classify the role',
+ '## Authority and safety floor',
+ '## Worker obligations',
+ '## Canonical supervised loop',
+ '## Task-spec contract',
+ '## Completion accounting',
+ '## Conditional references'
+ ]
- expect(toolBoundary).toContain('must create or bind a Run')
- expect(toolBoundary).toContain('create the Task with `orca orchestration task-create`')
- expect(toolBoundary).toContain('preferred `orca orchestration worker-start` composition')
- expect(toolBoundary).toContain('low-level `orca orchestration dispatch --inject` path')
- expect(toolBoundary).not.toContain('or `orca orchestration run`')
- expect(skill).toContain(
- '`coordinator-start`, `coordinator-stop`, `run`, and `run-stop` are retired scheduler commands'
- )
- expect(toolBoundary).toContain(
- 'Do not substitute non-Orca subagent tools, generic agent-spawn APIs, or chat-only parallel worker features'
- )
- expect(toolBoundary).toContain('do not create Orca task/dispatch provenance')
- expect(toolBoundary).toContain('injected lifecycle preambles')
- expect(toolBoundary).toContain('`worker_done` authority')
- expect(toolBoundary).toContain('decision gates')
- expect(toolBoundary).toContain('orca orchestration task-list --json')
- expect(toolBoundary).toContain('orca orchestration dispatch-show --task --json')
- expect(toolBoundary).toContain(
- 'do not retroactively describe the external worker as orchestrated'
- )
- })
-
- it('teaches attested adoption without reviving the retired scheduler', () => {
- const skill = readSkill()
- const migration = getSection(skill, 'Contract Migration')
-
- expect(migration).toContain(
- 'adopts a live pre-update orchestration assignment into an ordinary Run'
- )
- expect(migration).toContain(
- 'preserves the existing agent process, PTY/session, terminal handle, tab/leaf/pane, worktree or folder workspace, Task, and Dispatch'
- )
- expect(migration).toContain('never restarts or replaces the worker')
- expect(migration).toContain('The retired scheduler is not revived')
- expect(migration).toContain('[LEGACY COMPATIBILITY]')
- expect(migration).toContain('[LEGACY READ-ONLY]')
- expect(migration).toContain(
- 'Loss of lifecycle authority does not invalidate the existing assignment, process, or filesystem work.'
- )
- expect(migration).toContain(
- 'It must not spawn, write, signal, stop, switch, focus, split, or inject a terminal.'
- )
- expect(migration).not.toContain('task-list --run run_legacy_local')
- expect(migration).toContain('run_legacy_local is an empty audit tombstone')
- expect(migration).toContain('Recovered orchestration work from a contract update')
- expect(migration).toContain('run-show --id ')
- expect(migration).toContain('task-list --run ')
- expect(migration).toContain('Legacy inspection remains available without consuming mail')
- expect(migration).toContain('run-use --id --takeover-legacy')
- expect(migration).toContain('Takeover fences only the old coordinator')
- expect(migration).toContain('Live legacy workers keep their original Tasks, Dispatches')
- expect(migration).toContain(
- 'keep the original worker as the only editor until it reaches a stable handoff point'
- )
- expect(migration).toContain('a conflict-free placement for any remaining work')
- })
-
- it('treats long-running worker waits as liveness checkpoints, not failures', () => {
- const skill = readSkill()
-
- expect(skill).toContain('Treat a `check --wait` timeout or `{count:0}` as a checkpoint')
- expect(skill).toContain('Do not stop, close, kill, or restart a worker')
- expect(skill).toContain('keep waiting instead of retrying the task')
- expect(skill).not.toContain(
- 'If `check --wait` times out with no `worker_done` or `escalation`, fall back to `terminal wait --for tui-idle`, then `terminal read`.'
- )
- })
-
- it('keeps full handoffs out of dispatch lifecycle and off the active branch base', () => {
- const skill = readSkill()
- const fullHandoffs = getSection(skill, 'Full Handoffs')
-
- expect(skill).toContain('Full handoff means ownership transfer, not supervised dispatch.')
- expect(fullHandoffs).toContain(
- 'Do not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs.'
- )
- expect(fullHandoffs).toContain(
- '`task-create` is also forbidden because it records coordinator-owned tracking state'
- )
- expect(fullHandoffs).toContain('Do not create a `taskId`/`dispatchId`')
- expect(fullHandoffs).toContain(
- 'read the worker terminal after prompt delivery except to avoid losing the initial prompt'
- )
- expect(skill).toContain(
- '`--no-parent` only controls Orca lineage; it does not choose the Git base.'
- )
- expect(skill).toContain(
- 'never base it on the current feature branch unless the user explicitly asks'
- )
- expect(skill).toContain(
- 'orca worktree create --name --no-parent --agent codex --prompt'
- )
- expect(fullHandoffs).toContain(
- 'Before creating a new worktree from an active feature branch, decide and state whether the desired Orca lineage is child or top-level'
- )
- expect(fullHandoffs).toContain(
- 'Use child worktree lineage only when the new work is conceptually stacked under or dependent on the active worktree'
- )
- expect(fullHandoffs).toContain(
- 'For independent repo-wide fixes, standalone feature work, or unrelated follow-up tasks, create a top-level worktree with `--no-parent`'
- )
- expect(fullHandoffs).toContain('If the work should start from the repo default base')
- expect(fullHandoffs).toContain('omit `--base-branch`')
- })
-
- it('classifies handoff wording as ownership transfer unless supervision is explicit', () => {
- const skill = readSkill()
- const fullHandoffs = getSection(skill, 'Full Handoffs')
-
- for (const phrase of [
- 'hand off',
- 'handoff',
- 'handover',
- 'give this to another agent',
- 'give this to another worktree',
- 'another agent',
- 'another worktree'
- ]) {
- expect(fullHandoffs).toContain(phrase)
+ // Why: 202 is the budget after the anti-loop nextAction rule; the kernel is always in context.
+ expect(kernel.split('\n').length).toBeLessThanOrEqual(202)
+ for (let index = 1; index < headings.length; index += 1) {
+ expect(kernel.indexOf(headings[index])).toBeGreaterThan(kernel.indexOf(headings[index - 1]))
}
+ expect(kernel).not.toContain('## Contract Migration')
+ expect(kernel).not.toContain('## Full Handoffs')
+ expect(kernel).not.toContain('## Worker Terminals')
+ })
- for (const supervisionPhrase of [
- 'supervise',
- 'monitor',
- 'wait for worker_done',
- 'wait for results',
- 'track completion',
- 'DAG',
- 'decision gate',
- 'ask/reply'
+ it('classifies coordinator, dispatched worker, handoff, compatibility, and ordinary roles', () => {
+ const kernel = readKernel()
+
+ expect(kernel).toContain('explicitly asks to supervise, monitor, wait for results')
+ expect(kernel).toContain('live injected preamble with Task and Dispatch IDs')
+ expect(kernel).toContain('Handoff owner')
+ expect(kernel).toContain('create no Run, Task, or Dispatch and do not monitor completion')
+ expect(kernel).toContain('Compatibility operator')
+ expect(kernel).toContain('Ordinary terminal agent')
+ expect(kernel).toContain('Model or effort selection does not make a handoff supervised')
+ expect(squash(kernel)).toContain('Never substitute a non-Orca subagent tool')
+ })
+
+ it('makes Dispatch identity, remote uncertainty, folders, and mixed versions a safety floor', () => {
+ const kernel = readKernel()
+
+ expect(kernel).toContain('A Dispatch is one authoritative Task attempt')
+ expect(kernel).toContain('Lifecycle authority comes from the active Dispatch')
+ expect(kernel).toContain('execution host owns')
+ expect(squash(kernel)).toContain('`live` / `unverifiable` / `exited`')
+ expect(kernel).toContain('contact loss is not process death')
+ expect(kernel).toContain('Folder workspaces are valid')
+ expect(squash(kernel)).toContain('Treat unknown optional fields as absent')
+ expect(kernel).toContain('new stream operation requires advertised capability')
+ expect(kernel).toContain('Never fall back to local execution')
+ })
+
+ it('puts exactly-once worker completion and post-completion idle before coordinator mechanics', () => {
+ const kernel = readKernel()
+
+ expect(kernel.indexOf('## Worker obligations')).toBeLessThan(
+ kernel.indexOf('## Canonical supervised loop')
+ )
+ expect(kernel).toContain('The injected preamble is authoritative')
+ expect(kernel).toContain('Send `worker_done` exactly once')
+ expect(kernel).toContain('three-sentence executive summary')
+ expect(kernel).toContain('`--outcome succeeded` or `--outcome failed`')
+ // Why: the runnable worker_done command is the preamble's; its flag spellings are pinned
+ // on worker-contract.md by 'keeps heartbeat and worker_done recipes bound to the injected
+ // capability', so the kernel carries the obligations as prose and no third copy.
+ expect(kernel).not.toContain('--type worker_done')
+ expect(kernel).toContain('After `worker_done`, end the dispatched turn and idle')
+ expect(kernel).toContain('Do not reuse the settled lifecycle IDs')
+ })
+
+ it('teaches worker-start as the only normal-path launch and starts the wave before waiting', () => {
+ const kernel = readKernel()
+ const firstStart = kernel.indexOf('worker-start --spec ""')
+ const secondStart = kernel.indexOf('worker-start --spec ""')
+ const firstWait = kernel.indexOf('check --wait')
+
+ expect(firstStart).toBeGreaterThan(kernel.indexOf('run-create'))
+ expect(secondStart).toBeGreaterThan(firstStart)
+ expect(firstWait).toBeGreaterThan(secondStart)
+ expect(squash(kernel)).toContain('start the full independent wave before waiting')
+ expect(kernel).toContain('`worker-start` is the normal path')
+ expect(squash(kernel)).toContain(
+ "If `worker-start` exits non-zero, do not relaunch. Read the receipt's `failedStage` and `residualResources`"
+ )
+ expect(kernel).toContain('operator-created process unsupervised')
+ expect(kernel).not.toMatch(/^ORCA terminal create/mu)
+ })
+
+ it('makes worker-start --spec the default and keeps task-create for planned fan-out', () => {
+ const kernel = squash(readKernel())
+
+ expect(kernel).toContain('`worker-start --spec` creates the Task and its attempt in one call')
+ expect(kernel).toContain('Use `task-create` plus `worker-start --task `')
+ })
+
+ it('gives the supervised loop an exit condition for a live terminal with a dead agent', () => {
+ const kernel = squash(readKernel())
+
+ expect(kernel).toContain("`worker-list`'s `projection.liveness` is the fleet verdict")
+ expect(kernel).toContain("`worker-show`'s `observation.status` is PTY liveness only")
+ expect(kernel).toContain('After three consecutive empty waits')
+ expect(kernel).toContain('`ORCA orchestration worker-list --include-remote --json`')
+ expect(kernel).toContain('defaults to the bound Run; `--run ` overrides')
+ expect(kernel).toContain(
+ '`projection.attention` categories, `projection.attention.requiresAction`, and literal `projection.nextAction` argv'
+ )
+ expect(kernel).toContain(
+ 'An `inspect` `nextAction` on a `live` row with `attention.requiresAction` false is informational, not a command to re-run: keep waiting with `check --wait`'
+ )
+ expect(kernel).toContain('choose `worker-stop` or `worker-abandon`')
+ })
+
+ it('lets only positive evidence of exit end a wait', () => {
+ const kernel = squash(readKernel())
+
+ expect(kernel).toContain('Leave the wait only on positive proof the agent stopped')
+ expect(kernel).toContain('`exited` liveness')
+ expect(kernel).toContain("the worker's own observation of process exit")
+ expect(kernel).toContain('transcript whose final agent turn sent no `worker_done`')
+ expect(kernel).toContain(
+ '`unverifiable` is absence, including when `worker-show` reports `agentWait` null. Absence never authorizes stop, abandon, retry, or release'
+ )
+ })
+
+ it('names --terminal, never --from, as the check caller flag', () => {
+ const kernel = squash(readKernel())
+
+ expect(kernel).toContain('`check` names its caller with `--terminal `, never `--from`')
+ expect(kernel).not.toContain('check --from')
+ })
+
+ it('makes a dispatched worker read coordinator follow-ups on a cadence', () => {
+ const kernel = squash(readKernel())
+
+ expect(kernel).toContain('Read coordinator follow-ups at each natural checkpoint')
+ expect(kernel).toContain('once more immediately before `worker_done`')
+ expect(kernel).toContain('`ORCA orchestration check --terminal --json`')
+ })
+
+ it('requires full Delivery processing and settled-terminal accounting before ack', () => {
+ const kernel = readKernel()
+
+ expect(squash(kernel)).toContain(
+ 'oldest FIFO Delivery and replays that batch until acknowledged'
+ )
+ expect(squash(kernel)).toContain('Process every message')
+ expect(squash(kernel)).toContain("decide each settled terminal's next owner before the ack")
+ expect(squash(kernel)).toContain('reused, explicitly retained, or released')
+ expect(squash(kernel)).toContain(
+ 'the turn ends only when the report to that user names, per Task, its outcome, the evidence behind it, and any unresolved blocker'
+ )
+ expect(kernel).toContain('worker-release --dispatch ')
+ expect(kernel).toContain('check --ack --wait')
+ expect(squash(kernel)).toContain(
+ '`worker-list --run --terminal-state reclaimable --json`'
+ )
+ expect(squash(kernel)).toContain('do not follow it with `task-update --status completed`')
+ })
+
+ it('treats long waits and release uncertainty as safe checkpoints', () => {
+ const kernel = readKernel()
+
+ // Why: e92d7812d91 and c78f40fdd0b protect one rule; `## Outcome` states it once and each
+ // gate cites it, so these pin the condition rather than a per-gate list of non-proofs.
+ expect(squash(kernel)).toContain(
+ 'Only positive proof of exit authorizes stop, abandon, or retry, and only an accepted settlement authorizes release. Every other observation, absence included, is a checkpoint'
+ )
+ expect(squash(kernel)).toContain('A timeout or empty result is a checkpoint, not a failure')
+ expect(squash(kernel)).toContain('Do not stop, retry, release, or launch a duplicate editor')
+ expect(squash(kernel)).toContain('without the positive proof `## Outcome` requires')
+ expect(squash(kernel)).toContain(
+ 'Only an accepted settlement authorizes it; no other observation does'
+ )
+ expect(kernel).toContain('never substitute `terminal close`')
+ })
+
+ it('defines self-contained task specs and honest send attention semantics', () => {
+ const kernel = readKernel()
+
+ for (const field of [
+ '**Target:**',
+ '**Change:**',
+ '**Constraints:**',
+ '**Ownership:**',
+ '**Observable acceptance:**'
]) {
- expect(fullHandoffs).toContain(supervisionPhrase)
+ expect(kernel).toContain(field)
}
+ expect(kernel).toContain('successful `orchestration send` proves durable enqueue')
+ expect(kernel).toContain('best-effort attention only')
+ expect(squash(kernel)).toContain('does not prove the recipient read or accepted it')
+ })
+})
+
+describe('owned orchestration references', () => {
+ it('routes every conditional read to exactly one shipped reference', () => {
+ const kernel = readKernel()
+ const routed = [...kernel.matchAll(/`references\/([^`]+\.md)`/gu)].map((match) => match[1])
+ const shipped = readdirSync(referenceRoot)
+ .filter((name) => name.endsWith('.md'))
+ .sort()
+
+ const tableRoutes = [...kernel.matchAll(/^\|.*`references\/([^`]+\.md)`.*\|$/gmu)].map(
+ (match) => match[1]
+ )
+
+ expect([...new Set(routed)].sort()).toEqual(shipped)
+ // Why the table and not every mention: prose may cite a reference the gate table already routes.
+ expect(tableRoutes.sort()).toEqual(shipped)
+ expect(kernel).toContain('ORCA skills get orchestration --full')
+ // Why: the selector is the cheap path, so the kernel must teach it first and keep
+ // `--full` only as the fallback for a CLI build that predates it.
+ expect(squash(kernel)).toContain(
+ 'run `ORCA skills get orchestration --reference references/.md`'
+ )
+ expect(squash(kernel)).toContain(
+ 'If the CLI rejects `--reference`, run `ORCA skills get orchestration --full`'
+ )
+ expect(squash(kernel)).toContain('If an older CLI rejects `--full`')
})
- it('documents custom model and effort handoffs without completion monitoring', () => {
- const skill = readSkill()
- const fullHandoffs = getSection(skill, 'Full Handoffs')
+ it('owns expanded waves, launch preferences, reuse, and review boundaries', () => {
+ const reference = readReference('coordinator-loop.md')
- expect(fullHandoffs).toContain('Custom Codex model/effort handoff')
- expect(fullHandoffs).toContain(
- 'does not accept Codex-specific `--model` or `-c model_reasoning_effort=...` arguments'
- )
- expect(fullHandoffs).toContain('codex --model gpt-5.5 -c model_reasoning_effort="xhigh"')
- expect(fullHandoffs).toContain(
- 'Wait only for `tui-idle` when needed to avoid losing the prompt.'
- )
- expect(fullHandoffs).toContain('Do not monitor task completion.')
- })
-
- it('clarifies sidebar lineage for same-worktree orchestrated workers', () => {
- const skill = readSkill()
- const workerTerminals = getSection(skill, 'Worker Terminals')
-
- expect(workerTerminals).toContain(
- 'Sidebar lineage and orchestration lifecycle are related but not identical.'
- )
- expect(workerTerminals).toContain(
- 'A same-worktree worker may appear as a peer under that worktree in the sidebar'
- )
- expect(workerTerminals).toContain('while remaining a child dispatch in orchestration state')
- expect(workerTerminals).toContain(
- 'only an actual child worktree creates visible parent/child worktree lineage'
- )
- expect(workerTerminals).toContain(
- 'Create a new worktree only when the user explicitly requests one or a concrete checkout or filesystem conflict makes sharing unsafe or impossible'
- )
- expect(workerTerminals).toContain(
- 'Independent tasks, parallel execution, convenience, or a preference for separate checkouts are not isolation requirements.'
- )
- expect(workerTerminals).toContain(
- 'When a new worktree is allowed, use child lineage for isolated work that is stacked under or dependent on the active worktree'
- )
- expect(workerTerminals).toContain('use `--no-parent` when it is not stacked')
- })
-
- it('keeps review-only completions and named next-owner fixes in their lanes', () => {
- const skill = readSkill()
-
- expect(skill).toContain(
- 'A review-only `worker_done` reports findings; it does not authorize coordinator file edits.'
- )
- expect(skill).toContain('unless the user explicitly asked the coordinator to own fixes')
- expect(skill).toContain('dispatch or hand off fixes')
- expect(skill).toContain(
- "If the user's plan names a next owner agent " +
- '(for example, "then use opencode to create a PR")'
- )
- expect(skill).toContain('post-review corrections and PR prep belong to that named owner')
- expect(skill).toContain('the named owner edits files and creates the PR')
- })
-
- it('keeps post-completion workers idle without subordinating the user', () => {
- const skill = readSkill()
- const agentGuidance = getSection(skill, 'Agent Guidance')
-
- expect(agentGuidance).toContain('After sending `worker_done`, end that dispatched turn')
- expect(agentGuidance).toContain('idle at the agent prompt')
- expect(agentGuidance).toContain('Do not autonomously start more work, poll')
- expect(agentGuidance).toContain('A direct user instruction takes precedence')
- expect(agentGuidance).toContain('follow it without coordinator approval or a fresh Dispatch')
- expect(agentGuidance).toContain('never refuse it because of worker/coordinator roles')
- expect(agentGuidance).toContain("do not reuse the settled Dispatch's lifecycle IDs")
- expect(agentGuidance).toContain(
- 'A coordinator-supervised follow-up still arrives with a fresh preamble + TASK block'
- )
- expect(skill).not.toContain('post-completion polling messages')
- expect(skill).not.toContain('every 2 minutes')
- })
-
- it('makes settled worker terminal release an explicit coordinator step', () => {
- const skill = readSkill()
- const workerLoop = getSection(skill, 'Preferred Supervised Worker Loop')
- const agentGuidance = getSection(skill, 'Agent Guidance')
- const nextAction = getSection(skill, 'Next Action')
-
- expect(workerLoop).toContain(
- '# Process every message. For each accepted worker_done that is not immediately reused:\n' +
- 'orca orchestration worker-release --dispatch --json'
- )
- expect(workerLoop).toContain(
- 'Acknowledge only after every message and required release decision is handled'
- )
- expect(workerLoop).toContain(
- 'read the `worker.agent_terminal_handle` field of `worker-show --dispatch --json`'
- )
- expect(workerLoop).toContain(
- 'orca orchestration worker-start --task --terminal --json` so Orca ' +
- 'transfers cleanup ownership to the new Dispatch'
- )
- expect(workerLoop).toContain(
- 'Run `worker-release` after both succeeded and failed `worker_done` reports unless the user ' +
- 'explicitly asked to keep that worker live.'
- )
- expect(workerLoop).toContain('Release is post-completion cleanup, not cancellation')
- expect(workerLoop).toContain('orca orchestration worker-retain --dispatch --json')
- expect(workerLoop).toContain(
- 'the same Dispatch can be passed to `worker-release`, which clears the requested retention'
- )
- expect(agentGuidance).toContain(
- 'Coordinators must account for every settled worker terminal before waiting again or ending ' +
- 'the turn'
- )
- expect(agentGuidance).toContain('released workers remain readable through `worker-read`')
- expect(nextAction).toContain(
- 'After every accepted `worker_done`, either transfer the exact terminal to an immediate ' +
- 'follow-up Dispatch or run `worker-release` before the next wait.'
+ expect(reference).toContain('task-list --ready --brief --json')
+ expect(reference).toContain('`--effort` requires `--model`')
+ expect(reference).toContain('neither option combines with `--terminal`')
+ expect(reference).toContain('`launch.requested` with `launch.effective`')
+ expect(reference).toContain('worker-start --task --terminal')
+ expect(reference).toContain('A review-only `worker_done` authorizes synthesis')
+ expect(squash(reference)).toContain(
+ 'post-review fixes and PR preparation remain with that owner'
)
})
- it('documents per-invocation model and effort for supervised workers', () => {
- const workerLoop = getSection(readSkill(), 'Preferred Supervised Worker Loop')
+ it('owns worker heartbeat, ask resume, escalation, failure, and idle', () => {
+ const reference = readReference('worker-contract.md')
- expect(workerLoop).toContain('opaque provider model id with `--model`')
- expect(workerLoop).toContain('`--effort` requires `--model`')
- expect(workerLoop).toContain('neither option can combine with `--terminal`')
- expect(workerLoop).toContain('--agent claude --model opus --effort high --json')
- expect(workerLoop).toContain('`launch.requested` and `launch.effective`')
+ expect(reference).toContain('--type heartbeat')
+ expect(reference).toContain('--task-id --dispatch-id ')
+ expect(reference).toContain('--phase ""')
+ expect(reference).toContain('--resume ')
+ expect(reference).toContain('do not create a duplicate question')
+ expect(reference).toContain('--type escalation')
+ expect(reference).toContain('Send exactly one terminal report')
+ expect(reference).toContain('Use `--outcome failed`')
+ expect(reference).toContain('After `worker_done`, end the dispatched turn and idle')
+ expect(squash(reference)).toContain(
+ 'ORCA orchestration check --terminal --json'
+ )
+ expect(squash(reference)).toContain('once more immediately before `worker_done`')
+ expect(squash(reference)).toContain(
+ '`check` names its caller with `--terminal`, never `--from`'
+ )
+ expect(squash(reference)).toContain('If `check` returns `consumer_fenced`')
+ expect(squash(reference)).toContain('An empty `check` never means you were replaced')
})
- it('never authorizes release from idle, timeout, or worker-side triggers', () => {
- const skill = readSkill()
- const workerLoop = getSection(skill, 'Preferred Supervised Worker Loop')
- const agentGuidance = getSection(skill, 'Agent Guidance')
+ it('keeps heartbeat and worker_done recipes bound to the injected capability', () => {
+ const reference = readReference('worker-contract.md')
+ const recipes = [...reference.matchAll(/```text\n([\s\S]*?)```/gu)].map((match) => match[1])
+ const heartbeat = recipes.find((recipe) => recipe.includes('--type heartbeat'))
+ const workerDone = recipes.find((recipe) => recipe.includes('--type worker_done'))
- // The prohibition sentence is the guard the negative patterns below rely on.
- expect(workerLoop).toContain(
- 'Do not release a worker because of a timeout, TUI idle state, heartbeat, status, question, ' +
- 'escalation, or rejected/stale `worker_done`.'
- )
- expect(workerLoop).toContain(
- 'do not substitute `terminal close`; follow the exact recovery action in the receipt'
- )
- expect(skill).not.toMatch(
- /release[^.]*\bon (?:a |the )?(?:tui-?idle|idle|timeout|heartbeat|question|escalation)\b/iu
- )
- expect(skill).not.toMatch(
- /\b(?:after|on|upon) (?:a |the )?(?:tui-?idle|idle state|timeout|heartbeat)\b[^.]*\brelease/iu
- )
- expect(agentGuidance).toContain(
- 'Do not autonomously start more work, poll, or attempt to close the terminal yourself'
- )
- expect(agentGuidance).not.toMatch(/worker-release[^.]*\byourself\b/iu)
+ for (const recipe of [heartbeat, workerDone]) {
+ expect(recipe).toContain('--from ')
+ expect(recipe).toContain('--dispatch-capability ')
+ expect(recipe).toContain('--task-id --dispatch-id ')
+ }
+ expect(workerDone).not.toContain('--files-modified')
+ expect(workerDone).not.toContain('--report-path')
+ expect(squash(reference)).toContain('only when applicable, using actual paths')
+ expect(reference).toContain('Do not send documentation placeholders as metadata')
})
- it('documents @grok in the Messaging group address list', () => {
- const skill = readSkill()
- const messaging = getSection(skill, 'Messaging')
+ it('owns local, folder, worktree, SSH, WSL, remote, and mixed-version placement', () => {
+ const reference = readReference('placement-and-remote.md')
- expect(messaging).toContain('`@grok`')
+ expect(reference).toContain('--worktree current --agent codex')
+ expect(squash(reference)).toContain(
+ 'A worktree selector needs the full `::` value Orca returned, passed as `id:`; a bare repo id is not a worktree id'
+ )
+ expect(reference).toContain('--worktree new-child')
+ expect(reference).toContain('--worktree new-top-level')
+ expect(reference).toContain('Folder workspaces are first-class')
+ expect(reference).toContain('Remote `current` and `new-child` are invalid')
+ expect(squash(reference)).toContain("`--on` selects only the worker's execution server")
+ expect(squash(reference)).toContain(
+ 'route every follow-up, read, stop, and cleanup by Dispatch ID'
+ )
+ expect(reference).toContain('`live`, `unverifiable`, or `exited`')
+ expect(squash(reference)).toContain('unknown stream opcodes can be silently dropped')
+ expect(reference).toContain('printed `orca-ide`')
+ expect(squash(reference)).toContain(
+ 'ORCA project setup-existing-folder --project --host --path --kind folder --json'
+ )
+ expect(squash(reference)).toContain('and rejects a plain directory')
+ expect(reference).toContain(
+ 'ORCA orchestration worker-list --run --include-remote --json'
+ )
+ expect(squash(reference)).toContain(
+ 'enumerate remote workers with `--include-remote` or every one of them reads `unverifiable`'
+ )
})
- it('documents @cursor in the Messaging group address list', () => {
- const skill = readSkill()
- const messaging = getSection(skill, 'Messaging')
+ it('owns FIFO mail, Dispatch addresses, groups, questions, and gates', () => {
+ const reference = readReference('messaging-and-gates.md')
- expect(messaging).toContain('`@cursor`')
+ expect(reference).toContain('oldest FIFO Delivery')
+ expect(squash(reference)).toContain('Process every row')
+ expect(squash(reference)).toContain(
+ 'A Delivery therefore always carries the whole FIFO batch whatever its types, and a `check` without `--wait` hands that batch over unfiltered'
+ )
+ expect(reference).toContain('send --to dispatch:')
+ for (const group of ['@all', '@grok', '@cursor', '@worktree:']) {
+ expect(reference).toContain(group)
+ }
+ expect(reference).toContain('Dispatch lifecycle messages never target groups')
+ expect(reference).toContain('gate-create --task ')
+ expect(reference).toContain("Do not create a gate merely to answer a worker's `ask`")
+ expect(reference).toContain('successful `send` proves durable enqueue')
+ expect(squash(reference)).toContain('Wake and nudge are best-effort attention only')
+ expect(squash(reference)).toContain(
+ '`check` names its caller with `--terminal ` and is the only verb that rejects `--from`'
+ )
})
- it('keeps agent-first launch, handle recovery, and inbox injection distinct', () => {
- const skill = readSkill()
- const messaging = getSection(skill, 'Messaging')
- const workerTerminals = getSection(skill, 'Worker Terminals')
- const agentFirstExample = workerTerminals.match(
- /```bash\norca worktree create --name --agent codex --setup run --json\n[\s\S]*?```/
- )?.[0]
+ it('owns positive-evidence retry, unknown outcomes, retain/release, and no terminal close', () => {
+ const reference = readReference('recovery-and-cleanup.md')
- expect(workerTerminals).toContain('For an allowed new worktree, use agent-first:')
- expect(workerTerminals).toContain('fallback shell + agent pair')
- expect(workerTerminals).toContain(
- 'repo setup and default-terminal settings may add intentional tabs or splits'
+ expect(squash(reference)).toContain('| `ready` or active | Keep waiting')
+ expect(squash(reference)).toContain('| `outcome_unknown` | Inspect')
+ expect(squash(reference)).toContain('| Remote contact lost | Preserve `unverifiable`')
+ expect(reference).toContain('--retry-of ')
+ expect(squash(reference)).toContain('Placement is never silently inherited')
+ expect(reference).toContain('worker-abandon --dispatch')
+ expect(reference).toContain('worker-retain --dispatch')
+ expect(reference).toContain('worker-release --dispatch')
+ expect(squash(reference)).toContain('`release_pending` or `release_unknown`')
+ expect(squash(reference)).toContain('Never substitute `terminal close`')
+ })
+
+ it('owns the lost-response question and the request-show verdicts', () => {
+ const reference = squash(readReference('recovery-and-cleanup.md'))
+
+ expect(reference).toContain('request-show --request --json')
+ expect(reference).toContain('--retry-request ')
+ expect(reference).toContain('`completed` means the mutation already took effect')
+ expect(reference).toContain('`pending` means the original mutation is still running')
+ expect(reference).toContain('that is not proof nothing happened')
+ expect(reference).toContain('terminal send --wait-submit ')
+ })
+
+ it('names worker-list as the enumerating command and the agent-liveness authority', () => {
+ const reference = squash(readReference('recovery-and-cleanup.md'))
+
+ expect(reference).toContain('ORCA orchestration worker-list --run --json')
+ expect(reference).toContain("`worker-show`'s `observation.status` is PTY liveness only")
+ expect(reference).toContain(
+ '`projection.attention.categories`, `projection.attention.requiresAction`'
)
- expect(workerTerminals).toContain('without configured default tabs')
- expect(workerTerminals).toContain(
- 'only after `terminal list` or `terminal show` confirms it is an unused shell'
+ expect(reference).toContain('`projection.nextAction` argv')
+ expect(reference).toContain('the fleet verdict decides')
+ expect(reference).toContain(
+ 'ORCA orchestration worker-list --run --include-remote --json'
+ )
+ expect(reference).toContain('reads `unverifiable` until you enumerate with `--include-remote`')
+ expect(reference).toContain('follow `page.nextCursor` with `--cursor `')
+ })
+
+ it('requires positive evidence of exit before stop, abandon, retry, or release', () => {
+ const reference = squash(readReference('recovery-and-cleanup.md'))
+
+ expect(reference).toContain('Leave the wait only on positive proof the agent stopped')
+ expect(reference).toContain('`unverifiable` is always absence')
+ expect(reference).toContain('Absence never authorizes stop, abandon, retry, or release')
+ expect(reference).toContain(
+ '| `unverifiable` liveness | Keep waiting or inspect; never stop, abandon, retry, or release |'
+ )
+ })
+
+ it('owns the custom topology exception without claiming process ownership', () => {
+ const reference = readReference('low-level-topology.md')
+
+ expect(reference).toContain('only when `worker-start` cannot express')
+ expect(reference).toContain('terminal create --worktree active')
+ expect(reference).toContain('dispatch --task --to --inject')
+ expect(reference).toContain('operator-created process unsupervised')
+ expect(squash(reference)).toContain('creates no supervised worker resource row')
+ expect(reference).toContain('Use `worker-start --terminal `')
+ expect(squash(reference)).toContain('never use it for an ownership handoff')
+ })
+
+ it('owns legacy labels, read-only degradation, exact recovery, and takeover', () => {
+ const reference = readReference('legacy-contract-migration.md')
+
+ expect(reference).toContain('[LEGACY COMPATIBILITY]')
+ expect(reference).toContain('[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]')
+ expect(reference).toContain('[LEGACY READ-ONLY]')
+ expect(squash(reference)).toContain(
+ 'degrade to read-only inspection and never fall back to local execution'
+ )
+ expect(squash(reference)).toContain(
+ 'must not spawn, write, signal, stop, switch, focus, split, or inject'
+ )
+ expect(reference).toContain('launcher status `75`')
+ expect(reference).toContain('run_legacy_local')
+ expect(reference).toContain('Recovered orchestration work from a contract update')
+ expect(reference).toContain('run-use --id --takeover-legacy')
+ expect(reference).toContain(
+ 'Never take over while the original coordinator is actively coordinating'
)
- expect(workerTerminals).not.toContain('bare create opens a default shell')
- expect(workerTerminals).not.toContain('ends with **one** agent tab')
- expect(agentFirstExample).toBeDefined()
- expect(agentFirstExample).not.toContain('orca terminal list')
- expect(agentFirstExample).toContain('agentTerminalHandle')
- expect(agentFirstExample).toContain('startupTerminal.handle')
- expect(messaging).toContain('Prefer `agentTerminalHandle` from the create response')
- expect(messaging).toContain('Continue with the replacement handle only')
- expect(messaging).toContain('never writes to terminal input or remotely wakes another terminal')
- expect(messaging).toContain('Use `orchestration dispatch --inject` to deliver a tracked task')
})
})
describe('orchestration install stub', () => {
- it('points at the version-matched guide and preserves the safe resolver', () => {
+ it('preserves the safe version-matched resolver and bounded old-binary fallback', () => {
const stub = readFileSync(stubPath, 'utf8')
expect(stub).toContain('discovery stub')
expect(stub).toContain('ORCA skills get orchestration')
- // The safe CLI-resolution contract must survive in the stub, never a bare `orca`.
expect(stub).toContain('ORCA_CLI_COMMAND')
expect(stub).toContain('orca-dev')
expect(stub).toContain('orca-ide')
expect(stub).toContain('GNOME Orca screen reader')
+ expect(squash(stub)).toContain('explicitly reports that `skills get` is an unknown command')
+ expect(stub).toContain('do not invent commands')
expect(stub).not.toMatch(/^orca /mu)
})
- it('does not tell agents to mutate orchestration state before loading the guide', () => {
- const preGuide = readFileSync(stubPath, 'utf8').split('## Load the full guide')[0]
-
- expect(preGuide).not.toContain('orca orchestration task-create')
- expect(preGuide).not.toContain('orca orchestration dispatch')
- })
-
- it('gives older binaries a bounded fallback instead of a dead end', () => {
- const stub = readFileSync(stubPath, 'utf8').replace(/\s+/gu, ' ')
-
- expect(stub).toContain('explicitly reports that `skills get` is an unknown command')
- expect(stub).toContain('do not invent commands')
- expect(stub).toContain('ask the user rather than guessing')
- })
-
- it('drops the changing command reference from the installable file', () => {
+ it('performs no orchestration mutation before loading the guide', () => {
const stub = readFileSync(stubPath, 'utf8')
+ const preGuide = stub.split('## Load the full guide')[0]
- // Version-sensitive command detail lives in the binary-served guide now, not here.
- expect(stub).not.toContain('check --wait')
- expect(stub).not.toContain('dispatch-show')
- expect(stub.length).toBeLessThan(readFileSync(guidePath, 'utf8').length)
- })
-
- it('keeps the routing frontmatter identical to the guide', () => {
- const frontmatter = (text) => /^---\n[\s\S]*?\n---\n/u.exec(text)[0]
-
- expect(frontmatter(readFileSync(stubPath, 'utf8'))).toBe(
- frontmatter(readFileSync(guidePath, 'utf8'))
- )
+ expect(preGuide).not.toContain('orchestration task-create')
+ expect(preGuide).not.toContain('orchestration dispatch')
+ expect(frontmatter(stub)).toBe(frontmatter(readKernel()))
+ expect(stub.length).toBeLessThan(readKernel().length)
})
})
diff --git a/docs/site/content/docs/cli/orchestration.mdx b/docs/site/content/docs/cli/orchestration.mdx
index df093fab901..a8db0b06782 100644
--- a/docs/site/content/docs/cli/orchestration.mdx
+++ b/docs/site/content/docs/cli/orchestration.mdx
@@ -145,7 +145,7 @@ orca orchestration ask \
--json
```
-With `--json`, `ask` prints a single JSON object so workers can pipe it to `jq -r .answer`.
+With `--json`, `ask` prints the standard `{id, ok, result, _meta}` envelope, so workers read the answer with `jq -r .result.answer`.
## Decision gates
diff --git a/docs/site/content/docs/cli/reference.mdx b/docs/site/content/docs/cli/reference.mdx
index 5cbf19b82f9..5c0afa52b98 100644
--- a/docs/site/content/docs/cli/reference.mdx
+++ b/docs/site/content/docs/cli/reference.mdx
@@ -290,6 +290,8 @@ List bundled guides, print a version-matched guide, or install/update hybrid ski
```bash
orca skills list
orca skills get orca-cli
+orca skills get orchestration --references
+orca skills get orchestration --reference recovery-and-cleanup
orca skills get orchestration --full
orca skills install --skill orca-cli --skill orchestration
orca skills install --all --dry-run
diff --git a/docs/site/content/docs/cli/skills.mdx b/docs/site/content/docs/cli/skills.mdx
index 639b5099119..77ea47ce8ad 100644
--- a/docs/site/content/docs/cli/skills.mdx
+++ b/docs/site/content/docs/cli/skills.mdx
@@ -39,10 +39,14 @@ After `npx skills add`, agents see a short stub that says:
```bash
orca skills list
orca skills get orca-cli
+orca skills get orchestration --references
+orca skills get orchestration --reference recovery-and-cleanup
orca skills get orchestration --full
orca skills get orca-linear --json
```
+A guide's action gates name conditional references. `--reference ` prints one of them alone, so an agent pays for the kernel plus that document instead of the whole package; `--references` lists the names. The name may be bare (`recovery-and-cleanup`) or spelled as the guide writes it (`references/recovery-and-cleanup.md`). `--full` still prints the kernel plus every reference.
+
Add `--json` when an agent needs deterministic output for automation. `skills show` is an alias for `skills get`.
## Keep skills up to date
diff --git a/resources/skills/current-manifest.json b/resources/skills/current-manifest.json
index fdb54016a8f..925b09f75fe 100644
--- a/resources/skills/current-manifest.json
+++ b/resources/skills/current-manifest.json
@@ -131,17 +131,17 @@
"name": "orchestration",
"sourcePath": "skills/orchestration",
"releaseRevision": 29,
- "packageDigest": "689e31d84256aded123c801eaa87413474943a9a30d96bff9a19d0a321aefb54",
- "gitTreeSha": "902cc33dd65730b32ac234dd0ae7166d75498b46",
+ "packageDigest": "894d6f421cb96c2777e73055df867e2fdfca8dd05f0340d50a93cb33a8e85e3a",
+ "gitTreeSha": "da5b5c3f78634bbe12922e526ea227509faa9de0",
"files": [
{
"path": "SKILL.md",
- "size": 4398,
+ "size": 4539,
"executable": false,
"classification": "text",
- "exactSha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18",
- "textNormalizedSha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18",
- "identitySha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18"
+ "exactSha256": "937237cbb3449ff88f67efbcec0b6c6d64a23dbfb1b28c88260e4d0094f50954",
+ "textNormalizedSha256": "937237cbb3449ff88f67efbcec0b6c6d64a23dbfb1b28c88260e4d0094f50954",
+ "identitySha256": "937237cbb3449ff88f67efbcec0b6c6d64a23dbfb1b28c88260e4d0094f50954"
}
]
}
diff --git a/resources/skills/snapshot-registry.json b/resources/skills/snapshot-registry.json
index 5b3412a497b..520c9250fb2 100644
--- a/resources/skills/snapshot-registry.json
+++ b/resources/skills/snapshot-registry.json
@@ -1046,17 +1046,17 @@
},
{
"releaseRevision": 29,
- "packageDigest": "689e31d84256aded123c801eaa87413474943a9a30d96bff9a19d0a321aefb54",
- "gitTreeSha": "902cc33dd65730b32ac234dd0ae7166d75498b46",
+ "packageDigest": "894d6f421cb96c2777e73055df867e2fdfca8dd05f0340d50a93cb33a8e85e3a",
+ "gitTreeSha": "da5b5c3f78634bbe12922e526ea227509faa9de0",
"files": [
{
"path": "SKILL.md",
- "size": 4398,
+ "size": 4539,
"executable": false,
"classification": "text",
- "exactSha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18",
- "textNormalizedSha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18",
- "identitySha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18"
+ "exactSha256": "937237cbb3449ff88f67efbcec0b6c6d64a23dbfb1b28c88260e4d0094f50954",
+ "textNormalizedSha256": "937237cbb3449ff88f67efbcec0b6c6d64a23dbfb1b28c88260e4d0094f50954",
+ "identitySha256": "937237cbb3449ff88f67efbcec0b6c6d64a23dbfb1b28c88260e4d0094f50954"
}
]
}
diff --git a/skill-guides/orca-cli.md b/skill-guides/orca-cli.md
index 1dc918cbdf8..8cdeb18ec49 100644
--- a/skill-guides/orca-cli.md
+++ b/skill-guides/orca-cli.md
@@ -181,6 +181,7 @@ ORCA terminal read --terminal --json
ORCA terminal read --terminal --cursor --limit 1000 --json
ORCA terminal read --json
ORCA terminal send --terminal --text "continue" --enter --json
+ORCA terminal send --terminal --text "continue" --enter --wait-submit 10 --json
ORCA terminal send --text "echo hello" --enter --json
ORCA terminal wait --terminal --for exit --timeout-ms 5000 --json
ORCA terminal wait --terminal --for tui-idle --timeout-ms 300000 --json
@@ -204,7 +205,11 @@ Terminal rules:
- `terminal list --json` omits `visualLayouts` to keep the common agent payload bounded. Add `--include-visual-layouts` only when tab and pane topology is required.
- Use `terminal read` before `terminal send` unless the next input is obvious.
- Use `terminal send` only for direct terminal input or one-off prompts where no task state, inbox, or reply tracking is needed.
-- For structured coordination, invoke the `orchestration` skill; it uses `orca orchestration ...` commands for messages, handoffs, task DAGs, dispatches, inbox/reply flows, and coordinator loops. A receiving agent can run `orca orchestration check --unread --format` to render its unread mail in agent-readable form; this checks the caller's inbox and does not remotely deliver input to another terminal.
+- A text-plus-Enter agent prompt returns a durable request ID and additive stages: `input_accepted`, then `turn_started` once the agent's turn is proven. Raw text-only, bare Enter, interrupt, and terminal query replies keep their existing direct-input behavior.
+- A default send observes for 0 seconds, so a receipt that stops at `input_accepted` is expected and its warning means "unproven", not "failed". Pass `--wait-submit` when you need proof of submission.
+- `--wait-submit ` only observes the same accepted prompt. A timeout returns queued/input-accepted truth without resending; after an ambiguous transport failure, repeat the exact command with the reported `--retry-request `. Both text and `--json` receipts carry the same `warnings`.
+- An older host reports a legacy `old-host` fallback for an ordinary send and refuses `--wait-submit` or `--retry-request` before input, because it cannot provide durable replay.
+- For structured coordination, invoke the `orchestration` skill; it uses `orca orchestration ...` commands for messages, handoffs, task DAGs, dispatches, inbox/reply flows, and coordinator loops. A receiving agent can run `orca orchestration check --peek --format --json` to render its unread mail in agent-readable form; this checks the caller's inbox and does not remotely deliver input to another terminal.
- Use `terminal create --worktree active --command ""` for a fresh agent in the current worktree. Use `worktree create --agent ` only for a separate checkout (agent in the first terminal — do not also `terminal create` the same agent).
- Use `terminal wait --for tui-idle` for agent CLIs such as Claude Code, Gemini, Codex, OMP, Pi, and Grok; always pass `--timeout-ms`.
- Terminal handles are runtime-scoped. Use `startupTerminal.handle` as the sole agent handle when `worktree create --agent` returns it; if Orca restarts, omits the handle, or returns `terminal_handle_stale`, reacquire with `terminal list` and continue with the replacement only.
diff --git a/skill-guides/orchestration.md b/skill-guides/orchestration.md
index eab866f13d0..b06e2cc9143 100644
--- a/skill-guides/orchestration.md
+++ b/skill-guides/orchestration.md
@@ -1,449 +1,201 @@
---
name: orchestration
description: >-
- Use Orca orchestration for structured multi-agent coordination: threaded
- messages, blocking ask/reply flows, task dispatch, worker_done/escalation
- waits, task DAGs, decision gates, or coordinator loops. Use `orca-cli`
- instead for full ownership handoffs, including requests phrased as "hand
- off", "handoff", "handover", "give this to another agent", or "another
- worktree" when the user did not explicitly ask to supervise, monitor, wait
- for results, or coordinate a DAG. Use `orca-cli` for terminal control,
- lightweight terminal prompts, shell commands, Orca worktree management,
- reading or waiting on terminals, and the Orca embedded browser. Use Computer
- Use for external browser windows, webviews, Orca app UI, or desktop UI
- outside Orca's embedded browser only when the task requires OS/window-level
- control such as focus, menus, dialogs, coordinates, or screenshots. Use
- `orca-cli` for Orca's embedded pages and a page-automation tool such as
- Playwright or CDP for external pages.
+ Coordinate supervised Orca workers: threaded messages, blocking ask/reply,
+ task dispatch, worker_done/escalation waits, task DAGs, decision gates,
+ coordinator loops, and decomposing work across agents. Use `orca-cli` for full
+ ownership handoffs — "hand off", "handoff", "handover", "give this to another
+ agent", "another worktree" — unless asked to supervise, monitor, or coordinate
+ a DAG, and for terminal control, lightweight terminal prompts, shell commands,
+ Orca worktree management, and reading or waiting on terminals. Use Computer
+ Use for external browser windows, webviews, Orca app UI, or desktop UI outside
+ Orca's embedded browser only when the task requires OS/window-level control
+ such as focus, menus, dialogs, coordinates, or screenshots. Use `orca-cli` for
+ Orca's embedded pages and a page-automation tool such as Playwright or CDP for
+ external pages.
---
-# Orca Inter-Agent Orchestration
+# Orca orchestration
-Orchestration is Orca's structured coordination layer for agent messages, task ownership, dispatch state, and worker completion tracking.
+Orchestration is Orca's structured coordination layer. It records who owns work,
+which attempt is authoritative, and when supervised work has settled.
-Use this skill when coordination state matters. For lightweight terminal prompts or basic worktree/terminal/built-in-browser control, use `orca-cli`.
+## Outcome
-## Tool Boundary
+**Result:** every in-scope Task has one explicit outcome and every settled worker
+terminal has a next owner or cleanup decision. **Next consumer:** the user who
+requested supervision. **Done:** all expected Dispatches have settled, every
+delivered message was processed before acknowledgment, each settled worker was
+reused, explicitly retained, or released, and the turn ends only when the report
+to that user names, per Task, its outcome, the evidence behind it, and any
+unresolved blocker.
-If a task says to use Orca orchestration, the coordinator must create or bind a Run, create the Task with `orca orchestration task-create`, then attach the worker with either the preferred `orca orchestration worker-start` composition or the low-level `orca orchestration dispatch --inject` path.
+**Safe failure:** preserve work and authority and report the state as unknown or
+`unverifiable`. Only positive proof of exit authorizes stop, abandon, or retry,
+and only an accepted settlement authorizes release. Every other observation,
+absence included, is a checkpoint.
-Do not substitute non-Orca subagent tools, generic agent-spawn APIs, or chat-only parallel worker features. Those may create useful workers, but they do not create Orca task/dispatch provenance, injected lifecycle preambles, `worker_done` authority, or decision gates.
+## Classify the role
-Before claiming a worker was orchestrated, verify the task/dispatch exists:
+| Current context | Role | Route |
+| ---------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------- | ------------------------------------------------------------------------------ |
+| The user explicitly asks to supervise, monitor, wait for results, track completion, coordinate a DAG, use a decision gate, or manage ask/reply | Coordinator | Use the supervised loop below |
+| The current prompt contains a live injected preamble with Task and Dispatch IDs | Dispatched worker | Follow the preamble and the worker obligations below |
+| The user asks to hand off ownership or start another agent/worktree without supervision | Handoff owner | Use `orca-cli`; create no Run, Task, or Dispatch and do not monitor completion |
+| A message carries a legacy authority label | Compatibility operator | Load the legacy contract reference before any lifecycle mutation |
+| No live preamble and no explicit supervision | Ordinary terminal agent | Do not emit lifecycle messages; use `orca-cli` for terminal/worktree work |
-```bash
-orca orchestration task-list --json
-orca orchestration dispatch-show --task --json
+Model or effort selection does not make a handoff supervised. Never substitute a
+non-Orca subagent tool when Orca orchestration provenance was requested.
+
+## Authority and safety floor
+
+- A Run is a durable namespace and coordinator inbox; it does not schedule or
+ place workers. A Task is work. A Dispatch is one authoritative Task attempt.
+- Lifecycle authority comes from the active Dispatch, not a terminal title,
+ copied ID, old database row, provider transcript, or visible pane.
+- Workers use the exact executable, handle, capability, Task ID, and Dispatch ID
+ in the live preamble. Never reconstruct, translate, or broaden those arguments.
+- After remote start, address the worker by Dispatch ID. The execution host owns
+ process, filesystem, transcript, stop, and cleanup facts. Preserve the verdicts
+ `live` / `unverifiable` / `exited`; contact loss is not process death.
+- Liveness is layered: `worker-list`'s `projection.liveness` is the fleet verdict
+ for the agent; `worker-show`'s `observation.status` is PTY liveness only. A live
+ terminal can still hold a dead or stuck agent.
+- Folder workspaces are valid; never require Git or assume a worktree.
+- Clients and remote servers update independently. Treat unknown optional fields
+ as absent. A new stream operation requires advertised capability because old
+ decoders may silently drop unknown opcodes. Never fall back to local execution
+ when remote authority or capability is unproven.
+- Use the executable you used to run `skills get` for the entire run. In the
+ examples below, replace `ORCA` with it; do not create a shell variable or run
+ `ORCA` literally. If it fails, report that exact error instead of switching.
+- A successful `orchestration send` proves durable enqueue; its wake or nudge is
+ best-effort attention only and does not prove the recipient read or accepted it.
+
+## Worker obligations
+
+The injected preamble is authoritative. A dispatched worker must:
+
+1. Do only the current Task and use the preamble's `ask` command for a blocking
+ coordinator question. Never open a local question TUI the coordinator cannot
+ answer. Resume the same message ID after an ask timeout.
+2. Send heartbeats only at the cadence in the preamble. A heartbeat proves
+ liveness, not completion.
+3. Read coordinator follow-ups at each natural checkpoint — before starting a
+ new file, after a test run — and once more immediately before `worker_done`:
+ `ORCA orchestration check --terminal --json`.
+4. Send `worker_done` exactly once, from the dispatched terminal, with a
+ three-sentence executive summary, both lifecycle IDs, and explicit
+ `--outcome succeeded` or `--outcome failed`. Never encode failure only in prose.
+5. Append `--files-modified` and `--report-path` only with real values when
+ applicable. After `worker_done`, end the dispatched turn and idle; do not poll
+ or start new work.
+
+A direct user instruction after completion starts new user-owned work and takes
+precedence over the idle rule. Do not reuse the settled lifecycle IDs.
+
+## Canonical supervised loop
+
+Confirm the runtime, bind one Run, and start the full independent wave before
+waiting. `worker-start --spec` creates the Task and its attempt in one call:
+
+```text
+ORCA status --json
+ORCA orchestration run-create --objective "" --json
+ORCA orchestration worker-start --spec "" --worktree current --agent codex --json
+ORCA orchestration worker-start --spec "" --worktree current --agent claude --model sonnet --json
+ORCA orchestration check --wait --types "worker_done,escalation,question" --timeout-ms 900000 --json
```
-If the work was accidentally run outside Orca orchestration, say so plainly. To repair provenance, rerun or revalidate the needed work through a fresh Orca terminal plus injected dispatch; do not retroactively describe the external worker as orchestrated.
+If `worker-start` exits non-zero, do not relaunch. Read the receipt's
+`failedStage` and `residualResources`, then load
+`references/recovery-and-cleanup.md`.
-## When To Use
+Use `task-create` plus `worker-start --task ` for planned fan-out with
+dependencies or a retry of a known Task. Use dependencies only for real ordering
+and prefer parallel waves over chains deeper than three or four steps; nested
+workers obey the depth limit, and a new Run does not reset the caller's depth.
-- Send/reply/ask between agent terminals with persistent messages.
-- Dispatch structured tasks to workers and wait for `worker_done` or `escalation`.
-- Track task DAGs with dependencies.
-- Run coordinator loops or decision gates.
+A consuming `check` names its caller with `--terminal `, never `--from`;
+omit it inside the coordinator's own Orca terminal. It returns the bound Run's
+oldest FIFO Delivery and replays that batch until acknowledged. Process every
+message: reply to questions, validate each `worker_done` against the expected
+active Dispatch, and decide each settled terminal's next owner before the ack:
-Do not use orchestration merely because the user says "hand off", "handoff", "handover", "give this to another agent", or asks for another worktree/agent/model/effort. Those are full ownership transfers unless the user explicitly asks to supervise, monitor, wait for worker completion/results, coordinate a DAG, use decision gates, or keep a blocking ask/reply loop.
-
-## Preconditions
-
-- `orca status --json` should show a running runtime.
-- `orca` must be on PATH (`orca-ide` on Linux).
-- The orchestration experimental feature must be enabled in Settings > Experimental.
-- `orca orchestration` commands are RPC calls to the running Orca runtime.
-
-## Contract Migration
-
-Orca adopts a live pre-update orchestration assignment into an ordinary Run. Adoption preserves the existing agent process, PTY/session, terminal handle, tab/leaf/pane, worktree or folder workspace, Task, and Dispatch; it never restarts or replaces the worker. The retired scheduler is not revived, and a newly created attempt uses the current grammar.
-
-Treat the authority label on injected or formatted messages as definitive:
-
-- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported command printed with the message, using the same CLI executable and arguments that the original prompt supplied.
-- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded, at-least-once cutover replay. Process it idempotently and acknowledge it only through the exact displayed guidance.
-- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or lifecycle action.
-- An unlabeled current message uses the current guide and current grammar.
-
-An explicitly selected current Run, attested current Run binding, current Dispatch, or federated attachment takes precedence over legacy fallback. A retained adoption record alone never turns a current command into a legacy call.
-
-Database provenance, an old-looking terminal, or a legacy Run ID does not prove mutation authority. If the runtime cannot prove liveness, principal ownership, capability, or the exact legacy contract, it degrades to read-only inspection and must not fall back to local execution. Exact recovery may restore the already-live PTY once in its original inactive background tab. It must not spawn, write, signal, stop, switch, focus, split, or inject a terminal. Loss of lifecycle authority does not invalidate the existing assignment, process, or filesystem work.
-
-Compatibility retries have narrow guarantees. A pending ask, a reply, a final Dispatch settlement, and a consuming check have durable recovery identities. A-era heartbeat and escalation calls remain at-least-once across a manual A-to-B retry because identical later signals may be intentional. If an A-era ask may already have been answered, run the exact non-consuming recovery check printed by the runtime first; after its answer is printed and acknowledged, a new invocation with the same question creates a new question. Never guess among multiple identical question threads.
-
-When a compatibility or recovery command returns structured next-step arguments, run those exact arguments with the same CLI executable. The arguments intentionally omit the executable name so the guidance works with `orca`, `orca-ide`, `orca-dev`, or another configured Orca CLI command. Do not translate the command from memory, broaden its recipient, or retry it as a current mutation unless the returned guidance explicitly says to.
-
-On packaged Windows, a legacy ask uses a two-step commit/resume protocol. The initial command durably commits the question, prints its exact `ask --resume ` command, and exits with launcher status `75`; it does not wait for the answer. Run that exact resume command after the launcher or update boundary. Resume is idempotent and read-oriented: it waits for the already-committed question and does not create another one. For a WSL process that received compatibility proof at launch, use the printed executable `orca-ide` WSL resume command so the same distro and packaged launcher authority are preserved; do not substitute a PATH-resolved local CLI. Older WSL processes that never received the hidden launch token remain lifecycle read-only after the update, even while their terminal and filesystem work continue.
-
-Legacy inspection remains available without consuming mail:
-
-```bash
-orca orchestration run-list --json
-# run_legacy_local is an empty audit tombstone after adoption.
-orca orchestration run-show --id run_legacy_local --json
-# In run-list, find the ordinary Run whose objective is:
-# "Recovered orchestration work from a contract update"
-orca orchestration run-show --id --json
-orca orchestration task-list --run --json
-orca orchestration inbox --full --json
-orca orchestration check --terminal --peek --format --json
-orca terminal read --terminal --json
-orca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json
+```text
+ORCA orchestration reply --id --body "" --json
+ORCA orchestration worker-release --dispatch --json
+ORCA orchestration check --ack --wait --types "worker_done,escalation,question" --timeout-ms 900000 --json
```
-If the original coordinator is unavailable or cannot prove its retained authority, a current coordinator may explicitly take over the adopted Run from its own live agent terminal:
-
-```bash
-orca orchestration run-use --id --takeover-legacy --json
-orca orchestration check --run --json
-```
-
-Takeover fences only the old coordinator, binds the current one, and moves pending worker mail into current Run Delivery. It is bound to the authenticated invoking terminal; `--from` cannot name another coordinator. Live legacy workers keep their original Tasks, Dispatches, processes, filesystems, and old prompt commands; their later questions, escalations, and completion reports route to the current coordinator. Do not use takeover while the original coordinator is still actively coordinating, because its later lifecycle mutations are rejected.
-
-Do not launch a replacement editor merely because the desktop app or runtime was updated. If adoption cannot prove continuing authority, keep the original worker as the only editor until it reaches a stable handoff point, then use a new current Dispatch in a conflict-free placement for any remaining work.
-
-## Ownership
-
-New orchestration messages and tasks belong to one explicitly bound Run. A Run is only a durable namespace and coordinator inbox; it never schedules or places workers. Lifecycle authority comes from the active Dispatch, and terminal handles remain routing metadata rather than durable identity. Send `worker_done` and `heartbeat` from the worker's own terminal; Orca routes them to that Dispatch's Run.
-
-Classify inherited context before sending lifecycle messages:
-
-- Coordinated subtask: a live coordinator owns the DAG and waits on this dispatch. Follow the preamble exactly, including `worker_done`, heartbeat/status, `ask`, and `escalation`.
-- Full handoff means ownership transfer, not supervised dispatch. The original actor is not monitoring a DAG, so do not create lifecycle obligations unless the user explicitly asks you to supervise.
-- Classify requests containing "hand off", "handoff", "handover", "give this to another agent", "give this to another worktree", "another agent", or "another worktree" as full handoffs by default, even when the user names a custom model or reasoning effort.
-- Use supervised orchestration only when the user explicitly asks you to "supervise", "monitor", "wait", "track completion", "wait for worker_done", return results, coordinate a DAG, use a decision gate, or manage ask/reply flow.
-- Do not use `orca orchestration dispatch --inject` for full handoffs. It injects a coordinator preamble that tells the worker to send `worker_done`, heartbeat, and `ask` messages, then end its turn under the original terminal's dispatch lifecycle.
-- Do not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. Do not peek at terminal output after prompt delivery to monitor progress.
-- A review-only `worker_done` reports findings; it does not authorize coordinator file edits. After a review-only completion, synthesize findings, ask a decision gate if ownership is unclear, and dispatch or hand off fixes unless the user explicitly asked the coordinator to own fixes.
-- If the user's plan names a next owner agent (for example, "then use opencode to create a PR"), post-review corrections and PR prep belong to that named owner. The coordinator routes, synthesizes, asks decision gates when needed, and supervises; the named owner edits files and creates the PR.
-
-If unclear, inspect orchestration state before sending lifecycle messages:
-
-```bash
-orca orchestration task-list --json
-orca terminal list --json
-# If inherited context includes a task id:
-orca orchestration dispatch-show --task --json
-```
-
-## Messaging
-
-```bash
-orca orchestration send --subject [--to ] [--from ] [--body ] [--type ] [--priority ] [--thread-id ] [--payload ] [--json]
-orca orchestration check [--terminal ] [--ack ] [--peek|--all] [--types ] [--format] [--wait] [--timeout-ms ] [--json]
-orca orchestration reply --id --body [--from ] [--json]
-orca orchestration ask (--question |--resume ) [--options ] [--timeout-ms ] [--from ] [--json]
-orca orchestration inbox [--limit ] [--json]
-```
-
-Rules:
-
-- Omit `--from` unless impersonating another terminal; Orca auto-resolves it from the current terminal.
-- A coordinator `check` returns the bound Run's oldest FIFO Delivery (up to 50 messages) and replays that exact batch until `--ack `. Process every message before acknowledging; `check --ack --wait` acknowledges, checks, and waits in one operation.
-- Use `--peek` and `--all` only for read-only history/debugging. Type filters decide when a waiter wakes; the returned actionable Delivery is still the oldest full batch.
-- Use `dispatch:` for coordinator guidance to one supervised worker. Orca routes that stable address locally or through the connected-server relay; do not substitute a remote terminal handle.
-- Terminal handles remain appropriate for low-level pre-Dispatch messaging. Prefer `agentTerminalHandle` from the create response, fall back to `startupTerminal.handle` for older runtimes, then re-resolve with `orca terminal list --worktree ... --json` if missing or stale. Continue with the replacement handle only; never dual-send to old and new handles.
-- `terminal list --json` omits `visualLayouts` because handle recovery does not need topology. Add `--include-visual-layouts` only for explicit tab and pane inspection.
-- `orca orchestration check --peek --format --json` returns locally formatted unread mail without consuming it; it never writes to terminal input or remotely wakes another terminal. Use `orchestration dispatch --inject` to deliver a tracked task, or `terminal send` when an existing agent needs a free-form prompt.
-- While supervising workers manually, use `check --wait --types worker_done,escalation,question --timeout-ms ` instead of sleep/poll loops. Process the whole Delivery, reply to `question` messages with `orca orchestration reply --id --body --json`, then acknowledge and keep waiting.
-- `check --json` prints exactly one JSON document on stdout. While `--wait` blocks it also prints keepalive lines (`{"_keepalive":true,...}`) to stderr so you can tell the process is alive; those are never on stdout. Do not merge the streams before a parser — `check --wait --json 2>&1 | ` fails with "Extra data: line 2". Pipe stdout only.
-- Treat a `check --wait` timeout or `{count:0}` as a checkpoint, not a worker failure. Long coding tasks routinely run 15-60 minutes; keep using rolling waits unless you receive `worker_done`/`escalation`, the terminal exits or disappears, or the user explicitly asks you to stop.
-- Heartbeats and visible terminal activity mean the worker is alive, not done. Do not stop, close, kill, or restart a worker just because it has not produced a completion message yet.
-- Use `ask` when a worker needs a blocking answer from the coordinator; it defaults to the active Dispatch's Run. Timeout or disconnect leaves the question pending, so resume by its original message ID instead of asking again.
-- `check --wait` returns one bounded Delivery, not every future completion. Process every message, acknowledge it, then keep waiting until every expected Dispatch settles.
-- Group addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`, `@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`.
-- Message types include `status`, `dispatch`, `worker_done`, `merge_ready`, `escalation`, `handoff`, `question`, `decision_gate` (legacy/gates), and `heartbeat`.
-- Use group addresses only for messages that are genuinely useful to many terminals, such as `status` broadcasts or intentional fan-out questions. Do not send dispatch lifecycle messages to groups.
-- `worker_done` belongs to the active Dispatch and defaults to its Run mailbox; never target a group.
-- A valid `worker_done` for the active `taskId` + `dispatchId` marks the task and dispatch completed automatically. Do not follow it with `task-update --status completed`; reserve manual updates for explicit recovery or overrides.
-- `heartbeat` is also Dispatch-scoped. Include both IDs and omit `--to` so Orca uses the owning Run; use `status` for broad progress updates.
-
-## Tasks And Dispatch
-
-A Run is the namespace/inbox, a Task is the work item, and a Dispatch assigns one Task attempt to a terminal. Create or bind a Run once before the common loop.
-
-```bash
-orca orchestration run-create --objective --json
-orca orchestration task-create --spec [--deps ] [--parent ] [--json]
-orca orchestration task-list [--status ] [--ready] [--brief] [--json]
-orca orchestration task-update --id --status [--result ] [--json]
-orca orchestration dispatch --task --to [--from ] [--inject] [--json]
-orca orchestration dispatch-show --task [--json]
-```
-
-Task statuses: `pending`, `ready`, `dispatched`, `completed`, `failed`, `blocked`.
-
-Dispatch rules:
-
-- `--inject` sends the task spec plus preamble into a recognized agent CLI so it can report `worker_done`.
-- If the target is a bare shell, omit `--inject`, dispatch for tracking if needed, then send the prompt manually with `orca terminal send --terminal --text --enter --json`.
-- After 3 consecutive failures on one task, the dispatch context circuit-breaks and the task is marked failed.
-- Use `task-list --brief --json` for coordinator sweeps; it collapses whitespace and caps each echoed spec at 160 characters (`spec_truncated` marks shortened rows). Omit `--brief` when the full spec is required, or when an older CLI rejects it as an unknown flag.
-
-`dispatch` and `worker-start` refuse the following preflight cases with a stable `error.code`; read it before choosing a recovery, and treat `error.data.nextSteps` as the exact recovery text. Older hosts may omit `data`, so treat every field as optional.
-
-| Code | Meaning | Recovery |
-| -------------------- | --------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------ |
-| `task_not_found` | No Task with that id, or not in the bound Run (`data.taskId`, `data.runId`) | Check `task-list --json`; create the Task with `task-create` if it does not exist |
-| `task_not_startable` | Task cannot start now: not `ready`, or invalid `--retry-of` (`data.status`, `data.unmetDependencies`, `data.retryOf`) | Wait for running dependencies with `check --wait`; retry or unblock failed ones; inspect `dispatch-show` if already dispatched |
-| `inject_rejected` | `--inject` refused because no recognized agent runs in the target (`data.terminal`, `data.reason`) | Start a recognized agent there or pick another terminal; or dispatch without `--inject` and use `terminal send` |
-| `runtime_error` | Any other failure, including a target terminal that already owns an active Dispatch | Read the message, inspect state, and do not retry unchanged |
-
-## How deep workers can nest
-
-A dispatched worker normally cannot dispatch sub-workers. Attempting it fails with
-`nested_worker_depth_exceeded` and a message telling the worker to complete the task
-itself. Do that — do not try to route around it.
-
-The limit is a number, not an on/off switch. `Settings -> Orchestration -> Nested worker depth`
-sets how many generations are allowed:
-
-- `1` (default): a coordinator dispatches workers; those workers do not dispatch.
-- `2`: workers may dispatch one further generation.
-
-Depth is counted from the terminal that issues the command, not from the Run. Creating a
-new Run does not reset it — a worker that runs `run-create` then `worker-start` is still a
-worker, and still counted. This is the part that changed: the old behaviour rejected
-sub-dispatch only because a worker's terminal was not bound to a Run, so creating a Run was
-enough to slip past it.
-
-Two limits worth knowing:
-
-- **It is a guardrail, not a security boundary.** A caller that declares another terminal's
- handle while its own launch evidence is unverifiable (an ordinary restored terminal, for
- example) can be counted as that terminal instead. Orca does not treat workers as hostile.
-- **It applies while a Dispatch is active.** After `worker_done`, or after a coordinator
- settles the task, the terminal is no longer a worker and is counted as a root again. The
- process may still be alive; that is the documented boundary, not an accident.
-
-## Preferred Supervised Worker Loop
-
-Use `worker-start` for the normal supervised path. It composes the existing worktree, terminal, readiness, and dispatch primitives while returning exact created/reused effects. Agents still choose placement and concurrency; Orca does not schedule workers or infer conflicts.
-
-Create the Run and every independent Task first, then start all independent workers before waiting:
-
-```bash
-orca orchestration run-create --objective "" --json
-orca orchestration task-create --spec "" --json
-orca orchestration task-create --spec "" --json
-orca orchestration worker-start --task --worktree current --agent codex --json
-orca orchestration worker-start --task --worktree current --agent claude --json
-```
-
-`current` and exact existing worktrees create a fresh agent terminal and do not rerun setup. Reuse an existing agent only with `--terminal `.
-
-For a per-invocation Claude, Codex, or Cursor launch, pass an opaque provider model id with `--model`; add `--effort` only when that agent/model supports the level. These options apply only to fresh agent terminals, override general agent default arguments, and are reported under `launch.requested` and `launch.effective` in the receipt:
-
-```bash
-orca orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json
-```
-
-`--effort` requires `--model`, and neither option can combine with `--terminal`. A connected worker server must advertise launch-preference support before Orca forwards either option.
-
-For a new worktree, setup runs by default and agent-first creation reuses the returned startup agent terminal:
-
-```bash
-orca orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json
-# Independent/top-level:
-orca orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json
-```
-
-Setup normally starts alongside the agent. Only a repository explicitly configured with `wait-for-setup` delays agent launch until setup succeeds. Use `--setup skip` or `--setup inherit` only for a concrete reason.
-
-Read the returned receipt before continuing: `ready` plus setup `running` is normal for start-immediately, while wait-for-setup returns setup `succeeded` before accepting task input. A failed or unknown start exits nonzero; inspect its `stage`, `effects`, and `residualResources` instead of guessing or automatically retrying. A wait-for-setup timeout can honestly leave setup `running`, which is not proof of failure.
-
-To run the worker on another connected Orca server, add `--on `. The Run and Tasks remain authoritative on the current server; later commands route by Dispatch ID, so never repeat `--on`:
-
-```bash
-# Mac Run home -> Windows worker (the reverse is identical from a Windows Run home)
-orca orchestration worker-start --task --on windows --worktree new-top-level --repo --name --agent codex --setup run --json
-orca orchestration worker-show --dispatch --json
-orca orchestration worker-read --dispatch --limit 50 --json
-orca orchestration send --to dispatch: --subject "Follow-up" --body "" --json
-```
-
-Remote `current` and `new-child` are intentionally invalid because those words are ambiguous across servers. Use an exact discovered remote worktree selector or `new-top-level` with an explicit remote repo selector.
-
-The follow-up is structured inbox mail, not prompt injection. The worker's next
-`orchestration check` receives it even when the Dispatch is on another connected Orca server.
-
-`worker-read` defaults to `--source auto`: Orca returns the exact hook-reported Codex, Claude, OpenClaude, or Grok transcript when it can prove the worker session, otherwise it returns bounded terminal output with `source: "terminal"` and a typed `fallbackReason`. Continue with the returned top-level `cursor`; it stays pinned to that exact source. If Orca reports `source_changed`, start a fresh read without the old cursor. Never supply or guess a provider session ID or transcript path.
-
-Wait until every expected Dispatch settles, not for a fixed number of batches:
-
-```bash
-orca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json
-# Process every message. For each accepted worker_done that is not immediately reused:
-orca orchestration worker-release --dispatch --json
-# Acknowledge only after every message and required release decision is handled:
-orca orchestration check --ack --wait --types worker_done,escalation,question --timeout-ms 900000 --json
-```
-
-After processing each accepted `worker_done`, choose the terminal's next owner before you acknowledge the Delivery or wait again. If the same exact agent has an immediate follow-up Task, read the `worker.agent_terminal_handle` field of `worker-show --dispatch --json`, then run `orca orchestration worker-start --task --terminal --json` so Orca transfers cleanup ownership to the new Dispatch. Otherwise run `orca orchestration worker-release --dispatch --json`.
-
-Run `worker-release` after both succeeded and failed `worker_done` reports unless the user explicitly asked to keep that worker live. Release is post-completion cleanup, not cancellation: Orca first preserves inspectable output, then closes only the exact agent terminal owned by that settled Dispatch. Reused or pre-existing terminals, setup terminals, coordinators, active workers, user-taken-over terminals, and identities Orca cannot prove are retained. If the user explicitly asks to keep the live terminal for debugging, record that exception with `orca orchestration worker-retain --dispatch --json` instead of silently skipping cleanup. When the user is finished, the same Dispatch can be passed to `worker-release`, which clears the requested retention and releases the terminal.
-
-Do not release a worker because of a timeout, TUI idle state, heartbeat, status, question, escalation, or rejected/stale `worker_done`. If release returns `release_pending` or `release_unknown`, do not substitute `terminal close`; follow the exact recovery action in the receipt. A replayed Delivery may repeat `worker-release` safely.
-
-Workers report exactly once using the IDs and capability injected by Orca; they do not supply Run/server/terminal identity:
-
-```bash
-orca orchestration send --type worker_done --subject "" --body "" --task-id --dispatch-id --outcome succeeded --files-modified "path/a,path/b" --json
-# On failure, use --outcome failed; never encode failure only in prose.
-```
-
-A worker question defaults to its owning Run. Timeout leaves it pending:
-
-```bash
-orca orchestration ask --question "" --options "yes,no" --timeout-ms 600000 --json
-orca orchestration ask --resume --timeout-ms 600000 --json
-# Coordinator:
-orca orchestration reply --id --body "" --json
-```
-
-Recovery is conditional, never a fixed destructive sequence:
-
-- The response was lost and named no Dispatch: run `orca orchestration request-show --request --json` first. It is read-only. `completed` means the mutation already took effect. `pending` means the original mutation is still running or Orca restarted before recording its outcome. For either state, replaying the original command with `--retry-request ` reuses the same operation identity so Orca can replay, join, or safely recover it without starting a separate duplicate. `absent` means this runtime holds no receipt under your caller identity and is not proof that nothing happened; inspect the affected state before deciding whether to retry.
-- `worker-show --dispatch ` says `ready`: keep waiting or read bounded output.
-- It proves `failed` or `stopped`: start a replacement with `worker-start --task --retry-of ` plus an explicit `--on`/`--worktree` and `--agent`/`--terminal` choice. Retry does not silently inherit placement.
-- It remains `outcome_unknown`: either `worker-stop --dispatch ` and inspect again, or explicitly `worker-abandon --dispatch ` while accepting that resources may still be live. Abandon performs no remote, process, or filesystem action.
-- `worker-stop` closes only the exact supervised agent terminal. It never deletes the worktree, setup terminal, configured tabs, or unrelated processes.
-
-Low-level `worktree create`, `terminal create`, and `dispatch --inject` remain valid recipes for custom argv or topology that `worker-start` does not express.
-
-`dispatch --inject` deliberately keeps an operator-started terminal unsupervised: it never creates a `worker_dispatches` row and `worker-stop`/`worker-abandon` never close that process. The dispatch context is still authoritative, so `worker-show`, `worker-read`, and `worker-list` report it as `unsupervised`; settled `worker-retain` and `worker-release` report `retained` with `no_owned_resource` and take no process action. Use `worker-start --terminal ` when supervision and worker lifecycle state are required.
-
-## Gates And Legacy Inspection
-
-```bash
-orca orchestration gate-create --task --question [--options ] [--json]
-orca orchestration gate-resolve --id --resolution [--json]
-orca orchestration gate-list [--task ] [--status ] [--json]
-```
-
-Use `ask` for worker-to-coordinator questions; it creates a `question` message that the coordinator answers with `reply`. Use `gate-create` only for coordinator-managed task DAG decisions, not for answering a worker's `ask`.
-
-`coordinator-start`, `coordinator-stop`, `run`, and `run-stop` are retired scheduler commands. They perform no effects and return the current-skill recovery action. They are not aliases for lightweight Run creation or binding.
-
-Recovery only: `orca orchestration reset --tasks|--messages|--all --json` clears the selected local orchestration database state. Do not run it during active coordination unless explicitly abandoning that state.
-
-## Full Handoffs
-
-For full ownership transfer, use non-lifecycle terminal/worktree commands and then stop monitoring unless the user asks for supervision.
-
-Treat these as full handoff requests by default: "hand off", "handoff", "handover", "give this to another agent", "give this to another worktree", "send this to another agent", "another agent", "another worktree", or "launch another agent to own this." Custom model or reasoning effort words such as `gpt-5.5`, `high`, or `xhigh` do not make the handoff supervised.
-
-Supervised orchestration remains available only when the user explicitly asks for supervision or coordination: "supervise", "monitor", "wait for worker_done", "wait for results", "track completion", "DAG", "decision gate", "ask/reply", or "coordinate workers."
-
-Do not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. `task-create` is also forbidden because it records coordinator-owned tracking state; if a task row is needed, the user asked for supervised orchestration. Do not create a `taskId`/`dispatchId`, inject a lifecycle preamble, wait for completion, or read the worker terminal after prompt delivery except to avoid losing the initial prompt.
-
-New top-level worktree handoff:
-
-```bash
-orca worktree create --name --no-parent --agent codex --prompt "" --setup run --json
-```
-
-Before creating a new worktree from an active feature branch, decide and state whether the desired Orca lineage is child or top-level. Use child worktree lineage only when the new work is conceptually stacked under or dependent on the active worktree. For independent repo-wide fixes, standalone feature work, or unrelated follow-up tasks, create a top-level worktree with `--no-parent`.
-
-Existing terminal handoff:
-
-```bash
-orca terminal send --terminal --text "" --enter --json
-```
-
-Custom Codex model/effort handoff:
-
-`orca worktree create --agent codex --prompt ...` launches the known Codex agent but does not accept Codex-specific `--model` or `-c model_reasoning_effort=...` arguments. When the user asks for a specific Codex model or effort, create the independent worktree first, launch Codex with the requested command in that worktree, wait only for TUI readiness if prompt delivery would otherwise race startup, send the prompt, and stop.
-
-The two-step custom-argv path cannot enforce a repository's explicit `wait-for-setup` startup policy because the later `terminal create` is not the startup owned by `worktree create`. Use it only when the repository starts agents immediately. If the repository requires `wait-for-setup`, use an agent-first configured launcher that can preserve sequencing, or stop and ask rather than silently bypassing the policy.
-
-Note: when no repo default-terminal configuration supplies a primary terminal, bare create opens a fallback shell before `terminal create` adds the agent. Configured default tabs are materialized instead and may run real commands. Prefer `--agent` whenever custom argv is not required. With the two-step path, target only the agent handle; close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell.
-
-Use the exact full `::` worktree id returned by `orca worktree create --json`; a bare repo id cannot target the new worktree.
-
-```bash
-orca worktree create --name --no-parent --setup run --json
-orca terminal create --worktree id: --title --command 'codex --model gpt-5.5 -c model_reasoning_effort="xhigh"' --json
-orca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json
-orca terminal send --terminal --text "" --enter --json
-```
-
-Wait only for `tui-idle` when needed to avoid losing the prompt. Do not monitor task completion.
-
-`--no-parent` only controls Orca lineage; it does not choose the Git base. If the work should start from the repo default base, omit `--base-branch` so Orca uses that default, or explicitly pass the repo default base (`origin/main`, `origin/master`, or the `orca repo show --repo --json` value); never base it on the current feature branch unless the user explicitly asks for stacked work or "branch from current". Put current-branch context in the prompt instead.
-
-## Worker Terminals
-
-Choose the worker location before creating a terminal. `Fresh worker` means a fresh agent session, not a new git worktree. For parallel work, create one fresh agent terminal per worker in the same required worktree, falling back to the active worktree when none is named. If the task says current worktree only, depends on uncommitted files/artifacts, or must validate/PR the current branch, keep every worker in the active worktree:
-
-```bash
-orca terminal create --worktree active --title --command "codex" --json
-orca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json
-orca orchestration dispatch --task --to --inject --json
-```
-
-Reuse an idle agent in the required worktree only if the prompt allows reuse; otherwise create a fresh terminal there. Create a new worktree only when the user explicitly requests one or a concrete checkout or filesystem conflict makes sharing unsafe or impossible; if the user did not request it, state that conflict before running `worktree create`. Independent tasks, parallel execution, convenience, or a preference for separate checkouts are not isolation requirements.
-
-When a new worktree is allowed, use child lineage for isolated work that is stacked under or dependent on the active worktree, and use `--no-parent` when it is not stacked. Decide the Git base separately: `--no-parent` makes the worktree top-level in Orca, while omitted `--base-branch` uses the repo default base.
-
-For every new worktree, pass `--setup run` so any configured repository setup hook runs. This does not mean waiting for setup before agent launch: preserve the repository's startup policy, whose default starts setup and the agent side by side. Use `--setup skip` or `--setup inherit` only when there is a concrete task-specific reason, and state that reason before creating the worktree. This rule does not rerun setup for current or existing worktrees.
-
-```bash
-orca worktree create --name --agent codex --setup run --json
-# or: --agent claude | omp | pi | grok | ...
-# Read from agentTerminalHandle, falling back to startupTerminal.handle.
-orca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json
-orca orchestration dispatch --task --to --inject --json
-```
-
-For new-worktree workers, read the id and `agentTerminalHandle` from `worktree create`, falling back to `startupTerminal.handle` for older runtimes. Use that as the sole worker handle when present; otherwise use `terminal list` to resolve the agent handle. Omit `--repo` only inside an Orca-managed worktree; otherwise pass `--repo `.
-
-**For an allowed new worktree, use agent-first:** `--agent` reveals the new worktree and launches the selected agent **in its first terminal**, without adding a separate fallback shell for that worker. Pass `--setup run`; repo setup and default-terminal settings may add intentional tabs or splits. Do **not** run bare `worktree create` and then `terminal create --command ` for the same worker when agent-first create is available: without configured default tabs, that two-step path leaves a fallback shell + agent pair. Only use it when custom agent argv is required (for example Codex model/effort flags) or when an older CLI rejects `--agent`; if you must, message only the agent handle. Configured default tabs are intentional surfaces, so close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell. Do not run `worktree create` when the task must stay in the current worktree.
-
-Use `orca worktree create --prompt ...` or `orca terminal send ...` for full handoffs or untracked/lightweight prompts. Those paths do not attach `taskId`/`dispatchId`; the worker should not send lifecycle messages unless the prompt supplies a live orchestration preamble.
-
-Sidebar lineage and orchestration lifecycle are related but not identical. A same-worktree worker may appear as a peer under that worktree in the sidebar while remaining a child dispatch in orchestration state; only an actual child worktree creates visible parent/child worktree lineage.
-
-Other terminal commands coordinators often need:
-
-```bash
-orca terminal list [--worktree ] [--include-visual-layouts] [--json]
-orca terminal create [--worktree ] [--title ] [--command ] [--json]
-orca terminal split --terminal [--direction horizontal|vertical] [--command ] [--json]
-orca terminal wait --terminal --for tui-idle --timeout-ms --json
-orca terminal read --terminal --json
-orca terminal send --terminal --text --enter --json
-```
-
-If an older CLI rejects `worktree create --agent`, create the worktree normally, then run `orca terminal create --worktree