Rebase custom-agents onto main (4/4): e2e, CI, config

SSH custom-agent e2e job, reliability gates, package scripts, plan docs.

Co-authored-by: Orca <help@stably.ai>
This commit is contained in:
Jinjing
2026-09-01 03:52:23 -07:00
co-authored by Orca
parent 6c1da09ac1
commit a8b08ffc36
14 changed files with 1500 additions and 371 deletions
+207 -129
View File
@@ -17,10 +17,6 @@ on:
description: JSON array of changed specs; empty runs the full suite
required: false
type: string
ssh_source_changed:
description: '"true" when the PR touches SSH execution source; gates the Docker-SSH lane'
required: false
type: string
workflow_dispatch:
inputs:
ref:
@@ -44,10 +40,36 @@ jobs:
with:
ref: ${{ inputs.ref || github.ref }}
# Why: the build's plain-Node daemon smoke load resolves node-pty.
- uses: ./.github/actions/install-node-dependencies
# Why: the E2E build compiles native modules via node-gyp. Mirrors the
# install step in pr.yml's verify job so E2E doesn't hit missing-toolchain
# errors.
- name: Install native build tools
run: sudo apt-get update && sudo apt-get install -y build-essential python3
# Why pnpm first: setup-node needs pnpm on PATH to locate the store it caches.
# Without that cache every E2E job re-downloaded the whole dependency set.
- name: Setup pnpm
uses: pnpm/action-setup@v6
with:
native-runtime: node
run_install: false
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
cache: pnpm
# Why: this job runs the same pnpm install path as pr.yml's verify
# job, so it needs the same pinned node-gyp override to avoid pnpm's
# broken bundled gyp_main.py on Linux.
- name: Use external node-gyp to avoid pnpm's bundled copy (Linux only)
if: runner.os == 'Linux'
run: |
npm install -g node-gyp@11.5.0
echo "npm_config_node_gyp=$(npm root -g)/node-gyp/bin/node-gyp.js" >> "$GITHUB_ENV"
- name: Install dependencies
run: pnpm install --frozen-lockfile
# Why: building here avoids parallel builds inside Playwright globalSetup;
# paired-browser specs also need the standalone web bundle.
@@ -55,13 +77,9 @@ jobs:
env:
VITE_EXPOSE_STORE: 'true'
run: |
status=0
pnpm run build:relay &
relay_pid=$!
npx electron-vite build --mode e2e || status=1
pnpm run build:web-from-renderer || status=1
wait "$relay_pid" || status=1
exit "$status"
npx electron-vite build --mode e2e
pnpm run build:web-from-renderer
pnpm run build:relay
- name: Upload E2E build output
uses: actions/upload-artifact@v7
@@ -75,29 +93,9 @@ jobs:
retention-days: 1
if-no-files-found: error
# Build Electron-native dependencies once per workflow. Consumer shards restore
# this immutable cache instead of compiling the same ABI concurrently.
prepare-native-cache:
name: prepare Electron native cache
runs-on: ubuntu-latest
timeout-minutes: 15
steps:
- name: Checkout
uses: actions/checkout@v6
with:
ref: ${{ inputs.ref || github.ref }}
- name: Install native build tools
run: sudo apt-get update && sudo apt-get install -y build-essential python3
- uses: ./.github/actions/install-node-dependencies
with:
native-runtime: electron
e2e:
name: e2e ${{ matrix.shard_name }}
needs: [build, prepare-native-cache]
needs: build
if: inputs.test_files == ''
runs-on: ubuntu-latest
timeout-minutes: 30
@@ -105,37 +103,26 @@ jobs:
fail-fast: false
matrix:
include:
# Fourteen scheduled runs averaged 24.6 minutes per shard; shards 4
# and 9 repeatedly hit the 30-minute cap. A 12-way trial still left
# one 30-minute shard, so 14 gives the suite enough failure headroom.
- shard: '1/14'
shard_name: 1-of-14
- shard: '2/14'
shard_name: 2-of-14
- shard: '3/14'
shard_name: 3-of-14
- shard: '4/14'
shard_name: 4-of-14
- shard: '5/14'
shard_name: 5-of-14
- shard: '6/14'
shard_name: 6-of-14
- shard: '7/14'
shard_name: 7-of-14
- shard: '8/14'
shard_name: 8-of-14
- shard: '9/14'
shard_name: 9-of-14
- shard: '10/14'
shard_name: 10-of-14
- shard: '11/14'
shard_name: 11-of-14
- shard: '12/14'
shard_name: 12-of-14
- shard: '13/14'
shard_name: 13-of-14
- shard: '14/14'
shard_name: 14-of-14
- shard: '1/10'
shard_name: 1-of-10
- shard: '2/10'
shard_name: 2-of-10
- shard: '3/10'
shard_name: 3-of-10
- shard: '4/10'
shard_name: 4-of-10
- shard: '5/10'
shard_name: 5-of-10
- shard: '6/10'
shard_name: 6-of-10
- shard: '7/10'
shard_name: 7-of-10
- shard: '8/10'
shard_name: 8-of-10
- shard: '9/10'
shard_name: 9-of-10
- shard: '10/10'
shard_name: 10-of-10
steps:
- name: Checkout
@@ -143,14 +130,46 @@ jobs:
with:
ref: ${{ inputs.ref || github.ref }}
# Native cache misses need the compiler, Electron needs Xvfb, and paired
# Quick Open needs ripgrep. Install them in one apt transaction per shard.
- name: Install native build and headless UI tools
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk python3 ripgrep xvfb zsh
# Why: pnpm install rebuilds native modules, and those postinstall
# scripts still need the Linux toolchain even though this shard reuses
# the prebuilt Electron output.
- name: Install native build tools
# Why: paired Quick Open coverage exercises the resource-bounded host search.
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk python3 ripgrep zsh
- uses: ./.github/actions/install-node-dependencies
# Why: Electron on Linux needs an X display even when the app
# suppresses mainWindow.show() via ORCA_E2E_HEADLESS. xvfb provides a
# virtual framebuffer so Chromium can initialize without a real display.
- name: Install xvfb
run: sudo apt-get install -y xvfb
# Why pnpm first: setup-node needs pnpm on PATH to locate the store it caches.
# Without that cache every E2E job re-downloaded the whole dependency set.
- name: Setup pnpm
uses: pnpm/action-setup@v6
with:
native-runtime: electron
run_install: false
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
cache: pnpm
# Why: this job runs the same pnpm install path as pr.yml's verify
# job, so it needs the same pinned node-gyp override to avoid pnpm's
# broken bundled gyp_main.py on Linux. Gate on runner.os matches
# release.yml so the invariant "this workaround is Linux-only" is
# consistent across all three workflows, even though this job
# currently pins runs-on: ubuntu-latest.
- name: Use external node-gyp to avoid pnpm's bundled copy (Linux only)
if: runner.os == 'Linux'
run: |
npm install -g node-gyp@11.5.0
echo "npm_config_node_gyp=$(npm root -g)/node-gyp/bin/node-gyp.js" >> "$GITHUB_ENV"
- name: Install dependencies
run: pnpm install --frozen-lockfile
- name: Download E2E build output
uses: actions/download-artifact@v8
@@ -183,7 +202,7 @@ jobs:
changed-e2e:
name: changed e2e specs
needs: [build, prepare-native-cache]
needs: build
if: inputs.test_files != ''
runs-on: ubuntu-latest
# Why 45: pr.yml now maps SSH source edits onto Docker-backed specs, so this lane can
@@ -203,9 +222,25 @@ jobs:
# lane now receives those specs from pr.yml's SSH source mapping.
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 ripgrep xvfb zsh
- uses: ./.github/actions/install-node-dependencies
# Why pnpm first: setup-node needs pnpm on PATH to locate the store it caches.
# Without that cache every E2E job re-downloaded the whole dependency set.
- name: Setup pnpm
uses: pnpm/action-setup@v6
with:
native-runtime: electron
run_install: false
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
cache: pnpm
- name: Use external node-gyp to avoid pnpm's bundled copy
run: |
npm install -g node-gyp@11.5.0
echo "npm_config_node_gyp=$(npm root -g)/node-gyp/bin/node-gyp.js" >> "$GITHUB_ENV"
- name: Install dependencies
run: pnpm install --frozen-lockfile
- name: Download E2E build output
uses: actions/download-artifact@v8
@@ -217,16 +252,12 @@ jobs:
env:
TEST_FILES_JSON: ${{ inputs.test_files }}
run: |
# Why the native IME spec is dropped: it test.skip()s itself without
# ORCA_E2E_NATIVE_IBUS_HANGUL, which this lane cannot set because it has no ibus
# session. Running it here reported a green skip as coverage.
mapfile -t TEST_FILES < <(jq -r '.[] | select(
. != "tests/e2e/ssh-startup-exec-readiness.spec.ts" and
. != "tests/e2e/paired-startup-exec-readiness.spec.ts" and
. != "tests/e2e/terminal-ibus-hangul-native.spec.ts"
. != "tests/e2e/paired-startup-exec-readiness.spec.ts"
)' <<<"$TEST_FILES_JSON")
if [ "${#TEST_FILES[@]}" -eq 0 ]; then
echo "Changed specs are all owned by dedicated lanes."
echo "Changed startup-readiness specs are owned by the dedicated live lane."
exit 0
fi
E2E_ENV=(SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 ORCA_E2E_WEB_CLIENT=1 ORCA_RELAY_PATH="$GITHUB_WORKSPACE/out/relay")
@@ -255,21 +286,15 @@ jobs:
ssh-docker-watcher-isolation:
name: ssh docker watcher isolation
needs: [build, prepare-native-cache]
# effect of one route listing a startup-readiness spec — pruning that spec would have
# silently retired the whole lane. The signal is now derived from the SSH routes directly.
# The two spec clauses stay for their honest purpose: changed-e2e hands these specs to this
# lane, so editing one must still run it here.
needs: build
if: >-
inputs.test_files == '' ||
inputs.ssh_source_changed == 'true' ||
contains(inputs.test_files, 'tests/e2e/ssh-startup-exec-readiness.spec.ts') ||
contains(inputs.test_files, 'tests/e2e/paired-startup-exec-readiness.spec.ts')
runs-on: ubuntu-latest
# Why 60: this lane now also runs the remaining Docker-SSH specs serially. They average
# ~18s but several budget 4-10 minutes per test, so a slow run lands far above the old 35
# — and the sharded lanes already show that a lane which times out is a lane nobody trusts.
timeout-minutes: 60
# Why 35: the parking, retention, startup-exec, and paired parity specs run
# serially on isolated Electron/SSH fixtures after watcher isolation.
timeout-minutes: 35
steps:
- name: Checkout
@@ -280,9 +305,29 @@ jobs:
- name: Install native build and headless UI tools
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 xvfb zsh
- uses: ./.github/actions/install-node-dependencies
# Why pnpm first: setup-node needs pnpm on PATH to locate the store it caches.
# Without that cache every E2E job re-downloaded the whole dependency set.
- name: Setup pnpm
uses: pnpm/action-setup@v6
with:
native-runtime: electron
run_install: false
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
cache: pnpm
# Why: same Linux-only node-gyp pin as build/e2e jobs so the workaround
# stays consistent across workflows even while this job is ubuntu-latest.
- name: Use external node-gyp to avoid pnpm's bundled copy (Linux only)
if: runner.os == 'Linux'
run: |
npm install -g node-gyp@11.5.0
echo "npm_config_node_gyp=$(npm root -g)/node-gyp/bin/node-gyp.js" >> "$GITHUB_ENV"
- name: Install dependencies
run: pnpm install --frozen-lockfile
- name: Download E2E build output
uses: actions/download-artifact@v8
@@ -295,52 +340,85 @@ jobs:
- name: Run Docker SSH watcher isolation E2E
run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-watcher-isolation
# Why: Playwright empties test-results/ when it starts, so each step here used to
# destroy the previous step's traces. Only the last lane's failure was ever
# diagnosable from the artifact; set each lane aside before the next one runs.
- name: Keep watcher-isolation traces
if: always()
run: |
if [ -d test-results ]; then
mkdir -p e2e-traces
mv test-results "e2e-traces/watcher-isolation"
fi
# Why always(): this lane gates SSH parking/retention plus startup-exec
# readiness across live SSH, headed paired, and headless serve topologies.
- name: Run Docker SSH terminal parking + startup readiness E2E
if: always()
run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-terminal-parking
- name: Keep terminal-parking traces
if: always()
run: |
if [ -d test-results ]; then
mkdir -p e2e-traces
mv test-results "e2e-traces/terminal-parking"
fi
# Why here rather than the sharded lanes: the shards set no ORCA_E2E_SSH_DOCKER, so every
# spec below skipped itself while the shard still reported green. Running them on this one
# VM pays the fixture image build once instead of ten times, and keeps an SSH regression
# legible as an SSH-named failure.
- name: Run remaining Docker SSH E2E
if: always()
run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker
- name: Keep remaining-ssh-docker traces
if: always()
run: |
if [ -d test-results ]; then
mkdir -p e2e-traces
mv test-results "e2e-traces/remaining-ssh-docker"
fi
- name: Upload watcher isolation traces
if: failure()
uses: actions/upload-artifact@v7
with:
name: playwright-traces-ssh-docker-watcher-isolation
path: e2e-traces/
# Why: §U10 live-evidence job for the SSH cell of agent-launch.resolution-
# attribution. It boots a throwaway Docker SSH container, seeds a custom agent
# via the persisted profile, launches it through the host agentLaunch boundary
# into the REMOTE worktree, and observes the spawned process's argv+env from
# /proc inside the container — proving one host resolution produced one remote
# PTY with the exact custom executable/args/env, and that reconnect never
# duplicates the launch. Ubuntu-only because macOS/Windows GitHub runners
# cannot host a Linux Docker container (the platform matrix in pr.yml covers
# the pure-resolution cross-OS cells instead).
ssh-custom-agent:
name: e2e ssh custom agent
needs: build
runs-on: ubuntu-latest
timeout-minutes: 25
steps:
- name: Checkout
uses: actions/checkout@v6
with:
ref: ${{ inputs.ref || github.ref }}
- name: Install native build tools
run: sudo apt-get update && sudo apt-get install -y build-essential python3
# Why: Electron on Linux needs an X display even when the app suppresses
# mainWindow.show() via ORCA_E2E_HEADLESS; xvfb provides a virtual
# framebuffer so Chromium can initialize without a real display.
- name: Install xvfb
run: sudo apt-get install -y xvfb
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
- name: Setup pnpm
uses: pnpm/action-setup@v6
with:
run_install: false
# Why: same pinned node-gyp override as pr.yml's verify job to avoid pnpm's
# broken bundled gyp_main.py on Linux.
- name: Use external node-gyp to avoid pnpm's bundled copy (Linux only)
if: runner.os == 'Linux'
run: |
npm install -g node-gyp@11.5.0
echo "npm_config_node_gyp=$(npm root -g)/node-gyp/bin/node-gyp.js" >> "$GITHUB_ENV"
- name: Install dependencies
run: pnpm install --frozen-lockfile
- name: Download E2E build output
uses: actions/download-artifact@v8
with:
name: e2e-build-out
path: out/
# Why: SKIP_BUILD reuses the single build job's Electron artifact;
# globalSetup still builds the SSH relay bundle because the runner sets
# ORCA_E2E_SSH_DOCKER=1. Docker is preinstalled on ubuntu-latest runners.
- name: Run SSH custom-agent e2e
run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-custom-agent
- name: Upload Playwright traces
if: failure()
uses: actions/upload-artifact@v7
with:
name: playwright-traces-ssh-custom-agent
path: test-results/
retention-days: 7
if-no-files-found: ignore
+165 -237
View File
@@ -16,30 +16,15 @@ permissions:
contents: read
jobs:
# Why: a README/docs-only PR used to start the full matrix (test shards,
# Why: a README/docs-only PR used to start the full matrix (32 test shards,
# two package jobs, typecheck, git compat, xterm, shell contracts). Path
# filters on `on.pull_request` would drop the `verify` check entirely; this
# detector keeps verify as the required aggregate and skips the expensive jobs.
# Per-job outputs also skip git-compat/xterm/packaging/shell when those
# inputs are unchanged; empty diffs fail closed and run everything.
code_paths:
name: detect code-relevant changes
runs-on: ubuntu-latest
outputs:
should_run: ${{ steps.filter.outputs.should_run }}
native_cache_changed: ${{ steps.filter.outputs.native_cache_changed }}
static_analysis: ${{ steps.filter.outputs.static_analysis }}
typecheck: ${{ steps.filter.outputs.typecheck }}
git_compatibility: ${{ steps.filter.outputs.git_compatibility }}
codex_index_heal_contract: ${{ steps.filter.outputs.codex_index_heal_contract }}
xterm_patch_sync: ${{ steps.filter.outputs.xterm_patch_sync }}
shell_contracts: ${{ steps.filter.outputs.shell_contracts }}
test: ${{ steps.filter.outputs.test }}
orcad_browser: ${{ steps.filter.outputs.orcad_browser }}
cross-version-wire: ${{ steps.filter.outputs.cross-version-wire }}
managed_hook_node18: ${{ steps.filter.outputs.managed_hook_node18 }}
package: ${{ steps.filter.outputs.package }}
package_windows: ${{ steps.filter.outputs.package_windows }}
steps:
- name: Checkout
uses: actions/checkout@v6
@@ -63,12 +48,14 @@ jobs:
CHANGED="$(git diff --name-only --no-renames --diff-filter=ACDMR --merge-base "$BASE_SHA" "$HEAD_SHA")"
echo "Changed paths:"
printf '%s\n' "$CHANGED"
printf '%s\n' "$CHANGED" | node config/scripts/pr-code-change-scope.mjs | tee -a "$GITHUB_OUTPUT"
SHOULD_RUN="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-code-change-scope.mjs)"
echo "should_run=$SHOULD_RUN" >> "$GITHUB_OUTPUT"
echo "should_run=$SHOULD_RUN"
static_analysis:
name: static analysis
needs: [code_paths]
if: needs.code_paths.outputs.static_analysis == 'true'
if: needs.code_paths.outputs.should_run == 'true'
runs-on: ubuntu-latest
steps:
@@ -83,8 +70,6 @@ jobs:
persist-credentials: false
- uses: ./.github/actions/install-node-dependencies
with:
native-runtime: node
- name: Lint
run: pnpm exec oxlint --format github
@@ -131,9 +116,6 @@ jobs:
- name: Enforce max-lines ratchet
run: pnpm run check:max-lines-ratchet
- name: Enforce ts-nocheck ratchet
run: pnpm run check:ts-nocheck-ratchet
- name: Enforce runtime Electron-import ratchet
run: pnpm run check:runtime-electron-ratchet
@@ -206,7 +188,7 @@ jobs:
typecheck:
needs: [code_paths]
if: needs.code_paths.outputs.typecheck == 'true'
if: needs.code_paths.outputs.should_run == 'true'
runs-on: ubuntu-latest
steps:
@@ -218,23 +200,23 @@ jobs:
- uses: ./.github/actions/install-node-dependencies
# Why: every project is `composite`, so tsc already writes a .tsbuildinfo that lets
# the next run skip unchanged files. Share one cache entry across commits while the
# PR base stays stable; actions/cache keeps the first successful graph and the
# compiler still invalidates stale files from its content hashes.
# the next run skip unchanged files. CI threw it away each time. The key is per-SHA
# so each run saves its own; restore-keys inherit the newest prior graph to diff against.
- name: Cache TypeScript incremental state
uses: actions/cache@v5
with:
path: config/*.tsbuildinfo
key: tsbuildinfo-${{ runner.os }}-${{ hashFiles('pnpm-lock.yaml', 'config/tsconfig*.json') }}-${{ github.event.pull_request.base.sha }}
key: tsbuildinfo-${{ runner.os }}-${{ hashFiles('pnpm-lock.yaml', 'config/tsconfig*.json') }}-${{ github.sha }}
restore-keys: |
tsbuildinfo-${{ runner.os }}-${{ hashFiles('pnpm-lock.yaml', 'config/tsconfig*.json') }}-
tsbuildinfo-${{ runner.os }}-
- run: pnpm run typecheck
git_compatibility:
name: Git compatibility
needs: [code_paths]
if: needs.code_paths.outputs.git_compatibility == 'true'
if: needs.code_paths.outputs.should_run == 'true'
runs-on: ubuntu-latest
steps:
@@ -298,51 +280,10 @@ jobs:
done
exit "$status"
# Why this job: Orca's session index-heal depends on a Codex behavior — a
# `thread/read` of an unindexed rollout performs a read-repair that inserts the
# `threads` row. Every unit test drives a stub app-server and asserts only that the
# call did not error, so if Codex dropped the repair they would all stay green while
# the subsystem went inert. This runs the pinned real binary and fails when the
# repair stops happening. Pinned because the binary is the thing expected to drift.
codex_index_heal_contract:
name: Codex index-heal contract
needs: [code_paths]
if: needs.code_paths.outputs.codex_index_heal_contract == 'true'
runs-on: ubuntu-latest
env:
CODEX_CLI_VERSION: '0.150.1'
steps:
- name: Checkout
uses: actions/checkout@v6
with:
persist-credentials: false
- uses: ./.github/actions/install-node-dependencies
- name: Install pinned Codex CLI
run: |
set -euo pipefail
npm install --no-audit --no-fund --prefix "$RUNNER_TEMP/codex-cli" \
"@openai/codex@$CODEX_CLI_VERSION"
- name: Verify Codex index-heal contract
env:
# Why REQUIRED: without a binary the suite skips, and a job that skips
# reports success. This turns a failed or missing install into a red test
# instead of a green no-op.
ORCA_CODEX_CONTRACT_REQUIRED: '1'
ORCA_CODEX_CONTRACT_VERSION: ${{ env.CODEX_CLI_VERSION }}
run: |
set -euo pipefail
ORCA_CODEX_CONTRACT_BINARY="$RUNNER_TEMP/codex-cli/node_modules/.bin/codex" \
pnpm exec vitest run --config config/vitest.config.ts \
src/main/codex/codex-index-heal-binary-contract.test.ts
xterm_patch_sync:
name: xterm patch sync
needs: [code_paths]
if: needs.code_paths.outputs.xterm_patch_sync == 'true'
if: needs.code_paths.outputs.should_run == 'true'
runs-on: ubuntu-latest
steps:
@@ -353,12 +294,9 @@ jobs:
- uses: ./.github/actions/install-node-dependencies
# Why: the check rebuilds every package in the manifest from a pinned upstream
# commit — @xterm/xterm and the two addons, each built twice (once unmodified to
# prove the toolchain still reproduces the published bundles, once patched). Caching
# the npm metadata and the shallow clone keeps the repeated cost to the builds
# themselves; the key is the manifest, so a commit, package or toolchain bump
# invalidates it.
# Why: the check rebuilds xterm.js from a pinned upstream commit. Caching the
# npm metadata and the shallow clone turns a ~4 min cold run into well under a
# minute; the key is the manifest, so a commit or toolchain bump invalidates it.
- name: Restore upstream xterm build inputs
uses: actions/cache@v5
with:
@@ -375,7 +313,7 @@ jobs:
shell_contracts:
name: shell contracts
needs: [code_paths]
if: needs.code_paths.outputs.shell_contracts == 'true'
if: needs.code_paths.outputs.should_run == 'true'
runs-on: ubuntu-latest
# Why: this job's cost is almost entirely package download, and a stalled mirror has
# no wall-clock bound of its own. A successful run finishes in ~4.5 minutes, so this
@@ -487,17 +425,20 @@ jobs:
src/main/zsh-wrapper-version-mismatch.live-shell.test.ts \
src/renderer/src/components/terminal-pane/fish-color-scheme-child-stdin.node-pty.test.ts \
src/shared/fish-query-reply-child-stdin.node-pty.test.ts \
src/shared/pty-reply-echo-shapes.node-pty.test.ts \
src/shared/startup-shell-portability.live-shell.test.ts \
src/shared/posix-command-path-lookup.test.ts
# Cache-key input changes would otherwise make every shard compile the same
# native addon concurrently. Prime the supported Node ABI before the matrix fans out.
test_native_cache:
name: prepare test native cache node 24
test:
name: tests node ${{ matrix.node }} ${{ matrix.shard }}/${{ matrix.shard_total }}
needs: [code_paths]
if: needs.code_paths.outputs.native_cache_changed == 'true'
if: needs.code_paths.outputs.should_run == 'true'
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
node: ['24', '26']
shard: [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16]
shard_total: [16]
steps:
- name: Checkout
@@ -508,24 +449,38 @@ jobs:
- uses: ./.github/actions/install-node-dependencies
with:
native-runtime: node
node-version: '24'
node-version: ${{ matrix.node }}
test:
needs: [code_paths, test_native_cache]
if: >-
always() &&
needs.code_paths.outputs.test == 'true' &&
(needs.test_native_cache.result == 'success' || needs.test_native_cache.result == 'skipped')
uses: ./.github/workflows/unit-tests.yml
with:
node_versions: '["24"]'
- name: Install Electron package binary for tests
run: node config/scripts/install-electron-package-binary.mjs
# Why a separate job: the test needs a real Chrome, and the sharded `test` matrix
# would pay for it on every shard to run one file in whichever shard it landed in.
- name: Test shard
run: |
pnpm exec vitest run --config config/vitest.config.ts \
--exclude=src/main/daemon/repro-13767-shell-ready-marker-lost-to-exec.test.ts \
--exclude=src/main/daemon/shell-ready.test.ts \
--exclude=src/main/daemon/node-pty-fd-leak.test.ts \
--exclude=src/main/providers/local-pty-shell-ready-zsh-launch-environment.test.ts \
--exclude=src/main/providers/__tests__/shell-ready-framework-example.test.ts \
--exclude=src/main/pty/omp-shell-wrapper.node-pty.test.ts \
--exclude=src/main/shell-startup-feature-channel.test.ts \
--exclude=src/main/terminal-history-fish-session.node-pty.test.ts \
--exclude=src/main/zsh-scoped-histfile.live-shell.test.ts \
--exclude=src/main/zsh-startup-hook-user-config-equivalence.live-shell.test.ts \
--exclude=src/main/zsh-wrapper-version-mismatch.live-shell.test.ts \
--exclude=src/renderer/src/components/terminal-pane/fish-color-scheme-child-stdin.node-pty.test.ts \
--exclude=src/shared/fish-query-reply-child-stdin.node-pty.test.ts \
--exclude=src/shared/startup-shell-portability.live-shell.test.ts \
--exclude=src/shared/posix-command-path-lookup.test.ts \
--exclude=tests/e2e/cross-version-wire/** \
--shard=${{ matrix.shard }}/${{ matrix.shard_total }}
# Why a separate job: the test needs a real Chrome, and the 32-way `test` matrix
# would pay for it 32 times to run one file in whichever shard it landed in.
orcad_browser:
name: orcad browser provider
needs: [code_paths]
if: needs.code_paths.outputs.orcad_browser == 'true'
if: needs.code_paths.outputs.should_run == 'true'
runs-on: ubuntu-latest
steps:
@@ -561,7 +516,7 @@ jobs:
cross-version-wire:
name: cross-version wire compatibility
needs: [code_paths]
if: needs.code_paths.outputs.cross-version-wire == 'true'
if: needs.code_paths.outputs.should_run == 'true'
runs-on: ubuntu-latest
steps:
@@ -584,19 +539,13 @@ jobs:
# A path filter that matches nothing exits 1 ("No test files found"), so this
# lane cannot report success while running zero tests.
- name: Old/new client and server compatibility journeys
run: >-
pnpm exec vitest run --config config/vitest.config.ts
tests/e2e/cross-version-wire/release-checkout.unit.test.ts
tests/e2e/cross-version-wire/cross-version-browser-placement.unit.test.ts
tests/e2e/cross-version-wire/cross-version-terminal-wire.unit.test.ts
tests/e2e/cross-version-wire/reported-lossy-initial-snapshot.unit.test.ts
tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts
- name: Old/new client and server terminal journey
run: pnpm exec vitest run --config config/vitest.config.ts tests/e2e/cross-version-wire/cross-version-terminal-wire.unit.test.ts
managed_hook_node18:
name: managed hooks on Node 18
needs: [code_paths]
if: needs.code_paths.outputs.managed_hook_node18 == 'true'
if: needs.code_paths.outputs.should_run == 'true'
runs-on: ubuntu-latest
steps:
@@ -621,7 +570,7 @@ jobs:
package:
name: package
needs: [code_paths]
if: needs.code_paths.outputs.package == 'true'
if: needs.code_paths.outputs.should_run == 'true'
runs-on: ubuntu-latest
steps:
@@ -633,7 +582,9 @@ jobs:
- name: Cache electron-builder downloads
uses: actions/cache@v5
with:
path: ~/.cache/electron-builder
path: |
~/.cache/electron
~/.cache/electron-builder
key: electron-builder-linux-${{ hashFiles('pnpm-lock.yaml') }}
restore-keys: |
electron-builder-linux-
@@ -642,19 +593,6 @@ jobs:
with:
native-runtime: electron
# Why --no-file-parallelism: every file here launches a full Electron stack twice, and each
# probe carries its own in-process deadline. Four at once on a 4-vCPU runner starve each other
# past those deadlines; serial, every probe owns the runner.
- name: Test Linux Electron lifecycle boundary
run: >-
xvfb-run --auto-servernum pnpm exec vitest run --config config/vitest.config.ts
--no-file-parallelism
src/main/browser/browser-client-page-renderer-lifecycle.electron.test.ts
src/main/browser/browser-route-tcp-egress.electron.test.ts
src/main/browser/browser-route-webrtc-egress.electron.test.ts
src/main/browser/browser-route-h3-egress.electron.test.ts
src/main/browser/browser-route-dns-prefetch.electron.test.ts
- name: Build package inputs
run: |
status=0
@@ -695,7 +633,7 @@ jobs:
package_windows:
name: package (windows)
needs: [code_paths]
if: needs.code_paths.outputs.package_windows == 'true'
if: needs.code_paths.outputs.should_run == 'true'
runs-on: windows-2022
timeout-minutes: 30
@@ -705,6 +643,17 @@ jobs:
with:
persist-credentials: false
- name: Setup pnpm
uses: pnpm/action-setup@v6
with:
run_install: false
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
cache: pnpm
- name: Cache electron-builder downloads
uses: actions/cache@v5
with:
@@ -715,37 +664,26 @@ jobs:
restore-keys: |
electron-builder-windows-
# Why persist-native-cache false: this job later rebuilds the same path for
# Electron. A post-job save would store the Electron ABI under the Node key.
- uses: ./.github/actions/install-node-dependencies
id: deps
with:
native-runtime: node
persist-native-cache: 'false'
- name: Install dependencies
run: pnpm install --frozen-lockfile
- name: Save compiled Node native modules
if: steps.deps.outputs.native-cache-hit != 'true'
uses: actions/cache/save@v5
with:
path: |
node_modules/.pnpm/node-pty@*/node_modules/node-pty/build
node_modules/.pnpm/windows-native-registry@*/node_modules/windows-native-registry/build
node_modules/.pnpm/@vscode+windows-process-tree@*/node_modules/@vscode/windows-process-tree/build
key: native-modules-${{ runner.os }}-${{ steps.deps.outputs.native-cache-scope }}-${{ runner.arch }}-node-node${{ steps.deps.outputs.node-version }}-${{ hashFiles('pnpm-lock.yaml', '.github/actions/install-node-dependencies/action.yml', 'config/scripts/ensure-native-runtime.mjs', 'config/scripts/rebuild-native-deps.mjs', 'config/patches/node-pty@1.1.0.patch', 'config/patches/@vscode__windows-process-tree@0.8.0.patch') }}
# Why: node-pty prefers its upstream prebuild, which does not contain
# Orca's Windows patch, so the job-object exports would be absent and the
# suite below would test an unpatched binary. build_from_source removes
# the prebuild, and the package's postinstall restores the ConPTY runtime
# files that a bare node-gyp rebuild would miss.
- name: Rebuild node-pty from patched source
env:
npm_config_build_from_source: 'true'
run: pnpm rebuild node-pty
- name: Test Windows-specific boundaries
run: >-
pnpm exec vitest run --config config/vitest.config.ts
config/scripts/rebuild-native-deps.test.mjs
src/main/browser/browser-client-page-renderer-lifecycle.electron.test.ts
src/main/browser/browser-route-tcp-egress.electron.test.ts
src/main/browser/browser-route-webrtc-egress.electron.test.ts
src/main/browser/browser-route-h3-egress.electron.test.ts
src/main/browser/browser-route-dns-prefetch.electron.test.ts
src/main/providers/windows-conpty-wide-char-duplication.node-pty.test.ts
src/main/providers/pty-repaint-wide-char-buffer.node-pty.test.ts
src/shared/child-process/windows-command-line.win32.test.ts
src/main/agent-hooks/windows-hook-payload-delivery.test.ts
src/main/windows/windows-pty-job.win32.test.ts
src/main/windows/windows-host-job.win32.test.ts
src/main/wsl/wsl-runner.test.ts
@@ -755,7 +693,6 @@ jobs:
src/main/wsl/wsl-w1-w3-contract.test.ts
src/shared/source-scan/source-tree-scan.test.ts
src/main/cli/wsl-cli-powershell-boundary.test.ts
src/main/cursor/hook-service.test.ts
src/main/orca-profiles/profile-index-store.test.ts
src/main/runtime/repo-worktree-admin-fingerprint.test.ts
src/main/runtime/worktree-scan-admin-fingerprint-gate.test.ts
@@ -766,26 +703,9 @@ jobs:
# Why the :parallel variant: identical to build:release except the three
# electron-vite targets overlap instead of running back to back. The Linux package
# job already packages and smoke-tests an AppImage built that way.
- name: Cache Windows CLI launcher
uses: actions/cache@v5
with:
path: native/windows-cli-launcher/.build
key: windows-cli-launcher-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('native/windows-cli-launcher/**', 'config/scripts/build-windows-cli-launcher.mjs') }}
- name: Build package inputs
env:
ORCA_REUSE_WINDOWS_CLI_LAUNCHER: '1'
run: pnpm run build:release:parallel
- name: Restore compiled Electron native modules
uses: actions/cache@v5
with:
path: |
node_modules/.pnpm/node-pty@*/node_modules/node-pty/build
node_modules/.pnpm/windows-native-registry@*/node_modules/windows-native-registry/build
node_modules/.pnpm/@vscode+windows-process-tree@*/node_modules/@vscode/windows-process-tree/build
key: native-modules-${{ runner.os }}-${{ steps.deps.outputs.native-cache-scope }}-${{ runner.arch }}-electron-node${{ steps.deps.outputs.node-version }}-${{ hashFiles('pnpm-lock.yaml', '.github/actions/install-node-dependencies/action.yml', 'config/scripts/ensure-native-runtime.mjs', 'config/scripts/rebuild-native-deps.mjs', 'config/patches/node-pty@1.1.0.patch', 'config/patches/@vscode__windows-process-tree@0.8.0.patch') }}
- name: Prepare Electron native runtime
run: node config/scripts/ensure-native-runtime.mjs --runtime=electron
@@ -813,8 +733,6 @@ jobs:
outputs:
should_run: ${{ steps.filter.outputs.should_run }}
test_files: ${{ steps.filter.outputs.test_files }}
ssh_source_changed: ${{ steps.filter.outputs.ssh_source_changed }}
native_ime_source_changed: ${{ steps.filter.outputs.native_ime_source_changed }}
steps:
- name: Checkout
uses: actions/checkout@v6
@@ -837,16 +755,6 @@ jobs:
# authorities, exclusions, and sentinels without evaluating workflow shell.
TEST_FILES_JSON="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs)"
echo "test_files=$TEST_FILES_JSON" >> "$GITHUB_OUTPUT"
# Why a separate signal: the Docker-SSH lane must trigger on SSH source, not on a
# spec name surviving in a route's list. Same routes, so the two cannot drift.
SSH_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --ssh-source)"
echo "ssh_source_changed=$SSH_SOURCE_CHANGED" >> "$GITHUB_OUTPUT"
echo "SSH source changed: $SSH_SOURCE_CHANGED"
# Why its own signal: the real-IME lane is a whole ibus session, not a spec, so it must
# trigger on IME source rather than on a spec name in some route's list.
NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)"
echo "native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED" >> "$GITHUB_OUTPUT"
echo "Native IME source changed: $NATIVE_IME_SOURCE_CHANGED"
if [ "$TEST_FILES_JSON" != '[]' ]; then
echo "should_run=true" >> "$GITHUB_OUTPUT"
echo "Changed E2E specs: $TEST_FILES_JSON"
@@ -865,22 +773,50 @@ jobs:
uses: ./.github/workflows/e2e.yml
with:
test_files: ${{ needs.e2e-paths.outputs.test_files }}
ssh_source_changed: ${{ needs.e2e-paths.outputs.ssh_source_changed }}
# Why this is not in verify's needs: it is the first PR-gate run of a harness whose reliability
# is only known from nightly main runs (20/20 green, 2026-08-09..2026-08-29, p50 3m25s). It
# reports a red X on the PR without blocking, exactly like `e2e` above. Deliberately no
# continue-on-error: that renders the check green and hides the signal it exists to give. To
# make it blocking, add it to verify.needs, add TERMINAL_IME_NATIVE to the env below, and
# require `success || skipped` outside the strict loop — see the note on `e2e`.
terminal_ime_native:
name: real IME
needs: e2e-paths
if: needs.e2e-paths.outputs.native_ime_source_changed == 'true'
# Why: the reusable workflow only checks out, builds, and uploads artifacts.
permissions:
contents: read
uses: ./.github/workflows/terminal-ime-e2e.yml
# Why: §U10 requires the pure custom-agent resolver/tokenizer/startup/runtime
# suites to pass on every supported desktop OS (macOS/Linux/Windows) so
# platform-dependent behavior — shell quoting on PowerShell/cmd, ~ and WSL/SSH
# path translation, drive-letter mapping — is proven cross-OS, not just on
# Linux. The SSH throwaway-container e2e is a SEPARATE Ubuntu-only job because
# macOS/Windows GitHub runners cannot host a Linux Docker container.
custom-agent-platform:
name: Custom agent platform (${{ matrix.os }})
strategy:
# Run every OS to completion so a platform-specific failure is not masked
# by a fail-fast cancel on another leg.
fail-fast: false
matrix:
os:
- ubuntu-latest
- windows-2022
- macos-15
runs-on: ${{ matrix.os }}
steps:
- name: Checkout
uses: actions/checkout@v6
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
- name: Setup pnpm
uses: pnpm/action-setup@v6
with:
run_install: false
# Why: these suites are pure TypeScript — their resolver/spawn/runtime
# import closure references neither node-pty, better-sqlite3, nor electron
# — so skip postinstall native builds. That keeps the cross-OS matrix fast
# and free of Windows/macOS native-build flakiness while still exercising
# the platform-dependent resolution/quoting logic.
- name: Install dependencies (no native postinstall)
run: pnpm install --no-frozen-lockfile --prefer-frozen-lockfile=false --ignore-scripts
- name: Custom agent platform suites (resolver, tokenizer, startup, runtime)
run: pnpm test:custom-agent-platform
verify:
if: always()
@@ -890,15 +826,14 @@ jobs:
- root_directory_guard
- typecheck
- git_compatibility
- codex_index_heal_contract
- xterm_patch_sync
- shell_contracts
- test
- orcad_browser
- cross-version-wire
- managed_hook_node18
- package
- package_windows
- custom-agent-platform
runs-on: ubuntu-latest
steps:
@@ -915,66 +850,59 @@ jobs:
CODE_PATHS: ${{ needs.code_paths.result }}
SHOULD_RUN: ${{ needs.code_paths.outputs.should_run }}
STATIC_ANALYSIS: ${{ needs.static_analysis.result }}
STATIC_ANALYSIS_SHOULD_RUN: ${{ needs.code_paths.outputs.static_analysis }}
ROOT_DIRECTORY_GUARD: ${{ needs.root_directory_guard.result }}
TYPECHECK: ${{ needs.typecheck.result }}
TYPECHECK_SHOULD_RUN: ${{ needs.code_paths.outputs.typecheck }}
GIT_COMPATIBILITY: ${{ needs.git_compatibility.result }}
GIT_COMPATIBILITY_SHOULD_RUN: ${{ needs.code_paths.outputs.git_compatibility }}
CODEX_INDEX_HEAL_CONTRACT: ${{ needs.codex_index_heal_contract.result }}
CODEX_INDEX_HEAL_CONTRACT_SHOULD_RUN: ${{ needs.code_paths.outputs.codex_index_heal_contract }}
XTERM_PATCH_SYNC: ${{ needs.xterm_patch_sync.result }}
XTERM_PATCH_SYNC_SHOULD_RUN: ${{ needs.code_paths.outputs.xterm_patch_sync }}
SHELL_CONTRACTS: ${{ needs.shell_contracts.result }}
SHELL_CONTRACTS_SHOULD_RUN: ${{ needs.code_paths.outputs.shell_contracts }}
TEST: ${{ needs.test.result }}
TEST_SHOULD_RUN: ${{ needs.code_paths.outputs.test }}
ORCAD_BROWSER: ${{ needs.orcad_browser.result }}
ORCAD_BROWSER_SHOULD_RUN: ${{ needs.code_paths.outputs.orcad_browser }}
CROSS_VERSION_WIRE: ${{ needs.cross-version-wire.result }}
CROSS_VERSION_WIRE_SHOULD_RUN: ${{ needs.code_paths.outputs.cross-version-wire }}
MANAGED_HOOK_NODE18: ${{ needs.managed_hook_node18.result }}
MANAGED_HOOK_NODE18_SHOULD_RUN: ${{ needs.code_paths.outputs.managed_hook_node18 }}
PACKAGE: ${{ needs.package.result }}
PACKAGE_SHOULD_RUN: ${{ needs.code_paths.outputs.package }}
PACKAGE_WINDOWS: ${{ needs.package_windows.result }}
PACKAGE_WINDOWS_SHOULD_RUN: ${{ needs.code_paths.outputs.package_windows }}
CUSTOM_AGENT_PLATFORM: ${{ needs.custom-agent-platform.result }}
run: |
if [ "$CODE_PATHS" != "success" ]; then
exit 1
fi
if [ "$ROOT_DIRECTORY_GUARD" != "success" ]; then
exit 1
fi
if [ "$SHOULD_RUN" != "true" ]; then
echo "Docs-only change; expensive PR checks skipped."
fi
failed=0
check_job() {
local name="$1" result="$2" should="$3"
if [ "$should" = "true" ]; then
if [ "$result" != "success" ]; then
echo "$name: expected success, got $result"
failed=1
fi
else
if [ "$result" != "skipped" ]; then
echo "$name: expected skipped, got $result"
failed=1
fi
if [ "$ROOT_DIRECTORY_GUARD" != "success" ]; then
exit 1
fi
}
for result in \
"$STATIC_ANALYSIS" \
"$TYPECHECK" \
"$GIT_COMPATIBILITY" \
"$XTERM_PATCH_SYNC" \
"$SHELL_CONTRACTS" \
"$TEST" \
"$ORCAD_BROWSER" \
"$MANAGED_HOOK_NODE18" \
"$PACKAGE" \
"$PACKAGE_WINDOWS"; do
if [ "$result" != "skipped" ]; then
exit 1
fi
done
exit 0
fi
# Require success when the PR has code-relevant changes
check_job static_analysis "$STATIC_ANALYSIS" "$STATIC_ANALYSIS_SHOULD_RUN"
check_job typecheck "$TYPECHECK" "$TYPECHECK_SHOULD_RUN"
check_job git_compatibility "$GIT_COMPATIBILITY" "$GIT_COMPATIBILITY_SHOULD_RUN"
check_job codex_index_heal_contract "$CODEX_INDEX_HEAL_CONTRACT" "$CODEX_INDEX_HEAL_CONTRACT_SHOULD_RUN"
check_job xterm_patch_sync "$XTERM_PATCH_SYNC" "$XTERM_PATCH_SYNC_SHOULD_RUN"
check_job shell_contracts "$SHELL_CONTRACTS" "$SHELL_CONTRACTS_SHOULD_RUN"
check_job test "$TEST" "$TEST_SHOULD_RUN"
check_job orcad_browser "$ORCAD_BROWSER" "$ORCAD_BROWSER_SHOULD_RUN"
check_job cross-version-wire "$CROSS_VERSION_WIRE" "$CROSS_VERSION_WIRE_SHOULD_RUN"
check_job managed_hook_node18 "$MANAGED_HOOK_NODE18" "$MANAGED_HOOK_NODE18_SHOULD_RUN"
check_job package "$PACKAGE" "$PACKAGE_SHOULD_RUN"
check_job package_windows "$PACKAGE_WINDOWS" "$PACKAGE_WINDOWS_SHOULD_RUN"
exit "$failed"
for result in \
"$CODE_PATHS" \
"$STATIC_ANALYSIS" \
"$ROOT_DIRECTORY_GUARD" \
"$TYPECHECK" \
"$GIT_COMPATIBILITY" \
"$XTERM_PATCH_SYNC" \
"$SHELL_CONTRACTS" \
"$TEST" \
"$ORCAD_BROWSER" \
"$MANAGED_HOOK_NODE18" \
"$PACKAGE" \
"$PACKAGE_WINDOWS" \
"$CUSTOM_AGENT_PLATFORM"; do
if [ "$result" != "success" ]; then
exit 1
fi
done
@@ -82,5 +82,13 @@
"text": "ghostty",
"dynamic": false,
"count": 1
},
{
"filePath": "src/renderer/src/lib/resume-sleeping-agent-session-test-fixtures.ts",
"kind": "object-property:title",
"text": "shell",
"dynamic": false,
"count": 1,
"reason": "Test-only fixture builder for a terminal-tab shape; 'shell' is fixed test data, not a user-facing string."
}
]
+132
View File
@@ -2,15 +2,147 @@
# This is a RATCHET: the list may only SHRINK. Do NOT add entries to get CI green —
# split the oversized file instead (AGENTS.md → "Do Not Disable Max Lines").
# Regenerate/prune: pnpm check:max-lines-ratchet --prune (removes stale entries only)
inline src/cli/handlers/automations.ts
inline src/cli/help.ts
inline src/main/agent-hooks/server.ts
inline src/main/automations/external-manager.ts
inline src/main/automations/hermes-cron-output.ts
inline src/main/browser/agent-browser-bridge.ts
inline src/main/browser/browser-cookie-import.ts
inline src/main/browser/browser-manager.ts
inline src/main/browser/cdp-bridge.ts
inline src/main/browser/grab-guest-script.ts
inline src/main/claude-accounts/runtime-auth-service.ts
inline src/main/cli/cli-installer.ts
inline src/main/cli/wsl-cli-installer.ts
inline src/main/codex-accounts/runtime-home-service.ts
inline src/main/codex-accounts/service.ts
inline src/main/codex/hook-service.ts
inline src/main/daemon/daemon-init.ts
inline src/main/git/runner.ts
inline src/main/git/status.ts
inline src/main/git/worktree.ts
inline src/main/github/project-view.ts
inline src/main/index.ts
inline src/main/ipc/filesystem-watcher.ts
inline src/main/ipc/filesystem.ts
inline src/main/ipc/repos.ts
inline src/main/ipc/ssh.ts
inline src/main/ipc/worktree-remote.ts
inline src/main/keybindings/keybinding-file.ts
inline src/main/linear/issues.ts
inline src/main/memory/collector.ts
inline src/main/ports/advertised-url-watcher.ts
inline src/main/ports/local-workspace-port-scanner.ts
inline src/main/project-groups/nested-repo-discovery.ts
inline src/main/providers/local-pty-provider.ts
inline src/main/rate-limits/claude-fetcher.ts
inline src/main/rate-limits/claude-pty.ts
inline src/main/rate-limits/codex-fetcher.ts
inline src/main/rate-limits/service.ts
inline src/main/runtime/orca-runtime-browser.ts
inline src/main/runtime/orca-runtime-files.ts
inline src/main/runtime/orca-runtime-git.ts
inline src/main/runtime/orca-runtime.test.ts
inline src/main/runtime/orca-runtime.ts
inline src/main/runtime/rpc/methods/orchestration.ts
inline src/main/runtime/runtime-rpc.ts
inline src/main/source-control/hosted-review-creation.ts
inline src/main/speech/model-manager.ts
inline src/main/speech/stt-service.ts
inline src/main/ssh/ssh-channel-multiplexer.ts
inline src/main/ssh/ssh-connection.ts
inline src/main/ssh/ssh-relay-deploy.ts
inline src/main/ssh/ssh-relay-session.ts
inline src/main/text-generation/commit-message-text-generation.ts
inline src/main/updater.ts
inline src/main/window/attach-main-window-services.ts
inline src/main/window/createMainWindow.ts
inline src/preload/index.ts
inline src/relay/dispatcher.ts
inline src/relay/git-handler.ts
inline src/relay/pty-handler.ts
inline src/relay/workspace-space-scan.ts
inline src/renderer/src/components/GitLabItemDialog.tsx
inline src/renderer/src/components/JiraIssueWorkspace.tsx
inline src/renderer/src/components/LinearItemDrawer.tsx
inline src/renderer/src/components/TaskPage.tsx
inline src/renderer/src/components/Terminal.tsx
inline src/renderer/src/components/WorktreeJumpPalette.tsx
inline src/renderer/src/components/activity/ActivityPrototypePage.tsx
inline src/renderer/src/components/automations/AutomationsPage.tsx
inline src/renderer/src/components/editor/CombinedDiffViewer.tsx
inline src/renderer/src/components/editor/EditorContent.tsx
inline src/renderer/src/components/editor/IpynbViewer.tsx
inline src/renderer/src/components/editor/MarkdownPreview.tsx
inline src/renderer/src/components/feature-wall/EditorAnimatedVisual.tsx
inline src/renderer/src/components/floating-terminal/FloatingTerminalPanel.tsx
inline src/renderer/src/components/new-workspace/SmartWorkspaceNameField.tsx
inline src/renderer/src/components/onboarding/use-onboarding-flow.ts
inline src/renderer/src/components/right-sidebar/ChecksPanel.tsx
inline src/renderer/src/components/right-sidebar/PortsPanel.tsx
inline src/renderer/src/components/right-sidebar/checks-panel-content.tsx
inline src/renderer/src/components/settings/AccountsPane.tsx
inline src/renderer/src/components/settings/RepositoryHooksSection.tsx
inline src/renderer/src/components/settings/RuntimeEnvironmentsPane.tsx
inline src/renderer/src/components/settings/Settings.tsx
inline src/renderer/src/components/sidebar/use-workspace-kanban-area-selection.ts
inline src/renderer/src/components/status-bar/ResourceUsageStatusSegment.tsx
inline src/renderer/src/components/status-bar/StatusBar.tsx
inline src/renderer/src/components/status-bar/WorkspaceSpaceManagerPanel.tsx
inline src/renderer/src/components/terminal-pane/TerminalPane.tsx
inline src/renderer/src/components/terminal-pane/agent-completion-coordinator.ts
inline src/renderer/src/components/terminal-pane/keyboard-handlers.ts
inline src/renderer/src/components/terminal-pane/pty-transport.ts
inline src/renderer/src/components/terminal-pane/remote-runtime-pty-transport.ts
inline src/renderer/src/components/terminal-pane/use-terminal-pane-lifecycle.ts
inline src/renderer/src/hooks/useAutomationDispatchEvents.ts
inline src/renderer/src/hooks/useEditorExternalWatch.ts
inline src/renderer/src/hooks/useSettingsNavigationMetadata.ts
inline src/renderer/src/lib/pane-manager/pane-terminal-output-scheduler.ts
inline src/renderer/src/lib/pane-manager/pane-tree-ops.ts
inline src/renderer/src/runtime/remote-runtime-terminal-multiplexer.ts
inline src/renderer/src/runtime/runtime-file-client.ts
inline src/renderer/src/runtime/runtime-git-client.ts
inline src/renderer/src/runtime/runtime-linear-client.ts
inline src/renderer/src/runtime/sync-runtime-graph.ts
inline src/renderer/src/runtime/web-runtime-session.ts
inline src/renderer/src/runtime/web-session-tabs-sync.ts
inline src/renderer/src/store/slices/agent-status.ts
inline src/renderer/src/store/slices/browser.ts
inline src/renderer/src/store/slices/diffComments.ts
inline src/renderer/src/store/slices/hosted-review.ts
inline src/renderer/src/store/slices/linear.ts
inline src/renderer/src/store/slices/repos.ts
inline src/renderer/src/store/slices/tabs.ts
inline src/renderer/src/store/slices/ui.ts
inline src/renderer/src/web/web-runtime-client.ts
inline src/shared/automation-schedules.ts
inline src/shared/commit-message-agent-spec.ts
inline src/shared/constants.ts
inline src/shared/keybindings.ts
inline src/shared/marine-creatures.ts
inline src/shared/remote-runtime-client.ts
inline src/shared/runtime-types.ts
inline src/shared/source-control-ai.ts
inline src/shared/telemetry-events.ts
inline src/shared/text-search.ts
inline tests/e2e/helpers/terminal.ts
mobile-config app/h/*/files/*.tsx
mobile-config app/h/*/index.tsx
mobile-config app/h/*/session/*.tsx
mobile-config app/h/*/source-control/*.tsx
mobile-config app/h/*/tasks.tsx
mobile-config app/index.tsx
mobile-config app/pair-scan.tsx
mobile-config app/troubleshoot.tsx
mobile-config scripts/mock-server.ts
mobile-config scripts/repro-terminal-colors.ts
mobile-config scripts/repro-worktree-startup-stream.ts
mobile-config src/browser/MobileBrowserPane.tsx
mobile-config src/components/CustomKeyModal.tsx
mobile-config src/components/NewWorktreeModal.tsx
mobile-config src/components/mobile-rich-markdown-editor-html.ts
mobile-config src/terminal/terminal-accessory-keys.ts
mobile-config src/terminal/terminal-webview-html.ts
mobile-config src/transport/rpc-client.ts
+154 -1
View File
@@ -7160,7 +7160,7 @@
"providers": ["local", "daemon", "ssh", "wsl", "remote-runtime"],
"coveredPlatforms": ["macos"],
"coveredProviders": ["local", "daemon", "remote-runtime"],
"coverageNotes": "Renderer ownership/dedupe contracts cover provider-session claims. Local and daemon attach-only contracts prove an existing stable-pane owner is adopted without provider creation, while remote-runtime transport contracts preserve adopted ownership through cancellation. The Electron oracle covers a local macOS runtime and daemon with real agent, Setup, and unrelated-canary processes; SSH, WSL, paired-server, Linux, and Windows remain contract-only or unrun.",
"coverageNotes": "Renderer ownership/dedupe contracts cover provider-session claims, extended in U5 to the host-side private record store authoritative for base+provider-session ownership. Local and daemon attach-only contracts prove an existing stable-pane owner is adopted without provider creation, while remote-runtime transport contracts preserve adopted ownership through cancellation. The Electron oracle covers a local macOS runtime and daemon with real agent, Setup, and unrelated-canary processes; SSH, WSL, paired-server, Linux, and Windows remain contract-only or unrun. Maturity stays experimental.",
"motivatingLinks": [
"https://github.com/stablyai/orca/pull/6800",
"https://github.com/stablyai/orca/pull/5240",
@@ -7175,6 +7175,8 @@
"pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/lib/resume-sleeping-agent-session.test.ts",
"pnpm exec vitest run --config config/vitest.config.ts tests/e2e/completed-worker-retirement-resume.unit.test.ts --reporter=verbose",
"pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/lib/resume-sleeping-agent-session.test.ts src/main/providers/local-pty-provider-spawn-session.test.ts src/main/daemon/terminal-host.test.ts src/main/daemon/daemon-pty-adapter.test.ts src/main/ipc/pty-pane-materialization-race.test.ts src/main/ipc/pty-persisted-incarnation-repair.test.ts src/main/ipc/pty-pane-reservation-settlement.test.ts src/main/runtime/orca-runtime.test.ts src/renderer/src/lib/pane-manager/pane-fit.test.ts src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-host-session-launch.test.ts",
"pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/lib/resume-sleeping-agent-session.test.ts src/renderer/src/lib/resume-sleeping-agent-session-base-ownership.test.ts",
"pnpm exec vitest run --config config/vitest.config.ts src/main/agent-launch/agent-session-record-store.test.ts",
"pnpm exec electron-vite build --mode e2e && SKIP_BUILD=1 pnpm exec playwright test tests/e2e/live-background-terminal-mount-authority.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1"
],
"testFiles": [
@@ -7190,6 +7192,8 @@
"src/renderer/src/lib/pane-manager/pane-fit.test.ts",
"src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts",
"src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-host-session-launch.test.ts",
"src/renderer/src/lib/resume-sleeping-agent-session-base-ownership.test.ts",
"src/main/agent-launch/agent-session-record-store.test.ts",
"tests/e2e/live-background-terminal-mount-authority.spec.ts"
],
"assertionRefs": [
@@ -7254,6 +7258,22 @@
"runtime inventory, renderer graph, persisted session, runtime epoch, and daemon PID converge without replacement or resume",
"the unrelated canary remains writable and receives no signal across target projection repair"
]
},
{
"file": "src/renderer/src/lib/resume-sleeping-agent-session-base-ownership.test.ts",
"assertions": [
"a live custom-id pane owns its base provider session so a second custom id on the same base does not re-resume",
"a live custom-id pane on a different resumable base cannot claim another base's session id"
]
},
{
"file": "src/main/agent-launch/agent-session-record-store.test.ts",
"assertions": [
"two custom ids on one base/provider session resolve to one owner record",
"preserves the requested custom identity while keying ownership on the base",
"never binds a non-resumable base",
"a fork binds a NEW provider session into its own record and never mutates the source"
]
}
],
"evidenceRuns": [
@@ -7265,6 +7285,24 @@
"result": "passed",
"durationSeconds": 1.9,
"summary": "1 test file(s) passed, 30 tests passed on main@1282f5c2d in a clean checkout."
},
{
"date": "2026-07-11",
"runner": "local",
"platform": "macos",
"command": "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/lib/resume-sleeping-agent-session.test.ts src/renderer/src/lib/resume-sleeping-agent-session-base-ownership.test.ts",
"result": "passed",
"durationSeconds": 1.3,
"summary": "2 test file(s), 32 tests passed on branch custom-agents (U5). Adds base-keyed ownership evidence: two custom ids on one base+provider-session collapse to one owner, and a different resumable base cannot claim another base's session id."
},
{
"date": "2026-07-11",
"runner": "local",
"platform": "macos",
"command": "pnpm exec vitest run --config config/vitest.config.ts src/main/agent-launch/agent-session-record-store.test.ts",
"result": "passed",
"durationSeconds": 0.2,
"summary": "1 test file, 17 tests passed on branch custom-agents (U5). Extends the gate to the HOST-side private record store now authoritative for base+provider-session ownership: two custom ids collapse to one owner record, ownership keys on the base while preserving the requested custom identity, a non-resumable base never binds, and a fork binds a new provider session into its own record without mutating the source. Evidence extended without promoting experimental maturity."
}
],
"runtimeBudget": {
@@ -8010,6 +8048,121 @@
],
"demotionRule": "Demote or block release if any non-Advanced path mutates Active Server, if local reveal or Add Project depends on the durable default instead of captured workspace/host ownership, if an Add Project operation reaches a different host after it begins, or if transient host state survives restart."
},
{
"id": "agent-launch.resolution-attribution",
"title": "One host resolution yields one immutable snapshot, one launch token, and at most one PTY",
"maturity": "experimental",
"protection": "partial",
"owner": "agent-session",
"layer": "host-launch-resolution",
"surfaces": [
"agent launch resolution",
"worktree and composer create",
"default agent and tab quick launch",
"terminal quick commands and floating terminals",
"background, automation, and orchestration launches",
"source control AI recipe launch",
"activation, restore, resume, and fork",
"mobile and paired web launch"
],
"platforms": ["macos", "linux", "windows"],
"providers": ["local", "daemon", "ssh", "wsl", "remote-runtime"],
"coveredPlatforms": ["macos"],
"coveredProviders": [],
"coverageNotes": "Local macOS evidence over the host resolution/attribution suites on branch custom-agents (U10): the pure resolver (resolve-agent-launch.performance.test.ts and resolve-agent-launch-assembly.test.ts), the spawn boundary (agent-launch-spawn.test.ts), and the runtime terminal-create path (terminal-agent-launch-resolution.test.ts). The unit/runtime suites carry the full platform/provider risk matrix by construction because the pure resolver takes target platform, shell, and isRemote as inputs; live cells are named representatives elsewhere in U10 and are NOT folded into this gate's covered scope: the local Electron custom-agent e2e (renderer half) and the throwaway-container SSH custom-agent e2e observe the fixture argv/env at the target and the single PTY, and the three-OS platform matrix runs the resolver and tokenizer suites on ubuntu-latest, windows-2022, and macos-15. Maturity stays experimental until soak and saved red/green land. Base-scoped custom identity ownership is registered on the separate agent-session.provider-ownership gate (extended in U5); the two are intentionally not merged because ownership/dedupe and launch resolution have different failure oracles and demotion rules.",
"motivatingLinks": [
"docs/plans/2026-07-09-001-feat-custom-agents-plan.md",
"docs/plans/2026-07-09-001-feat-custom-agents-plan.md#registered-gate-contract"
],
"invariant": "One host resolution produces one immutable structured snapshot, one launch token, and at most one PTY; recovery never guesses liveness or identity. The host resolves command, args, env, and base from its own authenticated catalog/settings state, never from a client-supplied command or env.",
"oracle": "Host-unit and runtime suites assert the resolved argv/env band order and shell-safe quoting derived from one fetched catalog entry, one deep-frozen snapshot per resolution that excludes prompt/process/env-control data, O(1) lookup whose cost and env size add no catalog scans, catalog indexing once per settings revision, source-record and default/live reference attribution, a client receipt that never carries a custom env key or value, an untrusted client that cannot escalate to an unattended intent, and typed pre-spawn failures that produce no plan. Live Electron and SSH representatives observe the exact fixture argv/env at the target and the single PTY/token; those live cells are the risk scope, not proof that every platform/provider has a live gate.",
"commands": [
"pnpm exec vitest run --config config/vitest.config.ts src/main/agent-launch/resolve-agent-launch.performance.test.ts src/main/agent-launch/resolve-agent-launch-assembly.test.ts src/main/agent-launch/agent-launch-spawn.test.ts src/main/runtime/terminal-agent-launch-resolution.test.ts"
],
"testFiles": [
"src/main/agent-launch/resolve-agent-launch.performance.test.ts",
"src/main/agent-launch/resolve-agent-launch-assembly.test.ts",
"src/main/agent-launch/agent-launch-spawn.test.ts",
"src/main/runtime/terminal-agent-launch-resolution.test.ts"
],
"assertionRefs": [
{
"file": "src/main/agent-launch/resolve-agent-launch.performance.test.ts",
"assertions": [
"resolves a custom launch in O(1): lookup cost does not grow with catalog size",
"derives every field from one fetched entry: env size adds no catalog lookups",
"indexes the catalog once per revision and reuses it across resolutions",
"records resolver throughput as PR evidence (never a CI wall-clock assertion)"
]
},
{
"file": "src/main/agent-launch/resolve-agent-launch-assembly.test.ts",
"assertions": [
"appends recipe args as a distinct band after the definition argv",
"deep-freezes the snapshot and excludes prompt/process/env-control data",
"stays identical across worktree paths while the admission fingerprint changes",
"fails base_agent_unavailable for stock argv when the base is not detected"
]
},
{
"file": "src/main/agent-launch/agent-launch-spawn.test.ts",
"assertions": [
"resolves the command from host state, never a client-supplied command/env",
"resolves a source-control recipe id to its stored agentArgs as perLaunchArgs (U7)",
"rejects an unknown recipe action id with untrusted_reference and never resolves (U7)",
"propagates a typed resolution failure without a plan"
]
},
{
"file": "src/main/runtime/terminal-agent-launch-resolution.test.ts",
"assertions": [
"maps a resolved launch to terminal fields with the settle token + receipt",
"never carries a custom env key/value in the client receipt (G7 oracle-12/13)",
"never lets an untrusted client escalate to an unattended intent (GP3)",
"fails invalid_launch_snapshot for an unknown session key without resolving"
]
}
],
"evidenceRuns": [
{
"date": "2026-07-12",
"runner": "local",
"platform": "macos",
"command": "pnpm exec vitest run --config config/vitest.config.ts src/main/agent-launch/resolve-agent-launch.performance.test.ts src/main/agent-launch/resolve-agent-launch-assembly.test.ts src/main/agent-launch/agent-launch-spawn.test.ts src/main/runtime/terminal-agent-launch-resolution.test.ts",
"result": "passed",
"durationSeconds": 1.4,
"summary": "4 test files, 77 tests passed on branch custom-agents (U10) in a clean checkout. Covers O(1) size-independent resolution and index-once-per-revision (algorithmic assertions; wall-clock recorded as PR evidence only), argv band order and shell-safe quoting with the deep-frozen snapshot, host-state command resolution with source-control recipe attribution and untrusted_reference rejection, and the runtime terminal-create mapping with the no-custom-env-in-receipt and no-unattended-escalation leak guards."
}
],
"runtimeBudget": {
"p95Seconds": 10,
"scope": "host-unit and runtime resolution suites"
},
"flakeHistory": {
"status": "unknown",
"evidence": "Registered at U10 from the resolution/attribution suites; the resolver is a pure function with no timers, sockets, or subprocesses, but the gate needs soak history before it can block promotion."
},
"redGreenEvidence": {
"status": "partial",
"evidence": "Suites encode the one-snapshot/one-token/O(1)-lookup invariant, the argv band order and shell-quoting failure sources, source-record attribution and untrusted_reference rejection, and the client-receipt env-leak and unattended-escalation guards. Needs a saved red/green artifact for the class-level 'one resolution yields at most one PTY and recovery never guesses identity' invariant."
},
"performanceBudget": {
"required": true,
"evidence": "resolve-agent-launch.performance.test.ts asserts algorithmic/count invariants only (O(1) catalog lookups independent of a 2-vs-1000-agent catalog, no lookups added by 64 env entries, index-once-per-revision) per the plan's CI-enforces-count-not-wall-clock rule; p95 resolver throughput (evidence run about 0.06 ms versus the 2 ms budget) is recorded as PR-machine evidence, never a CI threshold. PRs adding new resolution scans must show bounded work before blocking promotion."
},
"promotionCriteria": [
"Run in soak for at least 100 consecutive passes or 14 days across the required CI platforms.",
"Attach live Electron and throwaway-container SSH evidence that the fixture argv/env at the target and the single PTY/token match the resolved snapshot.",
"Attach a saved red/green artifact proving a missing resolved snapshot fails closed with invalid_launch_snapshot rather than a fallback shell."
],
"knownGaps": [
"Desktop-launches-remote stock base-agent detection is unavailable (pty.ts passes detectionUnavailable; on-demand remote detection never landed), so a desktop-originated remote launch takes the honest-unknown path and skips the stock-detection gate rather than silently guessing; accepted G10 known-gap (KG-1).",
"PTY registration cannot prove the child CLI reached a healthy TUI; this residual is named in the plan and stays a permanent known-gap (KG-2).",
"Platform/provider cells without a live representative (for example WSL and non-Ubuntu SSH, which no GitHub runner can host) remain explicit known-gap rows and are not claimed as live coverage; unit and runtime suites carry them by construction.",
"Mobile custom-agent behavior is proven at unit; there is no native mobile e2e harness in the repo, and the desktop embedded-simulator spec is a distinct surface, so the live mobile-emulator cell stays an accepted known-gap (KG-4)."
],
"demotionRule": "Demote or quarantine if the gate flakes without a product bug, if a spawn is ever admitted without a resolved snapshot, if a client-supplied command or env is ever honored, or if one resolution yields more than one launch token or PTY."
},
{
"id": "terminal-geometry.visible-convergence",
"title": "Visible desktop terminals converge across xterm, fit, PTY, shell, and runtime mirror size",
@@ -0,0 +1,39 @@
import { spawnSync } from 'node:child_process'
const extraArgs = process.argv.slice(2)
const pnpm = process.platform === 'win32' ? 'pnpm.cmd' : 'pnpm'
const env = {
...process.env,
ORCA_E2E_SSH_DOCKER: '1'
}
const runtime = spawnSync(pnpm, ['run', 'ensure:electron-runtime'], {
stdio: 'inherit',
env
})
if (runtime.status !== 0) {
process.exit(runtime.status ?? 1)
}
const result = spawnSync(
pnpm,
[
'exec',
'playwright',
'test',
'tests/e2e/ssh-custom-agent-launch.spec.ts',
'--config',
'tests/playwright.config.ts',
'--project',
'electron-headless',
'--workers=1',
...extraArgs
],
{
stdio: 'inherit',
env
}
)
process.exit(result.status ?? 1)
@@ -0,0 +1,63 @@
# Custom agents: forward-rollback runbook
Operational contract for rolling back a release that ships the agent-catalog **v1**
schema (`agentCatalogSchemaVersion: 1`). Grounded in plan §1021-1028 (Rollout and
rollback contract) and acceptance oracle 39.
## Why a plain downgrade is unsafe
The v1 data file is **forward-readable but not honestly writable** by an older Orca
binary: pre-v1 code does not understand custom identities in defaults, owner
records, or resume attribution. Runtime compatibility projections protect old
*clients*; they do not make a downgraded desktop binary schema-aware.
## Rules
1. **Production rollback = a forward-rollback build.** It keeps the v1
reader/resolver and may disable new authoring. It must never disable identity
resolution for already-saved defaults, automations, or sessions.
2. **Never install a pre-v1 binary over a live v1 data file.** Release automation
must not resolve a feature incident this way.
3. **Explicit user downgrade = restore the pinned backup** (below). This is the
only supported pre-v1 binary rollback and discards settings/workspace metadata
written after the backup point.
## The pinned pre-v1 backup
- Written once, before the first v1 write, beside the rotating backups at
`<data-file>.pre-agent-catalog-v1.backup` (`pinnedPreV1BackupPath`).
- Same filesystem permissions as the data file; `fsync` + atomic rename.
- If it cannot be created, migration performs **no v1 write** and Settings reports a
local migration error; launch behavior stays on the clean built-in baseline.
- Never synced, never exposed through RPC, never given weaker permissions.
- Removed only after the documented one-release rollback window (see follow-up).
## Crash safety (oracle 39)
Schema migration, backup creation, catalog/reference mutation, and snapshot
persistence are independently fault-injected. Any crash leaves **either** the
complete old file + usable backup **or** the complete v1 file + usable backup —
never a half-migrated file. The backup is created before any v1 write, so a crash
mid-migration always finds an intact v0 file to restart from.
## Explicit downgrade procedure
1. Quit Orca.
2. Copy `<data-file>.pre-agent-catalog-v1.backup` over `<data-file>`.
3. Install the pre-v1 binary and relaunch.
The restored file is byte-identical to the pre-v1 state; all post-backup metadata
is intentionally discarded.
## Verification
Rollback verification runs on a **disposable** profile, never on user data:
`src/main/agent-launch/agent-catalog-forward-rollback-fixture.test.ts` exercises
migrate → v1-reference resolve → forward-rollback resolve → pinned-backup restore →
crash-before-write restart. Unit field-mapping coverage lives in
`agent-catalog-schema-migration.test.ts`.
## Follow-up (fill before merge)
- Dated removal of the pinned backup and the end of the one-release rollback
window: **TODO(date)** — owner to set at release cut.
+2
View File
@@ -27,6 +27,7 @@
"test": "node config/scripts/ensure-native-runtime.mjs --runtime=node && vitest run --config config/vitest.config.ts",
"test:skill-sharing:release": "vitest run --config config/vitest.config.ts src/main/skills src/main/runtime/rpc/methods/skills.test.ts src/relay/skill-install-handler.test.ts src/shared/skill-bundle-install-contract.test.ts src/shared/skill-install-contract.test.ts src/shared/skill-install-failure.test.ts src/shared/skill-package-manifest.test.ts",
"test:repro:remote-agent-session": "pnpm run build:cli && pnpm run build:electron-vite && node config/scripts/remote-agent-session-authority-repro.mjs",
"test:custom-agent-platform": "vitest run --config config/vitest.config.ts src/main/agent-launch/resolve-agent-launch-assembly.test.ts src/main/agent-launch/resolve-agent-launch-snapshot-target.test.ts src/main/agent-launch/resolved-agent-startup-plan.test.ts src/main/agent-launch/compose-agent-launch-env.test.ts src/main/agent-launch/agent-launch-spawn.test.ts src/main/agent-launch/resolve-agent-launch.performance.test.ts src/shared/tui-agent-startup.test.ts src/shared/legacy-agent-prefix-tokenizer.test.ts src/main/runtime/terminal-agent-launch-resolution.test.ts",
"check:reliability-gates": "node config/scripts/check-reliability-gates.mjs",
"check:max-lines-ratchet": "node config/scripts/check-max-lines-ratchet.mjs",
"check:ts-nocheck-ratchet": "node config/scripts/check-ts-nocheck-ratchet.mjs",
@@ -119,6 +120,7 @@
"test:e2e:local-ssh-browser": "node config/scripts/run-local-ssh-browser-routing-e2e.mjs",
"test:e2e:ssh-docker-terminal-parking": "node config/scripts/run-ssh-docker-terminal-parking-e2e.mjs",
"test:e2e:nested-runtime-ssh": "node config/scripts/run-nested-runtime-ssh-e2e.mjs",
"test:e2e:ssh-custom-agent": "node config/scripts/run-ssh-docker-custom-agent-e2e.mjs",
"test:e2e:source-control-scale": "pnpm run ensure:electron-runtime && npx playwright test tests/e2e/source-control-large-file-count.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1",
"win-update-e2e": "node tests/tools/win-update-e2e/run.mjs",
"win-crash-survival-e2e": "node tests/tools/win-crash-survival-e2e/run.mjs",
+234
View File
@@ -0,0 +1,234 @@
import path from 'node:path'
import type { CustomTuiAgent, CustomTuiAgentId } from '../../src/shared/types'
import { test, expect } from './helpers/orca-app'
import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store'
import {
countVisibleTerminalPanes,
sendToTerminal,
waitForActivePanePtyId,
waitForActiveTerminalManager,
waitForPaneCount,
waitForTerminalOutput
} from './helpers/terminal'
// The custom agent's executable is `node <fixture>`: commandOverride is one argv
// element (the node binary running the test), and the fixture path rides the v1
// args template (space-free, so it tokenizes to a single argument).
const FIXTURE_PATH = path.join(process.cwd(), 'tests/e2e/fixtures/custom-agent-launch-fixture.cjs')
const CUSTOM_AGENT_ID =
'custom-agent:codex:0f1e2d3c-4b5a-4c6d-8e7f-9a0b1c2d3e4f' as CustomTuiAgentId
const CUSTOM_AGENT_LABEL = 'E2E Fixture Agent'
const READY_MARKER = 'CUSTOM_AGENT_FIXTURE_READY'
const CUSTOM_AGENT: CustomTuiAgent = {
id: CUSTOM_AGENT_ID,
baseAgent: 'codex',
label: CUSTOM_AGENT_LABEL,
commandOverride: process.execPath,
args: FIXTURE_PATH,
env: {},
syncEnv: false
}
// A well-formed custom id that is deliberately NEVER seeded: the host resolves it
// against its catalog, finds nothing, and rejects the launch with a client-safe
// unknown_agent failure — a deterministic launch-boundary rejection that needs no
// second (fragile) seed. The uuid segment is asserted absent from the notice.
const UNKNOWN_AGENT_ID =
'custom-agent:codex:1a2b3c4d-5e6f-4a7b-8c9d-0e1f2a3b4c5d' as CustomTuiAgentId
// A real source-control launch-action id seeded with stored agentArgs. Naming it
// via sourceRecord makes the host resolve those args into host-owned perLaunchArgs
// (the client never sends args), which append to the launched agent's argv.
const RECIPE_ACTION_ID = 'fixChecks'
const RECIPE_ARGS = '--recipe one'
test.use({
seededCustomAgents: { agents: [CUSTOM_AGENT] },
seededSourceControlActions: { [RECIPE_ACTION_ID]: { agentArgs: RECIPE_ARGS } }
})
test('launches a seeded custom agent through the host boundary and round-trips keyboard input', async ({
orcaPage
}) => {
await waitForSessionReady(orcaPage)
const worktreeId = await waitForActiveWorktree(orcaPage)
// Self-verify the seed loaded into the real catalog before launching, so
// schema drift fails loudly here instead of silently no-op'ing the launch.
const loadedAgent = await orcaPage.evaluate(async (agentId) => {
const snapshot = await window.api.settings.agentCatalog.getLocal()
const entry = snapshot.customAgents.find(
(candidate) => candidate.status === 'ready' && candidate.definition.id === agentId
)
return entry && entry.status === 'ready'
? { id: entry.definition.id, label: entry.definition.label }
: null
}, CUSTOM_AGENT_ID)
expect(loadedAgent).toEqual({ id: CUSTOM_AGENT_ID, label: CUSTOM_AGENT_LABEL })
// Launch the custom agent exactly as launchAgentInNewTab does: an empty
// command with an `agentLaunch` selection the host resolves at spawn (it owns
// command/arg assembly), driving the seeded commandOverride into a real PTY.
const launchedTabId = await orcaPage.evaluate(
({ worktreeId, agentId }) => {
const store = window.__store
if (!store) {
throw new Error('Store unavailable')
}
const state = store.getState()
const tab = state.createTab(worktreeId, undefined, undefined, { launchAgent: agentId })
state.queueTabStartupCommand(tab.id, {
command: '',
agentLaunch: {
selection: { kind: 'agent', agent: agentId },
allowEmptyPromptLaunch: true
},
telemetry: { launch_source: 'tab_bar_quick_launch', request_kind: 'new' }
})
state.setActiveTab(tab.id)
state.setActiveTabType('terminal')
return tab.id
},
{ worktreeId, agentId: CUSTOM_AGENT_ID }
)
await ensureTerminalVisible(orcaPage)
await waitForActiveTerminalManager(orcaPage, 30_000)
await waitForPaneCount(orcaPage, 1, 30_000)
const ptyId = await waitForActivePanePtyId(orcaPage, 30_000)
// Exact launch output: the fixture prints its readiness marker on spawn, proving
// the host resolved the custom executable and started it in the pane's PTY.
await waitForTerminalOutput(orcaPage, READY_MARKER, 30_000)
// Trusted keyboard input: typing into the PTY reaches the process, which echoes
// it back with a prefix the terminal itself never produces (raw mode, no local echo).
await sendToTerminal(orcaPage, ptyId, 'ping\r')
await waitForTerminalOutput(orcaPage, 'CUSTOM_AGENT_ECHO:ping', 15_000)
// Exactly one PTY/session identity: the launch produced a single pane, and the
// launched tab carries the custom agent's identity as its visible launch state.
expect(await countVisibleTerminalPanes(orcaPage)).toBe(1)
const tabIdentity = await orcaPage.evaluate(
({ worktreeId, tabId }) => {
const tab = (window.__store?.getState().tabsByWorktree[worktreeId] ?? []).find(
(candidate) => candidate.id === tabId
)
return tab ? { launchAgent: tab.launchAgent, ptyId: tab.ptyId } : null
},
{ worktreeId, tabId: launchedTabId }
)
expect(tabIdentity?.launchAgent).toBe(CUSTOM_AGENT_ID)
expect(typeof tabIdentity?.ptyId).toBe('string')
// Clean shutdown so the daemon/PTY teardown does not race the app close.
await sendToTerminal(orcaPage, ptyId, '\x03')
})
test('surfaces a client-safe recovery notice when a custom agent fails to launch', async ({
orcaPage
}) => {
await waitForSessionReady(orcaPage)
const worktreeId = await waitForActiveWorktree(orcaPage)
// Confirm the id is genuinely absent from the catalog, so the failure below is
// a live launch-boundary rejection (unknown_agent), not a seeding artifact.
const isSeeded = await orcaPage.evaluate(async (agentId) => {
const snapshot = await window.api.settings.agentCatalog.getLocal()
return snapshot.customAgents.some(
(candidate) => candidate.status === 'ready' && candidate.definition.id === agentId
)
}, UNKNOWN_AGENT_ID)
expect(isSeeded).toBe(false)
// Launch through the SAME production boundary as the happy path; the host
// resolves the unknown id, rejects it pre-spawn, and creates no PTY.
await orcaPage.evaluate(
({ worktreeId, agentId }) => {
const store = window.__store
if (!store) {
throw new Error('Store unavailable')
}
const state = store.getState()
const tab = state.createTab(worktreeId, undefined, undefined, { launchAgent: agentId })
state.queueTabStartupCommand(tab.id, {
command: '',
agentLaunch: {
selection: { kind: 'agent', agent: agentId },
allowEmptyPromptLaunch: true
},
telemetry: { launch_source: 'tab_bar_quick_launch', request_kind: 'new' }
})
state.setActiveTab(tab.id)
state.setActiveTabType('terminal')
},
{ worktreeId, agentId: UNKNOWN_AGENT_ID }
)
await ensureTerminalVisible(orcaPage)
await waitForActiveTerminalManager(orcaPage, 30_000)
// The pre-spawn failure renders the persistent in-pane recovery notice with
// localized, client-safe copy — it names no argv, env, or agent id.
await expect(orcaPage.getByText(/no longer exists/i)).toBeVisible({ timeout: 30_000 })
// Client-safe: the visible notice never leaks the requested agent id.
expect(await orcaPage.getByText('1a2b3c4d').count()).toBe(0)
})
test('threads a source-control recipe’s stored agentArgs into the launched agent argv', async ({
orcaPage
}) => {
await waitForSessionReady(orcaPage)
const worktreeId = await waitForActiveWorktree(orcaPage)
// Self-verify the seed loaded so a drifted catalog fails loudly here instead of
// silently launching without the custom executable the recipe args ride on.
const ready = await orcaPage.evaluate(async (agentId) => {
const snapshot = await window.api.settings.agentCatalog.getLocal()
return snapshot.customAgents.some(
(candidate) => candidate.status === 'ready' && candidate.definition.id === agentId
)
}, CUSTOM_AGENT_ID)
expect(ready).toBe(true)
// Launch through the same production boundary as the happy path, but name the
// seeded recipe via sourceRecord. The client sends only the owner locator; the
// host resolves the recipe's stored agentArgs into perLaunchArgs at spawn.
await orcaPage.evaluate(
({ worktreeId, agentId, actionId }) => {
const store = window.__store
if (!store) {
throw new Error('Store unavailable')
}
const state = store.getState()
const tab = state.createTab(worktreeId, undefined, undefined, { launchAgent: agentId })
state.queueTabStartupCommand(tab.id, {
command: '',
agentLaunch: {
selection: { kind: 'agent', agent: agentId },
allowEmptyPromptLaunch: true,
sourceRecord: { owner: 'source-control-recipe', id: actionId }
},
telemetry: { launch_source: 'tab_bar_quick_launch', request_kind: 'new' }
})
state.setActiveTab(tab.id)
state.setActiveTabType('terminal')
},
{ worktreeId, agentId: CUSTOM_AGENT_ID, actionId: RECIPE_ACTION_ID }
)
await ensureTerminalVisible(orcaPage)
await waitForActiveTerminalManager(orcaPage, 30_000)
const ptyId = await waitForActivePanePtyId(orcaPage, 30_000)
// The recipe's agentArgs reached THIS spawned process's argv — proving the
// host-owned recipe→perLaunchArgs→argv path lands through the real boundary,
// not just the resolver (host-unit owns the resolver-level assertion).
await waitForTerminalOutput(orcaPage, `CUSTOM_AGENT_ARGV:${RECIPE_ARGS}`, 30_000)
// Clean shutdown so the daemon/PTY teardown does not race the app close.
await sendToTerminal(orcaPage, ptyId, '\x03')
})
@@ -0,0 +1,45 @@
#!/usr/bin/env node
'use strict'
// Deterministic stand-in for a real agent CLI, launched through the host
// agentLaunch boundary via a seeded custom agent's commandOverride. It prints a
// fixed readiness marker so the exact launch output is assertable, and echoes
// each received input line back with a distinct prefix so a test can prove that
// trusted keyboard input reached THIS process rather than local terminal echo.
const READY_MARKER = 'CUSTOM_AGENT_FIXTURE_READY'
const ECHO_PREFIX = 'CUSTOM_AGENT_ECHO:'
const ARGV_PREFIX = 'CUSTOM_AGENT_ARGV:'
function write(text) {
process.stdout.write(text)
}
write(`${READY_MARKER}\r\n`)
// Echo the received extra argv (everything past `node <fixture>`) so a test can
// prove a host-resolved recipe's per-launch args reached THIS process's argv.
write(`${ARGV_PREFIX}${process.argv.slice(2).join(' ')}\r\n`)
process.stdin.setEncoding('utf8')
if (process.stdin.isTTY) {
// Raw mode disables the slave's line-discipline echo, so the only occurrence
// of the typed text is the transformed echo line this process emits.
process.stdin.setRawMode(true)
}
process.stdin.resume()
let line = ''
process.stdin.on('data', (chunk) => {
for (const char of chunk) {
if (char === '\x03') {
// Ctrl-C: exit cleanly so the test can tear the session down.
process.exit(0)
}
if (char === '\r' || char === '\n') {
write(`${ECHO_PREFIX}${line}\r\n`)
line = ''
continue
}
line += char
}
})
@@ -0,0 +1,175 @@
import { execFileSync } from 'node:child_process'
import type { Page } from '@stablyai/playwright-test'
import {
DOCKER_SSH_RELAY_REMOTE_REPO_PATH,
type DockerSshRelayTarget
} from './docker-ssh-relay-target'
// Container-side fixture location. Space-free so the seeded custom agent's v1
// args template tokenizes it to a single argument (mirrors the local spec).
export const REMOTE_CUSTOM_AGENT_FIXTURE_PATH = '/tmp/orca-custom-agent-fixture.cjs'
export type ConnectedRemoteWorktree = {
targetId: string
repoId: string
worktreeId: string
}
export type RemoteAgentProcess = {
pid: number
argv: string[]
env: Map<string, string>
}
function dockerRun(args: string[], timeoutMs = 60_000): string {
return execFileSync('docker', args, {
encoding: 'utf8',
stdio: ['ignore', 'pipe', 'pipe'],
timeout: timeoutMs
})
}
/** Copy the deterministic agent fixture into the container so the host-resolved
* custom executable (`node <fixture>`) runs on the REMOTE, not the local host. */
export function placeCustomAgentFixtureInContainer(
target: DockerSshRelayTarget,
localFixturePath: string
): void {
dockerRun(['cp', localFixturePath, `${target.containerName}:${REMOTE_CUSTOM_AGENT_FIXTURE_PATH}`])
}
/** Read the live agent process(es) inside the container via /proc — the real
* spawned argv (cmdline) and environment (environ), NUL-separated. This proves
* the host launched THIS executable on the remote with exactly this argv/env,
* which a mocked HTTP handler could never observe. */
export function observeRemoteAgentProcesses(target: DockerSshRelayTarget): RemoteAgentProcess[] {
const raw = dockerRun([
'exec',
target.containerName,
'bash',
'-lc',
// -f matches the full cmdline; `|| true` keeps a zero-match exit non-fatal.
`pgrep -f ${REMOTE_CUSTOM_AGENT_FIXTURE_PATH} || true`
]).trim()
if (raw.length === 0) {
return []
}
const processes: RemoteAgentProcess[] = []
for (const line of raw.split('\n')) {
const pid = Number(line.trim())
if (!Number.isInteger(pid) || pid <= 0) {
continue
}
let cmdline: string
let environ: string
try {
cmdline = dockerRun(['exec', target.containerName, 'bash', '-lc', `cat /proc/${pid}/cmdline`])
environ = dockerRun(['exec', target.containerName, 'bash', '-lc', `cat /proc/${pid}/environ`])
} catch {
// The process may exit between pgrep and the /proc read; skip it.
continue
}
const argv = cmdline.split('\0').filter((part) => part.length > 0)
// Only the node process running the fixture is the launched agent; a shell
// wrapper whose cmdline merely mentions the path is not argv[0]-node.
const executable = argv[0]?.split('/').at(-1)
if (executable !== 'node') {
continue
}
const env = new Map<string, string>()
for (const pair of environ.split('\0')) {
if (pair.length === 0) {
continue
}
const eq = pair.indexOf('=')
if (eq <= 0) {
continue
}
env.set(pair.slice(0, eq), pair.slice(eq + 1))
}
processes.push({ pid, argv, env })
}
return processes
}
/** Add the Docker SSH target, connect the relay, register the seeded remote
* repo, and activate its worktree — the connectDockerRemote pattern, returning
* the identifiers a custom-agent launch and a reconnect attempt both need. */
export async function connectDockerRemoteWorktree(
page: Page,
target: DockerSshRelayTarget
): Promise<ConnectedRemoteWorktree> {
return await page.evaluate(
async ({ target, remotePath }) => {
const store = window.__store
if (!store) {
throw new Error('Store unavailable')
}
const credentialUnsub = window.api.ssh.onCredentialRequest((request) => {
void window.api.ssh.submitCredential({ requestId: request.requestId, value: null })
})
try {
const createdTarget = await window.api.ssh.addTarget({
target: {
label: `Docker SSH Custom Agent ${Date.now()}`,
host: '127.0.0.1',
port: target.port,
username: 'root',
identityFile: target.identityFile,
identitiesOnly: true,
relayGracePeriodSeconds: 1
}
})
const state = await window.api.ssh.connect({ targetId: createdTarget.id })
if (!state || state.status !== 'connected') {
throw new Error(`SSH target did not connect: ${JSON.stringify(state)}`)
}
store.getState().setSshConnectionState(createdTarget.id, state)
const labels = new Map(store.getState().sshTargetLabels)
labels.set(createdTarget.id, createdTarget.label)
store.getState().setSshTargetLabels(labels)
const result = await window.api.repos.addRemote({
connectionId: createdTarget.id,
remotePath,
displayName: 'Docker SSH Custom Agent'
})
if ('error' in result) {
throw new Error(result.error)
}
await store.getState().fetchRepos()
await store.getState().fetchWorktrees(result.repo.id)
const worktree = (store.getState().worktreesByRepo[result.repo.id] ?? [])[0]
if (!worktree) {
throw new Error(`No remote worktree found for ${result.repo.path}`)
}
store.getState().setActiveWorktree(worktree.id)
return {
targetId: createdTarget.id,
repoId: result.repo.id,
worktreeId: worktree.id
}
} finally {
credentialUnsub()
}
},
{ target, remotePath: DOCKER_SSH_RELAY_REMOTE_REPO_PATH }
)
}
/** Kill and re-establish the relay connection so a reconnect-settle attempt can
* observe whether the launch token settles without a duplicate PTY. */
export async function reconnectDockerTarget(page: Page, targetId: string): Promise<void> {
await page.evaluate(async (targetId) => {
const store = window.__store
if (!store) {
throw new Error('Store unavailable')
}
await window.api.ssh.disconnect({ targetId })
const state = await window.api.ssh.connect({ targetId })
if (!state || state.status !== 'connected') {
throw new Error(`SSH target did not reconnect: ${JSON.stringify(state)}`)
}
store.getState().setSshConnectionState(targetId, state)
}, targetId)
}
@@ -1,6 +1,9 @@
import { ONBOARDING_FINAL_STEP, ONBOARDING_FLOW_VERSION } from '../../../src/shared/constants'
import { FEATURE_INTERACTION_IDS } from '../../../src/shared/feature-interactions'
import { FEATURE_TIP_IDS } from '../../../src/shared/feature-tips'
import { AGENT_CATALOG_SCHEMA_VERSION } from '../../../src/main/agent-launch/agent-catalog-schema-migration'
import type { CustomTuiAgent } from '../../../src/shared/types'
import type { SourceControlActionRecipe } from '../../../src/shared/source-control-ai-actions'
const SEEN_FIRST_RUN_CONTEXTUAL_TOUR_IDS = [
'workspace-board',
@@ -11,14 +14,32 @@ const SEEN_FIRST_RUN_CONTEXTUAL_TOUR_IDS = [
] as const
const SEEN_FIRST_RUN_FEATURE_INTERACTION_TIMESTAMP = Date.parse('2026-01-01T00:00:00.000Z')
export function getE2ECompletedOnboardingProfile() {
export function getE2ECompletedOnboardingProfile(overrides?: {
customTuiAgents?: readonly CustomTuiAgent[]
sourceControlActions?: Readonly<Record<string, SourceControlActionRecipe>>
}) {
const customTuiAgents = overrides?.customTuiAgents ?? []
const sourceControlActions = overrides?.sourceControlActions
return {
settings: {
telemetry: {
optedIn: true,
installId: '00000000-0000-4000-8000-000000000000',
existedBeforeTelemetryRelease: false
}
},
// Stamp the current catalog schema so the boot migration short-circuits to
// a no-op (no pinned pre-v1 backup) and leaves the seeded agents untouched.
...(customTuiAgents.length > 0
? {
customTuiAgents,
agentCatalogSchemaVersion: AGENT_CATALOG_SCHEMA_VERSION,
agentCatalogRevision: 1,
agentReferenceRevision: 1
}
: {}),
// Seed source-control action recipes so a launch naming the recipe id via
// sourceRecord resolves its stored agentArgs into host-owned perLaunchArgs.
...(sourceControlActions ? { sourceControlAi: { actions: sourceControlActions } } : {})
},
onboarding: {
flowVersion: ONBOARDING_FLOW_VERSION,
+37 -2
View File
@@ -34,6 +34,8 @@ import {
createElectronHomeIsolation
} from './electron-home-isolation'
import { createSeededTestRepo, isValidGitRepo } from './seeded-test-repo'
import type { CustomTuiAgent } from '../../../src/shared/types'
import type { SourceControlActionRecipe } from '../../../src/shared/source-control-ai-actions'
type OrcaTestFixtures = {
electronApp: ElectronApplication
@@ -62,6 +64,17 @@ type OrcaTestFixtures = {
// PATH/token environment. Keep this fixture-owned so tests never mutate the
// developer's shell or already-running Orca instance.
launchEnv: NodeJS.ProcessEnv
// Why: the custom-agent launch spec seeds a custom agent into the persisted
// profile file (not authored via UI) so the launch boundary is exercised
// against a real v1 catalog entry. Fixture-owned so no other spec is affected.
// Wrapped in an object: Playwright's option-tuple detection silently truncates
// a bare multi-element array to its first element, so a bare array here would
// seed a partial catalog. The object keeps the array out of that detection.
seededCustomAgents: { agents: readonly CustomTuiAgent[] }
// Why: the recipe-args case seeds a source-control action recipe into the
// profile so a launch naming it (sourceRecord) resolves its stored agentArgs
// into host-owned perLaunchArgs. Fixture-owned so no other spec is affected.
seededSourceControlActions: Readonly<Record<string, SourceControlActionRecipe>>
}
type OrcaWorkerFixtures = {
@@ -178,7 +191,9 @@ export const test = base.extend<OrcaTestFixtures, OrcaWorkerFixtures>({
launchEnv,
orcaAppExtraEnv,
orcaAppExtraArgs,
registerPostElectronShutdownCleanup
registerPostElectronShutdownCleanup,
seededCustomAgents,
seededSourceControlActions
},
provideFixture,
testInfo
@@ -195,9 +210,27 @@ export const test = base.extend<OrcaTestFixtures, OrcaWorkerFixtures>({
// every other test. Seed a completed-onboarding fresh-install profile:
// an empty file would make persistence treat the profile as an
// existing-user upgrade cohort and mount the telemetry notice overlay.
const seededAgents = seededCustomAgents.agents
// Guard: a non-array here means Playwright's option-tuple detection
// truncated the value (the wrapper object exists to prevent that). Throw
// loudly so this truncation class can never silently seed a partial catalog.
if (!Array.isArray(seededAgents)) {
throw new Error(
`seededCustomAgents.agents must be an array; received ${typeof seededAgents}. ` +
'Pass { agents: [...] } — a bare array is truncated by Playwright option detection.'
)
}
const hasSeededActions = Object.keys(seededSourceControlActions).length > 0
const profileOverrides =
seededAgents.length > 0 || hasSeededActions
? {
...(seededAgents.length > 0 ? { customTuiAgents: seededAgents } : {}),
...(hasSeededActions ? { sourceControlActions: seededSourceControlActions } : {})
}
: undefined
writeFileSync(
path.join(userDataDir, 'orca-data.json'),
`${JSON.stringify(getE2ECompletedOnboardingProfile(), null, 2)}\n`
`${JSON.stringify(getE2ECompletedOnboardingProfile(profileOverrides), null, 2)}\n`
)
}
const headful = shouldLaunchHeadful(testInfo)
@@ -282,6 +315,8 @@ export const test = base.extend<OrcaTestFixtures, OrcaWorkerFixtures>({
launchEnv: [{}, { option: true }],
orcaAppExtraEnv: [{}, { option: true }],
orcaAppExtraArgs: [[], { option: true }],
seededCustomAgents: [{ agents: [] }, { option: true }],
seededSourceControlActions: [{}, { option: true }],
// Test-scoped: grab the first BrowserWindow, add the test repo, and wait
// until the session is fully ready with a worktree active.
+216
View File
@@ -0,0 +1,216 @@
import path from 'node:path'
import type { Page } from '@stablyai/playwright-test'
import type { CustomTuiAgent, CustomTuiAgentId } from '../../src/shared/types'
import { test, expect } from './helpers/orca-app'
import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store'
import {
countVisibleTerminalPanes,
sendToTerminal,
waitForActivePanePtyId,
waitForActiveTerminalManager,
waitForPaneCount,
waitForTerminalOutput
} from './helpers/terminal'
import {
cleanupDockerSshRelayTarget,
startDockerSshRelayTarget,
type DockerSshRelayTarget
} from './helpers/docker-ssh-relay-target'
import {
connectDockerRemoteWorktree,
observeRemoteAgentProcesses,
placeCustomAgentFixtureInContainer,
reconnectDockerTarget,
REMOTE_CUSTOM_AGENT_FIXTURE_PATH
} from './helpers/docker-ssh-custom-agent-remote'
const RUN_DOCKER_SSH = process.env.ORCA_E2E_SSH_DOCKER === '1'
const LOCAL_FIXTURE_PATH = path.join(
process.cwd(),
'tests/e2e/fixtures/custom-agent-launch-fixture.cjs'
)
const CUSTOM_AGENT_ID =
'custom-agent:codex:2b3c4d5e-6f7a-4b8c-9d0e-1f2a3b4c5d6e' as CustomTuiAgentId
const CUSTOM_AGENT_LABEL = 'E2E SSH Fixture Agent'
const READY_MARKER = 'CUSTOM_AGENT_FIXTURE_READY'
// A declared env value that no base agent, shell profile, or relay would set on
// its own, so finding it in the remote /proc/<pid>/environ proves the host
// carried THIS custom agent's env across the relay to the spawned process. The
// key must avoid the reserved ORCA_* namespace (rejected by field validation).
const ENV_MARKER_KEY = 'E2E_CUSTOM_MARKER'
const ENV_MARKER_VALUE = 'orca-remote-env-proof'
// The custom executable is `node <fixture>` where the fixture lives INSIDE the
// container: commandOverride is a bare `node` resolved on the remote PATH (the
// node:22 image ships it), and the space-free container path rides the v1 args
// template as one argument. The host resolves and spawns this on the remote.
const CUSTOM_AGENT: CustomTuiAgent = {
id: CUSTOM_AGENT_ID,
baseAgent: 'codex',
label: CUSTOM_AGENT_LABEL,
commandOverride: 'node',
args: REMOTE_CUSTOM_AGENT_FIXTURE_PATH,
env: { [ENV_MARKER_KEY]: ENV_MARKER_VALUE },
syncEnv: false
}
test.use({ seededCustomAgents: { agents: [CUSTOM_AGENT] } })
async function launchSeededCustomAgent(page: Page, worktreeId: string): Promise<string> {
// Same production boundary as launchAgentInNewTab: an empty command with an
// agentLaunch selection the host resolves at spawn (it owns command/arg/env
// assembly), driving the seeded commandOverride into the remote PTY.
return await page.evaluate(
({ worktreeId, agentId }) => {
const store = window.__store
if (!store) {
throw new Error('Store unavailable')
}
const state = store.getState()
const tab = state.createTab(worktreeId, undefined, undefined, { launchAgent: agentId })
state.queueTabStartupCommand(tab.id, {
command: '',
agentLaunch: {
selection: { kind: 'agent', agent: agentId },
allowEmptyPromptLaunch: true
},
telemetry: { launch_source: 'tab_bar_quick_launch', request_kind: 'new' }
})
state.setActiveTab(tab.id)
state.setActiveTabType('terminal')
return tab.id
},
{ worktreeId, agentId: CUSTOM_AGENT_ID }
)
}
test.describe('SSH custom-agent launch', () => {
test.skip(!RUN_DOCKER_SSH, 'Set ORCA_E2E_SSH_DOCKER=1 to run Docker-backed SSH custom-agent e2e.')
test.skip(process.platform === 'win32', 'Docker SSH custom-agent e2e uses POSIX ssh tooling.')
test('resolves and spawns a seeded custom agent on the SSH remote with the exact argv and env', async ({
orcaPage
}, testInfo) => {
test.slow()
let target: DockerSshRelayTarget | null = null
try {
target = startDockerSshRelayTarget(testInfo)
placeCustomAgentFixtureInContainer(target, LOCAL_FIXTURE_PATH)
await waitForSessionReady(orcaPage)
await waitForActiveWorktree(orcaPage)
// Self-verify the seed loaded into the real host catalog before launching,
// so schema drift fails loudly here instead of silently no-op'ing.
const loadedAgent = await orcaPage.evaluate(async (agentId) => {
const snapshot = await window.api.settings.agentCatalog.getLocal()
const entry = snapshot.customAgents.find(
(candidate) => candidate.status === 'ready' && candidate.definition.id === agentId
)
return entry && entry.status === 'ready'
? { id: entry.definition.id, label: entry.definition.label }
: null
}, CUSTOM_AGENT_ID)
expect(loadedAgent).toEqual({ id: CUSTOM_AGENT_ID, label: CUSTOM_AGENT_LABEL })
const remote = await connectDockerRemoteWorktree(orcaPage, target)
const launchedTabId = await launchSeededCustomAgent(orcaPage, remote.worktreeId)
await ensureTerminalVisible(orcaPage, 45_000)
await waitForActiveTerminalManager(orcaPage, 60_000)
await waitForPaneCount(orcaPage, 1, 60_000)
const ptyId = await waitForActivePanePtyId(orcaPage, 60_000)
// The fixture prints its readiness marker on spawn: the host resolved the
// custom executable and started it in the REMOTE pane's PTY.
await waitForTerminalOutput(orcaPage, READY_MARKER, 60_000, 80_000)
// Observe the real remote process: exactly one node agent, launched with
// the host-assembled argv (the container fixture path) and carrying the
// custom agent's declared env — read from /proc inside the container.
await expect
.poll(() => observeRemoteAgentProcesses(target!).length, {
timeout: 30_000,
message: 'remote custom-agent process did not appear in the container'
})
.toBe(1)
const [remoteProcess] = observeRemoteAgentProcesses(target)
expect(remoteProcess.argv).toContain(REMOTE_CUSTOM_AGENT_FIXTURE_PATH)
expect(remoteProcess.env.get(ENV_MARKER_KEY)).toBe(ENV_MARKER_VALUE)
// Trusted keyboard input reaches THIS remote process, which echoes it back
// with a prefix the terminal never produces (raw mode, no local echo).
await sendToTerminal(orcaPage, ptyId, 'ping\r')
await waitForTerminalOutput(orcaPage, 'CUSTOM_AGENT_ECHO:ping', 30_000, 80_000)
// Exactly one PTY/identity: one pane, and the tab carries the custom
// agent's identity as its visible launch state.
expect(await countVisibleTerminalPanes(orcaPage)).toBe(1)
const tabIdentity = await orcaPage.evaluate(
({ worktreeId, tabId }) => {
const tab = (window.__store?.getState().tabsByWorktree[worktreeId] ?? []).find(
(candidate) => candidate.id === tabId
)
return tab ? { launchAgent: tab.launchAgent, ptyId: tab.ptyId } : null
},
{ worktreeId: remote.worktreeId, tabId: launchedTabId }
)
expect(tabIdentity?.launchAgent).toBe(CUSTOM_AGENT_ID)
expect(typeof tabIdentity?.ptyId).toBe('string')
testInfo.annotations.push({
type: 'ssh-custom-agent-launch',
description: `remote pid=${remoteProcess.pid} argv=${remoteProcess.argv.join(' ')}`
})
await sendToTerminal(orcaPage, ptyId, '\x03')
} finally {
cleanupDockerSshRelayTarget(target)
}
})
test('settles the launch token after a relay disconnect/reconnect without a duplicate remote agent', async ({
orcaPage
}, testInfo) => {
test.slow()
let target: DockerSshRelayTarget | null = null
try {
target = startDockerSshRelayTarget(testInfo)
placeCustomAgentFixtureInContainer(target, LOCAL_FIXTURE_PATH)
await waitForSessionReady(orcaPage)
await waitForActiveWorktree(orcaPage)
const remote = await connectDockerRemoteWorktree(orcaPage, target)
await launchSeededCustomAgent(orcaPage, remote.worktreeId)
await ensureTerminalVisible(orcaPage, 45_000)
await waitForActiveTerminalManager(orcaPage, 60_000)
await waitForActivePanePtyId(orcaPage, 60_000)
await waitForTerminalOutput(orcaPage, READY_MARKER, 60_000, 80_000)
await expect
.poll(() => observeRemoteAgentProcesses(target!).length, { timeout: 30_000 })
.toBe(1)
// Kill and re-establish the relay. Recovery must never guess liveness or
// identity into a SECOND launch: the settled state carries at most one
// remote agent process and one PTY identity, never a duplicate.
await reconnectDockerTarget(orcaPage, remote.targetId)
await ensureTerminalVisible(orcaPage, 45_000)
await waitForActiveTerminalManager(orcaPage, 60_000)
await expect
.poll(() => observeRemoteAgentProcesses(target!).length, {
timeout: 30_000,
message: 'reconnect produced a duplicate remote custom-agent process'
})
.toBeLessThanOrEqual(1)
testInfo.annotations.push({
type: 'ssh-custom-agent-reconnect-settle',
description: `remote agents after reconnect: ${observeRemoteAgentProcesses(target).length}`
})
} finally {
cleanupDockerSshRelayTarget(target)
}
})
})