diff --git a/.github/actions/setup-wsl-test-runtime/action.yml b/.github/actions/setup-wsl-test-runtime/action.yml new file mode 100644 index 00000000000..f919c2e75bc --- /dev/null +++ b/.github/actions/setup-wsl-test-runtime/action.yml @@ -0,0 +1,8 @@ +name: Set up WSL test runtime +description: Install a checksum-pinned Ubuntu WSL1 guest with executable Node and Git for real terminal tests. +runs: + using: composite + steps: + - name: Provision Ubuntu WSL1 + shell: pwsh + run: '& "${{ github.action_path }}/setup.ps1"' diff --git a/.github/actions/setup-wsl-test-runtime/setup.ps1 b/.github/actions/setup-wsl-test-runtime/setup.ps1 new file mode 100644 index 00000000000..2fd012eb246 --- /dev/null +++ b/.github/actions/setup-wsl-test-runtime/setup.ps1 @@ -0,0 +1,32 @@ +$ErrorActionPreference = 'Stop' +if (-not $IsWindows) { throw 'WSL test provisioning requires a Windows runner' } + +$rootfs = Join-Path $env:RUNNER_TEMP 'noble-rootfs.tar.gz' +Invoke-WebRequest 'https://releases.ubuntu.com/24.04.4/ubuntu-24.04.4-wsl-amd64.wsl' -OutFile $rootfs +if ((Get-FileHash $rootfs -Algorithm SHA256).Hash.ToLowerInvariant() -ne '9b2f7730dc68227dd04a9f3e5eab86ad85caf556b8606ad94f1f29ff5c4fd3f5') { throw 'Ubuntu rootfs checksum mismatch' } +$distroDir = Join-Path $env:RUNNER_TEMP 'orca-wsl-ubuntu' +wsl.exe --import Ubuntu $distroDir $rootfs --version 1 +if ($LASTEXITCODE -ne 0) { throw "WSL import failed: $LASTEXITCODE" } +wsl.exe --distribution Ubuntu --user root --exec /usr/bin/true +if ($LASTEXITCODE -ne 0) { throw "WSL guest did not start: $LASTEXITCODE" } +wsl.exe --distribution Ubuntu --user root --exec /usr/bin/apt-get update +if ($LASTEXITCODE -ne 0) { throw "WSL apt update failed: $LASTEXITCODE" } +wsl.exe --distribution Ubuntu --user root --exec /usr/bin/apt-get install --yes git curl xz-utils +if ($LASTEXITCODE -ne 0) { throw "WSL git install failed: $LASTEXITCODE" } +$kernelMsi = Join-Path $env:RUNNER_TEMP 'wsl_update_x64.msi' +Invoke-WebRequest 'https://wslstorestorage.blob.core.windows.net/wslblob/wsl_update_x64.msi' -OutFile $kernelMsi +if ((Get-FileHash $kernelMsi -Algorithm SHA256).Hash.ToLowerInvariant() -ne '4d09c776c8d45f70a202281d18e19be1118f53159b0c217a5274a31ce18525fe') { throw 'WSL kernel installer checksum mismatch' } +$installer = Start-Process msiexec.exe -ArgumentList @('/i', $kernelMsi, '/quiet', '/norestart') -Wait -PassThru +if ($installer.ExitCode -ne 0) { throw "WSL kernel installation failed: $($installer.ExitCode)" } +wsl.exe --status +if ($LASTEXITCODE -ne 0) { throw "WSL status failed: $LASTEXITCODE" } +wsl.exe --distribution Ubuntu --user root --exec /usr/bin/curl --fail --silent --show-error --location https://nodejs.org/dist/v22.14.0/node-v22.14.0-linux-x64.tar.xz --output /tmp/orca-node.tar.xz +if ($LASTEXITCODE -ne 0) { throw 'Node download failed' } +$nodeHash = wsl.exe --distribution Ubuntu --user root --exec /usr/bin/sha256sum /tmp/orca-node.tar.xz +if ($LASTEXITCODE -ne 0 -or -not ($nodeHash -match '^69b09dba5c8dcb05c4e4273a4340db1005abeafe3927efda2bc5b249e80437ec')) { throw 'Node checksum mismatch' } +wsl.exe --distribution Ubuntu --user root --exec /usr/bin/tar -xJf /tmp/orca-node.tar.xz -C /usr/local --strip-components=1 +if ($LASTEXITCODE -ne 0) { throw 'Node extraction failed' } +wsl.exe --distribution Ubuntu --user root --exec /usr/local/bin/node --version +if ($LASTEXITCODE -ne 0) { throw 'Node cannot execute in WSL' } +wsl.exe --list --verbose +if ($LASTEXITCODE -ne 0) { throw "WSL enumeration failed: $LASTEXITCODE" } diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml index a94a7ea2ba5..f75d7ba00bb 100644 --- a/.github/workflows/e2e.yml +++ b/.github/workflows/e2e.yml @@ -227,6 +227,11 @@ jobs: mapfile -t TEST_FILES < <(jq -r '.[] | select( . != "tests/e2e/ssh-startup-exec-readiness.spec.ts" and . != "tests/e2e/paired-startup-exec-readiness.spec.ts" and + . != "tests/e2e/local-ssh-browser-routing.spec.ts" and + . != "tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts" and + . != "tests/e2e/ssh-localhost.spec.ts" and + . != "tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts" and + . != "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts" and . != "tests/e2e/terminal-ibus-hangul-native.spec.ts" )' <<<"$TEST_FILES_JSON") if [ "${#TEST_FILES[@]}" -eq 0 ]; then @@ -262,12 +267,15 @@ jobs: needs: [build, prepare-native-cache] # effect of one route listing a startup-readiness spec — pruning that spec would have # silently retired the whole lane. The signal is now derived from the SSH routes directly. - # The two spec clauses stay for their honest purpose: changed-e2e hands these specs to this + # The explicit spec clauses stay for their honest purpose: changed-e2e hands these specs to this # lane, so editing one must still run it here. if: >- inputs.test_files == '' || inputs.ssh_source_changed == 'true' || + contains(inputs.test_files, 'tests/e2e/local-ssh-browser-routing.spec.ts') || + contains(inputs.test_files, 'tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts') || contains(inputs.test_files, 'tests/e2e/ssh-startup-exec-readiness.spec.ts') || + contains(inputs.test_files, 'tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts') || contains(inputs.test_files, 'tests/e2e/paired-startup-exec-readiness.spec.ts') runs-on: ubuntu-latest # Why 60: this lane now also runs the remaining Docker-SSH specs serially. They average @@ -348,3 +356,87 @@ jobs: path: e2e-traces/ retention-days: 7 if-no-files-found: ignore + + ssh-browser-network-route: + name: ssh browser network route + if: inputs.test_files == '' || contains(inputs.test_files, 'tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts') + runs-on: ubuntu-latest + timeout-minutes: 15 + steps: + - uses: actions/checkout@v6 + with: + ref: ${{ inputs.ref || github.ref }} + - uses: ./.github/actions/install-node-dependencies + with: + native-runtime: node + - name: Install SSH client + run: sudo apt-get update && sudo apt-get install -y openssh-client + - name: Run Docker SSH browser network route journeys + env: + ORCA_BACKGROUND_LAUNCH: '1' + ORCA_RUN_DOCKER_SSH_BROWSER_E2E: '1' + run: node_modules/.bin/vitest run --config config/vitest.config.ts tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts + + ssh-localhost: + name: localhost SSH terminal and hooks + needs: [build, prepare-native-cache] + if: inputs.test_files == '' || contains(inputs.test_files, 'tests/e2e/ssh-localhost.spec.ts') + runs-on: ubuntu-latest + timeout-minutes: 20 + steps: + - uses: actions/checkout@v6 + with: + ref: ${{ inputs.ref || github.ref }} + - name: Install SSH server and headless tools + run: sudo apt-get update && sudo apt-get install -y build-essential openssh-client openssh-server python3 ripgrep xvfb zsh openbox x11-utils + - uses: ./.github/actions/install-node-dependencies + with: + native-runtime: electron + - uses: actions/download-artifact@v8 + with: + name: e2e-build-out + path: out/ + - name: Start isolated localhost SSH server + shell: bash + run: | + # Bare shells install Pi extensions only for an existing agent home. + mkdir -p "$HOME/.pi/agent" + fixture="$RUNNER_TEMP/orca-localhost-sshd" + mkdir -p "$fixture" + ssh-keygen -q -t ed25519 -N '' -f "$fixture/host_key" + ssh-keygen -q -t ed25519 -N '' -f "$fixture/client_key" + cat > "$fixture/sshd_config" <> "$GITHUB_ENV" + - name: Run localhost SSH terminal and hook journey + env: + SKIP_BUILD: '1' + ORCA_E2E_SSH_LOCALHOST: '1' + ORCA_FEATURE_REMOTE_AGENT_HOOKS: '1' + ORCA_E2E_FORWARD_APP_LOGS: '1' + run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh pnpm exec playwright test --config tests/playwright.config.ts tests/e2e/ssh-localhost.spec.ts --project=electron-headless --workers=1 + - uses: actions/upload-artifact@v7 + if: failure() + with: + name: localhost-ssh-traces + path: test-results/ + retention-days: 7 + if-no-files-found: ignore diff --git a/.github/workflows/golden-e2e-experiment.yml b/.github/workflows/golden-e2e-experiment.yml index d46c80033fa..11cfa866c67 100644 --- a/.github/workflows/golden-e2e-experiment.yml +++ b/.github/workflows/golden-e2e-experiment.yml @@ -98,12 +98,17 @@ jobs: $env:SKIP_BUILD = '1' $env:ORCA_E2E_FORWARD_APP_LOGS = '1' pnpm run --if-present test:e2e:workspace-session-golden + if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } pnpm run --if-present test:e2e:windows-fresh-startup-golden + if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } pnpm run --if-present test:e2e:tab-bar-agent-launch-golden + if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } if (Test-Path tests/e2e/golden-fresh-profile-terminal.spec.ts) { pnpm run test:e2e -- tests/e2e/golden-fresh-profile-terminal.spec.ts tests/e2e/golden-shell-command.spec.ts + if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } } pnpm run --if-present test:e2e:source-control-golden + if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } - name: Upload Playwright traces if: failure() diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index bd655eb801a..1fd141ee4a0 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -45,6 +45,7 @@ jobs: test_files: ${{ steps.e2e_filter.outputs.test_files }} ssh_source_changed: ${{ steps.e2e_filter.outputs.ssh_source_changed }} native_ime_source_changed: ${{ steps.e2e_filter.outputs.native_ime_source_changed }} + wsl_source_changed: ${{ steps.e2e_filter.outputs.wsl_source_changed }} steps: - name: Checkout uses: actions/checkout@v6 @@ -92,6 +93,9 @@ jobs: # trigger on IME source rather than on a spec name in some route's list. NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)" echo "native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED" >> "$GITHUB_OUTPUT" + WSL_CHANGED="$(git diff --name-only --no-renames --diff-filter=ACDMR --merge-base "$BASE" "$HEAD")" + WSL_SOURCE_CHANGED="$(printf '%s\n' "$WSL_CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --wsl-source)" + echo "wsl_source_changed=$WSL_SOURCE_CHANGED" >> "$GITHUB_OUTPUT" echo "Native IME source changed: $NATIVE_IME_SOURCE_CHANGED" SHOULD_RUN="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --reusable-workflow)" if [ "$SHOULD_RUN" = true ]; then @@ -942,6 +946,16 @@ jobs: contents: read uses: ./.github/workflows/terminal-ime-e2e.yml + windows_wsl: + name: real WSL terminal + needs: code_paths + if: needs.code_paths.outputs.wsl_source_changed == 'true' + permissions: + contents: read + uses: ./.github/workflows/windows-wsl-e2e.yml + with: + ref: ${{ github.event.pull_request.head.sha }} + verify: if: always() needs: diff --git a/.github/workflows/windows-wsl-e2e.yml b/.github/workflows/windows-wsl-e2e.yml new file mode 100644 index 00000000000..fb781e25331 --- /dev/null +++ b/.github/workflows/windows-wsl-e2e.yml @@ -0,0 +1,74 @@ +name: Windows WSL terminal E2E + +on: + workflow_dispatch: + inputs: + ref: + description: Commit to validate + type: string + required: false + workflow_call: + inputs: + ref: + type: string + required: false + +permissions: + contents: read + +concurrency: + group: windows-wsl-e2e-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + +jobs: + wsl-terminal: + runs-on: windows-2022 + timeout-minutes: 30 + env: + NODE_OPTIONS: --max-old-space-size=4096 + steps: + - uses: actions/checkout@v6 + with: + ref: ${{ inputs.ref || github.sha }} + persist-credentials: false + - uses: ./.github/actions/setup-wsl-test-runtime + - uses: ./.github/actions/install-node-dependencies + with: + native-runtime: electron + - name: Build relay and Electron + run: | + pnpm run build:relay + if ($LASTEXITCODE -ne 0) { throw 'Relay build failed' } + pnpm exec electron-vite build --mode e2e + if ($LASTEXITCODE -ne 0) { throw 'Electron build failed' } + - name: Exercise real WSL launch and paste + env: + SKIP_BUILD: '1' + ORCA_E2E_FORWARD_APP_LOGS: '1' + PLAYWRIGHT_JSON_OUTPUT_FILE: test-results/wsl-results.json + run: >- + pnpm exec playwright test + tests/e2e/golden-tab-bar-agent-launch.spec.ts + tests/e2e/terminal-windows-shell-paste-ownership.spec.ts + --config tests/playwright.config.ts + --project=electron-headless + --grep "WSL" + --repeat-each=3 + --workers=1 + --reporter=list,json + - name: Require all nine WSL executions + if: always() + run: node config/scripts/verify-wsl-e2e-participation.mjs test-results/wsl-results.json + - name: Upload WSL participation report + uses: actions/upload-artifact@v7 + if: always() + with: + name: windows-wsl-participation-report + path: test-results/wsl-results.json + retention-days: 3 + - uses: actions/upload-artifact@v7 + if: failure() + with: + name: windows-wsl-terminal-traces + path: test-results/ + retention-days: 7 diff --git a/cloud/docs/relay-improvement-checklist-2026-09.md b/cloud/docs/relay-improvement-checklist-2026-09.md index 91f1cc742ef..8d86afd4699 100644 --- a/cloud/docs/relay-improvement-checklist-2026-09.md +++ b/cloud/docs/relay-improvement-checklist-2026-09.md @@ -4,18 +4,18 @@ Companion to [`relay-improvement-roadmap-2026-09.md`](./relay-improvement-roadma match). This file answers three questions per item: what are the concrete steps, what can run in parallel, and will a user notice. -## Status as of 2026-09-04 22:30Z +## Status as of 2026-09-06 16:30Z Three buckets. "Merged" means the code is on `main` and nothing in production has changed yet. "Deployed" means users are already getting it. "Awaiting owner" means I will not touch production without a go. **Deployed to production** +- Roll 2 relay image `4916ed67` (stablyai/orca #18959 + #18722 + #18720 flag unset): director since 2026-09-06 01:02Z, all 19 general cells by 16:29Z. Control lease 6 h ± 30 min, accept abandonment, per-cell inventory locks, pool `statement_timeout`. Record: findings doc, "Roll 2" section. - Auth instance cap 20 + dead-family audit fix (orca-cloud #474) as revision `orca-cloud-auth-00031-tox`. - Dynamic NAT ports in both regions (stablyai/orca #18693). Zero drops and zero proxy dial errors since. - Nine alert policies with log metrics: 4 auth (#475), 3 relay Cloud SQL/NAT (#18693), 1 cell process-exit (#18717), all on the relay Slack channel. -**Merged, ships with the next relay cell image roll (Roll 1 carries `519f4914`; Roll 2 needs a fresh image build)** -- Per-cell inventory locks, delta counters, pool `statement_timeout` (#18722). Roll 2. -- Cells dial Cloud SQL with `--private-ip` when configured (#18720). Inert until 2.1 applies. +**Merged, not yet live** +- Cells dial Cloud SQL with `--private-ip` when configured (#18720). Deployed in Roll 2 with the flag unset; inert until 2.1 applies. - Phone shows a clear "sign in on the desktop again" state when the desktop is signed out (#18698). **Merged, ships with the next auth deploy** @@ -158,9 +158,10 @@ independent. (2.2 deferred; if revived, do it after 2.1 so the new instance is p - [ ] Production: announce a window; same steps; verify `orca_relay_runtime_metrics` controls recover to pre-cutover count. - [ ] Update `production-cloud-sql-app-consumers` budget test and both alert policies' `database_id`. -### 2.3 Relay pool statement timeout (merged stablyai/orca #18722; ships Roll 2) +### 2.3 Relay pool statement timeout (deployed in Roll 2, 2026-09-06) - [x] `statement_timeout` on the relay `pg.Pool` (5 s, env-configurable; schema pool untimed; `57014` retryable), below the control-renewal deadline; DDL on an untimed connection (same pattern as auth #476). - [x] Postgres test on 55440: a held lock fails the query fast and the bounded retry takes over. +- [x] Deployed fleet-wide in Roll 2 (`4916ed67`), 2026-09-06. ### 3.1 Refresh rotation grace window (orca-cloud #478 merged 2026-09-04; deploy pending owner go) - [ ] Fix the deploy-script env strip for `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` (pre-existing; found by #478). @@ -169,14 +170,16 @@ independent. (2.2 deferred; if revived, do it after 2.1 so the new instance is p - [x] Tests: replay inside window returns same successor; outside revokes; concurrent double-present yields one successor. - [x] Deploy via `deploy-auth-production` (candidate → smoke → promote). Deployed 2026-09-04 23:15Z as `orca-cloud-auth-00035-gos`, cap 20 kept, 0 5xx; `successor_material` column present; sealed successors being written. (candidate → smoke → promote). -### 3.2 / 4.3 Desktop (merged stablyai/orca #18719; ships next desktop release) +### 3.2 / 4.3 Desktop (merged stablyai/orca #18719; ships next desktop release; relay side of 4.3 deployed in Roll 2) - [x] 3.2: on refresh timeout, re-read stored session before retrying; do not re-send a token already rotated locally. - [x] 4.3: ±10 % jitter on control lease renewal; unit test on the distribution; wire-compatible (server accepts early renewals already). +- [x] 4.3 relay side: control lease 55 min → 6 h ± 30 min (#18959), deployed in Roll 2, 2026-09-06. -### 4.1 Lock contention (partial: stablyai/orca #18722 merged; ships Roll 2) +### 4.1 Lock contention (partial: stablyai/orca #18722 deployed in Roll 2, 2026-09-06) - [x] Replace the global `FOR UPDATE` over `relay_cells` with per-cell row locks; counters delta-only. Remaining: `assignOnce` placement lock is still global (optimistic snapshot follow-up). with per-cell row locks or `pg_advisory_xact_lock(cell)`; counters delta-only. - [x] Postgres tests on 55440 with concurrent probes (in #18722). Staging load run still owed; `postgres_retries` per hour drops in staging load run. -- [ ] Ships in Roll 2; then 4.4 recalibrates the retries bar from a week of data. +- [x] Shipped in Roll 2 (2026-09-06). Director retries first 6 h on the new image: 13 vs 85 on the predecessor's prior 6 h. +- [ ] 4.4: recalibrate the retries bar from a week of data (after 2026-09-13). ### 4.2 Region preference - [ ] Director: honor requested region when the preferred region has headroom, else sticky. Behind the existing flag. diff --git a/cloud/docs/relay-reconnect-2026-09-findings.md b/cloud/docs/relay-reconnect-2026-09-findings.md index 580a4da84d8..efb23f380bd 100644 --- a/cloud/docs/relay-reconnect-2026-09-findings.md +++ b/cloud/docs/relay-reconnect-2026-09-findings.md @@ -999,4 +999,34 @@ Owner: "sure, feel free to drive these." Sequence chosen: Roll 1 first (highest | Image publish | run 34002233801 → `sha256:4916ed676d8389f694a648e750f1112d9002d68c84a1e0c7af828d5af129de62`; mirrored to staging (run 34002326150). | | | Staging cell smoke | **Dropped.** Staging C4 is pinned to the Asia launch digest by `relay-staging-c4-refresh-workflow.test.mjs` (with production c27–c29 tfvars and the C4 recovery workflow) and the only C4 image-refresh path pins its accepted predecessor to an older digest. Re-pinning all of it for a smoke widens into the Asia launch machinery; #18969 closed. Roll 2 follows the Roll 1 path: director first, c7 as the rehearsal cell. | | | Director deploy | run 34002673626 **success** 01:02Z: serving `orca-cloud-relay-00575-leq` on `4916ed67`, `00574-wag` (same image) tagged `selector-rollback`, `00569-ret` (`519f4914`) still deployable. Baseline before: 1 director Postgres retry in the prior hour, 0 `container die`. | | -| c7 `verify` (read-only) | run 34002885408 dispatched 01:03Z, target `4916ed67`, rollback `85bf6799`, protocol 1, gen 148. | | +| c7 `verify` (read-only) | run 34002885408 **success** (gate success, cell_1 rollout success, release_lease success), target `4916ed67`, rollback `85bf6799`, protocol 1, gen 148. | | +| Director go/no-go (01:02Z–07:00Z, 6 h on `00575-leq`) | **Go.** Presence confirmed (13.8k assign 200s, 410 cell + 90 director `runtime_metrics` rows/30 min). Postgres retries 13 (all `55P03` lock_timeout) vs 85 on `00570-siv` in the prior 6 h. `/v1/assign` mix 200/401/503 = 13820/5557/623 vs 14081/5256/663 before the deploy; 503s are the placement/sticky admission `Retry-After` path and cluster by source (top source 351), same shape as before. 0 `container die`, cell `sqlFailuresDelta` sum 0. The earlier all-zero read at 01:28Z was a dead gcloud credential, not a quiet fleet, and was discarded. | | +| Monitor dry-run (Roll 2 gate 1) | run 34018071984 dispatched 07:03Z at gen 148, **green** 07:18Z at `1326d6b40c`; main had moved to `b51bbf3fc6` with identical trusted code. | | +| c7 `canary-apply` (run 34018804481) | **Succeeded** 07:18–07:31Z, protocol 1, rollback `85bf6799`: gate, rollout, seal_canary, release_lease all success. Template `…-20260906072156…` on `4916ed67`; selector gen 148 → 150. Four `container die` at 07:29:16–25Z were the new container exiting during boot (`applyPostgresSchema`/`backfillRelayCellRegions` → `Connection terminated due to connection timeout`, exit 1, 2 s runtime each) while the `cloud-sql-proxy` sidecar warmed up; fifth start at 07:29:26 listening, readiness check passed 07:29:27. Same boot-order race as c13 in Roll 1 batch 1, no serving impact (cell was still drained). 139 controls by 07:34Z and climbing, `sqlFailuresDelta` 0, `sqlLatencyMsMax` ~40 ms. | | +| Monitor dry-run (Roll 2 gate 2) | run 34019568779 dispatched 07:36Z at gen 150, **green** 07:51Z at `57e34c7f03` (main `6494f2a4f0`, identical trusted code). | | +| c8 `canary-apply` (run 34020284092) | **Succeeded** 07:52–08:09Z, protocol 1, rollback `519f4914`: all jobs success. Template `…-20260906075820…` on `4916ed67`; gen 150 → 152. One boot-race `container die` at 08:05:51Z (2 s, exit 1), next start served. 101 controls by 08:10Z, `sqlFailuresDelta` 0. | | +| Monitor dry-run (Roll 2 gate 3) | run 34021119905 dispatched 08:11Z at gen 152, **green** 08:26Z at `ffbf35e0d2`. | | +| Batch 1 `batch-apply` c9,c10,c13,c14 (run 34021868303, canary 34020284092) | **Failed on cell 3 (c13); c9 and c10 succeeded.** c9 08:27–08:43Z → gen 154, c10 08:43–08:58Z → gen 156, both trust-proven and restored general. c13: isolate → gen 157, drain, template `…-20260906090225…` on `4916ed67`, one boot-race exit 09:09:50Z, readiness 09:09:51Z, transition verifier passed at migration-only 09:11:17Z (2 680 assignments, heartbeat fresh, image `4916ed67`), then `probe-relay-rehome-trust` got **409** from the director at 09:11:18Z (157 ms; c9/c10 got 200 in ~178 ms). Failsafe re-asserted migration-only at gen 157 (no change). c14 skipped, lease released. c13 is **serving on the new image but isolated**: 151 controls by 09:18Z, `sqlFailuresDelta` 0, no exits fleet-wide after 09:12Z. The probe script prints only the status, not the director's `error` body, and neither the director nor c13 logs the 409 reason; candidates are the director's source check (`runtime.ready`/`heartbeatFresh`/incarnation read ~1 s after the verifier passed) or c13's `host-drain` rejecting the probe (incarnation mismatch, shared-runtime-identity proof, or the probe host unexpectedly present). Monitor residual: the probe should print the error body. | | +| Monitor dry-run (Roll 2 gate 4) + c13 recovery | Gate run 34024459585 dispatched 09:26Z at gen 157 with c13 in migration-only. On green: `mode=rollback` for c13 with rollback digest `4916ed67` (what it already runs) and target `519f4914`, protocol 1 both ways: `ROLLBACK_RESUME=true` path, no restart, verify + trust probe + restore general. As in Roll 1 (c8 recovery), the rollback mode seals no canary authority, so c14 runs as its own `canary-apply` and the next batch is c15,c16,c19,c20 behind that. | | +| c13 recovery (run 34025225328, `mode=rollback`) | Gate 4 **green** 09:38Z. Recovery **succeeded** 09:38–09:42Z: `ROLLBACK_RESUME=true`, no restart, verifier passed at migration-only (2 679 assignments, heartbeat fresh, `4916ed67`), **trust probe passed** (`host-not-connected` ×2, idempotent, shared runtime identity rejected), activate → **gen 158**, c13 general, verifier passed again. 154 controls, `sqlFailuresDelta` 0, no exits fleet-wide since 09:12Z. The 09:11Z 409 was therefore transient: same cell, same incarnation, same image, ~30 min later the identical probe passed. Most likely the director's source check reading the runtime row within ~1 s of the verifier's pass (a `ready`/heartbeat edge), which a retry in the workflow step would absorb. Residual: retry the trust probe once on 409 and print the error body. | | +| Monitor dry-run (Roll 2 gate 5) | run 34025450523 dispatched 09:44Z at gen 158, **green** 09:59Z at `6933fd70d7` (main `d19be485d3`, identical trusted code). | | +| c14 `canary-apply` (run 34026157631) | **Succeeded** 09:59–10:20Z, protocol 1: trust-proven, gen 158 → 160, canary authority sealed. No boot exits, 102 controls by 10:22Z, fleet `sqlFailuresDelta` 0 over 30 min. | | +| Monitor dry-run (Roll 2 gate 6) | run 34027238190 dispatched 10:23Z at gen 160, **green** 10:38Z at `ec64df335e` (main `adcc30be3b`, identical trusted code). | | +| Batch 2 `batch-apply` c15,c16,c19,c20 (run 34027985784, canary 34026157631) | **All four succeeded** 10:38–11:31Z, protocol 1, four trust proofs, gen 160 → 168. Boot-race exits only: 3 at 10:50Z (c16) and 5 at 11:02Z (c19), all 2–4 s, exit 1, next start served. Controls at 11:32Z: c15 160, c16 164, c19 164, c20 87 (still refilling). Fleet `sqlFailuresDelta` 1 over 30 min. | | +| Monitor dry-run (Roll 2 gate 7) | run 34030557166 dispatched 11:33Z at gen 168, **green** 11:48Z at `adcc30be3b`. | | +| c22 `canary-apply` (run 34031304526) | **Succeeded** 11:48–12:02Z, protocol 1, trust-proven, gen 168 → 170, canary authority sealed. No boot exits, 134 controls by 12:03Z. One correlated 1 s lock-timeout blip at 11:35:17–27Z (c10, c13, c19, c25, c28: one `sqlFailuresDelta` each, `sqlLatencyMsMax` ≈1 000 ms) spanning old and new images, the known lock-wait shape, not roll-related. Director retries 4 in the last hour. | | +| Monitor dry-run (Roll 2 gate 8) | run 34032011250 dispatched 12:05Z at gen 170, **green** 12:20Z at `adcc30be3b`. | | +| Batch 3 `batch-apply` c23,c24,c25,c26 (run 34032799574, canary 34031304526) | **Failed on cell 4 (c26); c23, c24, c25 succeeded** (12:20–13:11Z, gen 170 → 176, three trust proofs). c26: isolate → gen 177, drain, template `…-20260906131159…` on `4916ed67`, one boot-race exit 13:19:17Z, readiness 13:19:19Z, transition verifier passed at migration-only 13:20:42Z (2 604 assignments, heartbeat fresh, `4916ed67`), then the very next call, `admin_post target-runtime` to `c26.relay.onorca.dev/v1/admin/runtime-status`, got **503 `unconditional drop overload`** (27-byte body) and the step failed. That string is not in the relay codebase and c26 logged nothing at 13:20:42Z (readiness at 13:19:19Z, metrics steady), so it is a front-end/LB shed on one request; curl's `--retry 3` logged no retry attempt. Failsafe re-asserted migration-only at gen 177 (no change). c26 is serving on the new image but isolated: 166 controls by 13:25Z and climbing, `sqlFailuresDelta` 0. Residual: the post-apply `admin_post` should retry on 503 (the pre-apply one already tolerates a transient 5xx by comment). | | +| c26 recovery (run 34036875433, `mode=rollback`) | Gate 9 (run 34036059275) **green** 13:41Z at gen 177 with c26 migration-only. Recovery **succeeded** 13:42–13:46Z: `ROLLBACK_RESUME=true`, no restart, verifier + trust probe passed, activate → **gen 178**, c26 general. 176 controls, `sqlFailuresDelta` 0, no exits since 13:25Z. **All 16 US general cells are on `4916ed67`.** | | +| Monitor dry-run (Roll 2 gate 10) | run 34037169783 dispatched 13:48Z at gen 178, **green** 14:03Z at `f952f1ac96`. | | +| c27 `canary-apply` (run 34037973681, Asia, protocol 0) | **Succeeded** 14:03–14:19Z, gen 178 → 180, canary authority sealed (unused; Asia cells roll as single canaries). Template on `4916ed67`, no boot exits, 51 controls by 14:20Z (Asia cell, refilling), `sqlFailuresDelta` 0, `sqlLatencyMsMax` ~1 040 ms (cross-region baseline, c28 on the old image reads ~1 055 ms). Fleet `sqlFailuresDelta` 5 over 30 min: c28 ×3 (~1.17 s), c8 and c9 ×1 (1 s bar), the known lock-wait singles. | | +| Monitor dry-run (Roll 2 gate 11) | run 34038869552 dispatched 14:21Z at gen 180, **green** 14:36Z at `f952f1ac96`. | | +| c28 `canary-apply` (run 34039710735, Asia, protocol 0) | **Succeeded** 14:36–14:53Z, gen 180 → 182. Template on `4916ed67`, no boot exits, 37 controls by 14:55Z (refilling), `sqlFailuresDelta` 0, `sqlLatencyMsMax` ~1 045 ms. Fleet `sqlFailuresDelta` 3 over 30 min. | | +| Monitor dry-run (Roll 2 gate 12) | run 34040698172 dispatched 14:56Z at gen 182, **green** 15:12Z at `1d2e00819f`. | | +| c29 `canary-apply` (run 34041558414, Asia, protocol 0) | **Succeeded** 15:12–15:28Z, gen 182 → 184. No boot exits, 55 controls by 15:29Z. | | +| Census 15:29Z | MIG templates: 18 of 19 general cells on `4916ed67`; **c21 still on `519f4914`**. When c13's recovery re-sealed the canary at c14, batch 2 took c15,c16,c19,c20 and c21 dropped out of the plan's wave (`c15 canary + c16,c19,c20,c21`). Fleet 23 cells, 2 971 controls. Roll 2 exits since 07:00Z: 20, all boot-race (<10 s), 0 serving. Director retries 5 in the last hour. c21 rolls next as a single canary. | | +| Monitor dry-run (Roll 2 gate 13) | run 34042460176 dispatched 15:30Z at gen 184, **green** 15:45Z at `3631f886a7`. | | +| c21 `canary-apply` (run 34043296422, protocol 1) | **Failed at the same post-apply step as c26.** Isolate → gen 185, drain, template `…-20260906155550…` on `4916ed67`, verifier passed at migration-only 16:04:46Z (2 607 assignments, heartbeat fresh, `4916ed67`), then `admin_post target-runtime` to c21 got **503 `unconditional drop overload`** again (27-byte body, ~160 ms after the verifier's own successful read). Failsafe held migration-only at gen 185. c21 serving on the new image, isolated, 111 controls by 16:07Z. Second occurrence in ~3 h on two different cells, both ~1.3 min after readiness: consistent with an edge shed on the first admin request after the LB backend flips healthy. The step needs the same transient-5xx tolerance as the pre-apply read. | | +| Monitor dry-run (Roll 2 gate 14) + c21 recovery | Gate run 34044440616 dispatched 16:08Z at gen 185 with c21 migration-only. On green: `mode=rollback` resume for c21 (rollback digest `4916ed67`, protocol 1). | | +| c21 recovery (run 34045296151, `mode=rollback`) | Gate 14 **green** 16:23Z. Recovery **succeeded** 16:24–16:28Z: no restart, verifier + trust probe passed, activate → **gen 186**, c21 general. 164 controls, `sqlFailuresDelta` 0. | | +| **Roll 2 complete** 16:29Z | **All 19 general cells on `4916ed67`** (c7–c10, c13–c16, c19–c29); existing-only c1–c6, c11, c12 and migration-only c17, c18 untouched. Selector gen 148 → 186. Fleet 23 cells, 2 927 controls. Container exits 07:00–16:29Z: 20, every one a boot-race exit (<10 s, `cloud-sql-proxy` sidecar not yet listening), **0 serving-process exits**. Director on `00575-leq` (`4916ed67`) since 01:02Z: Postgres retries 0 in the last hour (13 over the first 6 h vs 85 on the predecessor), 5xx in the last hour 104 `/v1/assign` 503s (admission `Retry-After` path, at the pre-roll rate). Three waves needed the no-restart `mode=rollback` resume (c13: transient trust-probe 409; c26 and c21: post-apply `runtime-status` 503 `unconditional drop overload`), each recovered in ~4 min with no drain. 14 monitor gates, 14 green, 0 freezes. | | diff --git a/cloud/docs/relay-roll2-plan-2026-09.md b/cloud/docs/relay-roll2-plan-2026-09.md index f84de39163f..ab275039fe9 100644 --- a/cloud/docs/relay-roll2-plan-2026-09.md +++ b/cloud/docs/relay-roll2-plan-2026-09.md @@ -133,6 +133,11 @@ Record every gate and wave in the findings doc as in Roll 1. dropped. - **Monitor residuals** already in the checklist: `probeEndpointHealth` retry decision still uses the flat 2 000 ms bar; operator protocol unbound for Asia; `probe-relay-rehome-trust` regex. +- **Same-cap job residuals found in Roll 2** (three of eleven mutating runs needed the resume path): + the post-apply `admin_post target-runtime` read has no transient-5xx tolerance and failed twice on a + one-request 503 `unconditional drop overload` from the edge ~80 s after readiness (c26, c21); and + `probe-relay-rehome-trust` prints only the status on a 409, so the transient c13 failure left no + reason on record. Retry both once and print the error body. - Update the checklist status header; tick 2.3, 4.1, 4.3 relay-side as deployed. ## Deferred, owner decision required diff --git a/config/packaged-runtime-node-modules.cjs b/config/packaged-runtime-node-modules.cjs index 1eca37b7c05..1ee443f8288 100644 --- a/config/packaged-runtime-node-modules.cjs +++ b/config/packaged-runtime-node-modules.cjs @@ -14,6 +14,7 @@ const projectDir = resolve(__dirname, '..') const requireFromProject = createRequire(join(projectDir, 'package.json')) const PACKAGED_RUNTIME_PACKAGE_ROOTS = [ + '@anthropic-ai/claude-agent-sdk', '@electron-toolkit/utils', '@linear/sdk', '@parcel/watcher', @@ -56,6 +57,11 @@ const ELECTRON_ARCHITECTURE_BY_ENUM = { 4: 'universal' } const PACKAGED_NATIVE_ARCHITECTURES = new Set(['ia32', 'x64', 'arm', 'arm64']) +const PACKAGED_MAIN_REQUIRED_FILES = [ + 'out/main/index.js', + 'out/main/agent-hooks/managed-agent-hook-controls.js' +] +const PACKAGED_MAIN_SOURCE_RE = /^out\/main\/.+\.js$/ const TYPE_DECLARATION_ARTIFACT_RE = /\.d\.(?:c|m)?ts(?:\.map)?$/ const JS_SOURCE_MAP_ARTIFACT_RE = /\.(?:c|m)?js\.map$/ const VERSIONED_ONNXRUNTIME_DYLIB_RE = /^libonnxruntime\.\d[\d.]*\.dylib$/ @@ -223,22 +229,39 @@ function verifyPackagedMainRuntimeDeps(resourcesDir, asar = require('@electron/a return } - const mainFiles = ['out/main/index.js', 'out/main/agent-hooks/managed-agent-hook-controls.js'] const entries = asar.listPackage(asarPath) - const missing = new Set() - - for (const file of mainFiles) { - const entry = findAsarEntry(entries, file) - if (!entry) { + for (const file of PACKAGED_MAIN_REQUIRED_FILES) { + if (!findAsarEntry(entries, file)) { throw new Error(`Packaged main file ${file} was not found in ${asarPath}`) } + } + + const missing = new Set() + // Why every emitted main file rather than the entry points alone: rolldown hoists + // modules shared by two entries into out/main/chunks, so an entry's own bare imports + // move out from under a fixed file list and silently stop being checked. + for (const entry of entries) { + if (!PACKAGED_MAIN_SOURCE_RE.test(normalizeAsarEntryPath(entry))) { + continue + } // Why: @electron/asar lists entries with host separators; Windows returns // backslashes, and extractFile expects that same host-style path. const internalPath = entry.replace(/^[\\/]+/, '') const source = asar.extractFile(asarPath, internalPath).toString('utf8') - for (const match of source.matchAll(/require\(["']([^"']+)["']\)/g)) { - const specifier = match[1] + // Why the lookbehind: Orca has its own registry methods named `require`, so a + // minified `registry.require('some-id')` must not read as a bare specifier. + // Why it readmits `...`: a dot that ends a spread is not member access, and + // the two error directions are not symmetric -- a false positive fails the + // release build loudly, a false negative is this guard going blind. + // Known limit: a specifier inside an embedded source string counts too, and + // ssh-relay-deploy's remote probe names node-pty that way. A remote-only + // dependency added to that script would fail desktop packaging here; telling + // the two apart needs a parser, not a wider pattern. + for (const match of source.matchAll( + /(?:(?` host resolves on the desktop resolver outside the tunnel, and a source census keeps any DoH host-resolver mode from widening that leak. Network-service restart and provider journeys remain uncovered.", + "coveredPlatforms": ["macos", "linux"], + "coveredProviders": ["ssh", "remote-runtime"], + "coverageNotes": "Deterministic main-process tests cover versioned delimiter-safe aggregate partition derivation, path-safe opaque names, durable collision metadata, oversized or corrupt metadata refusal, missing-sidecar Chromium-data refusal, bounded binding/live-page admission, immediate SOCKS5 setup with Chromium loopback bypass disabled, inherited-connection closure, exact proxy verification before allowlisting, concurrent setup coalescing, live proxy-retarget refusal, token-safe page replacement, existing browser-profile policy installation, active Orca-profile storage scoping, blank-only initial attachment, arbitrary initial-navigation denial, and fail-closed per-guest WebRTC policy through delayed or failed cleanup. The production client-page executor prepares, registers, and grants the exact route page. A real Electron A/B capture proves HTTP, HTTPS, WebSocket, redirects, subresources, downloads, and a `.test` hostname traverse SOCKS with no direct target connection. A two-launch control proves immediate setProxy routes a forced persisted-worker wake and later worker fetch. A separate capture proves the protected guest sends zero direct STUN packets. A further capture proves non-WebRTC UDP is also contained: a WebTransport session and a fetch forced onto QUIC both reach the desktop directly in the control arm and emit zero datagrams through the route partition, and the shipped disable-features list hides the Direct Sockets constructors whose mere construction kills a control-arm renderer. DNS prefetch is a tripwire over an accepted residual rather than a guard: Electron 43 inherits Chromium's PrefetchDNS, so a `` host resolves on the desktop resolver outside the tunnel, and a source census keeps any DoH host-resolver mode from widening that leak. Network-service restart and WSL provider journeys remain uncovered. Four Linux Docker SSH browser baseline scenarios passed: direct-host routing, unavailable-host local escape, forwarding refusal, and paired client-hosted reconnect. All four scenarios subsequently passed three repetitions each (12 passes, no skips or retries) in Linux CI run 34040309638, with unchanged assertions and timeouts.", "motivatingLinks": [ - "https://linear.app/stably/issue/STA-4150/refactor-remote-browser-to-client-hosted-electron-webviews" + "https://linear.app/stably/issue/STA-4150/refactor-remote-browser-to-client-hosted-electron-webviews", + "https://github.com/stablyai/orca/actions/runs/34039986047", + "https://github.com/stablyai/orca/actions/runs/34040309638" ], "invariant": "A client-hosted partition is derived only in main from stable Orca-profile, browser-profile, authority-connection, and execution-host identities. Raw identities and individually linkable component hashes never enter its path-safe partition name. Durable binding metadata must match and precede Chromium partition data before reuse. One live partition never changes execution host or proxy endpoint. Fixed SOCKS5 setup starts immediately after Session creation, before policy installation can yield or a persisted worker is awakened; no partition enters the webview allowlist until browser policy is installed, inherited connections are closed, and resolveProxy returns exactly that one listener. Initial route-partition attachment is blank-only. Its exact WebContents is quarantined before applying non-proxied WebRTC denial and remains navigation- and popup-denied if policy application or cleanup fails. Distinct live partitions, retained logical page generations, durable bindings, and binding-file reads remain bounded. No UDP transport a route-partition page can reach — WebRTC, WebTransport, or forced QUIC — emits a datagram to the desktop, the Direct Sockets constructors stay absent from every guest so no page can kill its renderer, and the process never enables a DoH host-resolver mode.", "oracle": "Derive two delimiter-adversarial identities and require distinct full-digest path-safe partitions with no raw IDs or component hashes. Persist one binding, reload it, and reject replacement, malformed or oversized state, Chromium data without matching metadata, and the 513th binding. Prepare one partition and require setProxy with <-loopback> to be invoked immediately after getSession and before policy setup, then closeAllConnections and exact SOCKS5 resolveProxy while isAllowedPartition remains false; only then may it become live. Under real Electron, require direct controls for HTTP, HTTPS, WebSocket, redirects, subresources, and downloads, then require the fixed SOCKS session to route every equivalent request plus an otherwise-unresolvable `.test` hostname with zero direct target connections. Across two Electron launches, require immediate setProxy to route a forced worker wake and post-verification fetch. Reject DIRECT, endpoint retargeting, and capacity overflow. Require quarantine before disable_non_proxied_udp and admission; under real Electron require the unprotected control to emit STUN and the protected guest to emit zero direct UDP packets. Under real Electron require a direct control to emit WebTransport and forced-QUIC datagrams and the SOCKS partition to emit none, require an explicitly enabled Direct Sockets control to expose the constructors and die on construction, and require the shipped disable-features list to leave them undefined with the renderer alive. Capture a route partition's netLog across a dns-prefetch load and require the prefetched host to appear on a local resolver task while an unreferenced control host appears nowhere; require no source file to set a non-'off' secureDnsMode.", @@ -4609,7 +4714,9 @@ "pnpm exec vitest run --config config/vitest.config.ts src/main/browser/browser-route-webcontents-registry.test.ts src/main/browser/browser-route-webrtc-egress.electron.test.ts", "pnpm exec vitest run --config config/vitest.config.ts src/main/browser/browser-route-session-registry.test.ts src/main/browser/browser-route-persisted-worker-egress.electron.test.ts src/main/browser/browser-route-webrtc-egress.electron.test.ts", "pnpm exec vitest run --config config/vitest.config.ts src/main/browser/browser-route-tcp-egress.electron.test.ts", - "pnpm exec vitest run --config config/vitest.config.ts src/main/browser/browser-route-h3-egress.electron.test.ts src/main/browser/browser-route-dns-prefetch.electron.test.ts src/main/startup/secure-dns-census.test.ts" + "pnpm exec vitest run --config config/vitest.config.ts src/main/browser/browser-route-h3-egress.electron.test.ts src/main/browser/browser-route-dns-prefetch.electron.test.ts src/main/startup/secure-dns-census.test.ts", + "ORCA_E2E_SSH_DOCKER=1 ORCA_E2E_LOCAL_SSH_BROWSER=1 ORCA_E2E_SSH_CLIENT_HOSTED_BROWSER=1 ORCA_E2E_WEB_CLIENT=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/local-ssh-browser-routing.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1 --repeat-each=3", + "ORCA_E2E_SSH_DOCKER=1 ORCA_E2E_LOCAL_SSH_BROWSER=1 ORCA_E2E_SSH_CLIENT_HOSTED_BROWSER=1 ORCA_E2E_WEB_CLIENT=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1 --repeat-each=3" ], "testFiles": [ "src/main/browser/browser-route-identity.test.ts", @@ -4624,7 +4731,9 @@ "src/main/browser/browser-route-webcontents-registry.test.ts", "src/main/browser/browser-session-registry.test.ts", "src/main/browser/browser-session-startup.test.ts", - "src/main/window/createMainWindow.test.ts" + "src/main/window/createMainWindow.test.ts", + "tests/e2e/local-ssh-browser-routing.spec.ts", + "tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts" ], "assertionRefs": [ { @@ -4719,6 +4828,20 @@ "a live route partition may attach only the normalized blank document", "an arbitrary URL cannot be the initial route-partition document" ] + }, + { + "file": "tests/e2e/local-ssh-browser-routing.spec.ts", + "assertions": [ + "a remote-only origin renders through direct SSH routing; cookies survive transport recovery", + "unavailable SSH hosts prevent premature webview attachment and offer a working explicit local escape hatch", + "real AllowTcpForwarding refusal is classified and Try anyway preserves the SSH route" + ] + }, + { + "file": "tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts", + "assertions": [ + "paired client-hosted browser pages render an SSH-only origin, preserve cookies and retire superseded route pages across a real transport drop" + ] } ], "evidenceRuns": [ @@ -4767,15 +4890,33 @@ "result": "passed", "durationSeconds": 0.5, "summary": "Six files passed 150 opaque identity, durable collision binding, bounded partition/page, proxy-before-allowlist, policy reuse, profile startup, and blank-only attach tests." + }, + { + "date": "2026-09-06", + "runner": "ci", + "platform": "linux", + "result": "passed", + "command": "ORCA_E2E_SSH_DOCKER=1 ORCA_E2E_LOCAL_SSH_BROWSER=1 ORCA_E2E_SSH_CLIENT_HOSTED_BROWSER=1 ORCA_E2E_WEB_CLIENT=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/local-ssh-browser-routing.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1 --repeat-each=3", + "durationSeconds": 252, + "summary": "Nine direct SSH cases passed: three repetitions each of routing/reconnect, unavailable-host local escape, and real TCP-forwarding refusal. Run 34040309638, head 259a5f6; unchanged tests and timeouts, zero skips or retries." + }, + { + "date": "2026-09-06", + "runner": "ci", + "platform": "linux", + "result": "passed", + "command": "ORCA_E2E_SSH_DOCKER=1 ORCA_E2E_LOCAL_SSH_BROWSER=1 ORCA_E2E_SSH_CLIENT_HOSTED_BROWSER=1 ORCA_E2E_WEB_CLIENT=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1 --repeat-each=3", + "durationSeconds": 138, + "summary": "Three paired client-hosted reconnect cases passed with remote-only origin and cookie-preservation assertions. Run 34040309638, head 259a5f6; unchanged tests and timeouts, zero skips or retries." } ], "runtimeBudget": { "p95Seconds": 2, - "scope": "deterministic identity, binding-store, session-policy, and window-boundary tests" + "scope": "deterministic identity, binding-store, session-policy, and window-boundary tests; this unit-test budget excludes SSH browser journeys, whose CI p95 is not yet established" }, "flakeHistory": { "status": "unknown", - "evidence": "The deterministic suite passes locally; CI and real Electron soak history have not started." + "evidence": "The deterministic suite passes locally. Linux SSH provider journeys passed four baseline cases and twelve repeated cases in CI runs 34039986047 and 34040309638 with zero skips or retries. Long-term and cross-platform soak history remains incomplete." }, "redGreenEvidence": { "status": "partial", @@ -4801,7 +4942,8 @@ "Partition deletion, download/transfer draining, idle route release, disk quotas, and browser-profile cloning are later lifecycle stages.", "Binding writes serialize in Electron main, and packaged hosts rely on Orca's per-userData single-instance lock. Activation still needs an explicit guard for dev instances that share userData or a cross-process CAS/lock.", "Sequential proxy or policy setup failures retain durable bindings and can exhaust the 512-binding ledger. Activation requires bounded tombstone recovery and partition garbage collection.", - "Each preparePage synchronously reads and parses bounded binding metadata on Electron main; activation requires latency evidence or a safely invalidated cache before this becomes frequent." + "Each preparePage synchronously reads and parses bounded binding metadata on Electron main; activation requires latency evidence or a safely invalidated cache before this becomes frequent.", + "New SSH browser journey evidence is limited to Linux CI with Docker; native macOS/Windows clients and WSL providers remain unverified by these scenarios." ], "demotionRule": "Keep experimental or demote if raw identities enter a partition path, a durable binding mismatch is reused, a partition retargets to another execution host or live listener, a route partition becomes attachable before exact proxy verification, initial attachment can navigate beyond blank, stale cleanup retires a replacement, admission exceeds a declared cap, or any browser request reaches desktop DNS, TCP, UDP, localhost, or system proxy outside the selected route." }, @@ -5044,9 +5186,9 @@ ], "platforms": ["macos", "linux", "windows", "ios"], "providers": ["remote-runtime", "ssh", "wsl"], - "coveredPlatforms": ["macos", "ios"], + "coveredPlatforms": ["macos", "ios", "linux"], "coveredProviders": ["remote-runtime", "ssh", "wsl"], - "coverageNotes": "Fresh-build Playwright journeys run the same production store action against an isolated headed Electron server and a real headless orca serve host. They prove one immutable client placement owns one real retained guest on the viewing desktop, the server owns no duplicate guest, no screencast frame renders, browser.snapshot reaches the client guest, disabling the setting preserves that guest, and the next page uses the legacy server engine. Deterministic contracts cover omitted placement, missing capabilities, explicit server placement, exact renderer-store materialization after delayed publication, no fallback after client-create failure, folder workspaces, git worktrees, browserless hosts, native and WSL routes, exact connected SSH authority, reconnect command replay, lease replacement, imported-inventory cleanup, bounded retirement, and shared remote screencast fanout for multiple independent viewers. A published v1.4.184 package runs both skew directions: an old client omits placement against the current host, while a current client capability-downgrades against the old host; each creates one server guest, no client guest, and returns the exact snapshot marker. A current iOS Simulator client paired to that legacy packaged host visibly loads Example Domain through the preserved server-hosted surface. A Docker OpenSSH target proves container-only DNS and localhost through both ssh2 and system-SSH routes. Physical Windows/Linux Electron and physical mobile journeys remain gaps.", + "coverageNotes": "Fresh-build Playwright journeys run the same production store action against an isolated headed Electron server and a real headless orca serve host. They prove one immutable client placement owns one real retained guest on the viewing desktop, the server owns no duplicate guest, no screencast frame renders, browser.snapshot reaches the client guest, disabling the setting preserves that guest, and the next page uses the legacy server engine. Deterministic contracts cover omitted placement, missing capabilities, explicit server placement, exact renderer-store materialization after delayed publication, no fallback after client-create failure, folder workspaces, git worktrees, browserless hosts, native and WSL routes, exact connected SSH authority, reconnect command replay, lease replacement, imported-inventory cleanup, bounded retirement, and shared remote screencast fanout for multiple independent viewers. A published v1.4.184 package runs both skew directions: an old client omits placement against the current host, while a current client capability-downgrades against the old host; each creates one server guest, no client guest, and returns the exact snapshot marker. A current iOS Simulator client paired to that legacy packaged host visibly loads Example Domain through the preserved server-hosted surface. A Docker OpenSSH target proves container-only DNS and localhost through both ssh2 and system-SSH routes. Physical Windows/Linux Electron and physical mobile journeys remain gaps. The two Docker remote-only SSH browser routing journeys now run in the dedicated Linux ssh-browser-network-route CI job on full runs and their mapped source/test changes; 2 baseline and 6 repeated cases passed with no skips/retries on 2026-09-06.", "motivatingLinks": [ "https://linear.app/stably/issue/STA-4150/refactor-remote-browser-to-client-hosted-electron-webviews" ], @@ -5061,7 +5203,8 @@ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/browser-network-tunnel-paired-runtime.integration.test.ts src/main/browser/paired-runtime-browser-network-route.test.ts src/main/browser/browser-network-execution-route.test.ts src/main/browser/wsl-browser-network-execution-route.test.ts src/main/browser/wsl-browser-network-relay-launch.test.ts src/main/runtime/runtime-browser-network-execution-host.test.ts", "pnpm exec vitest run --config config/vitest.config.ts src/main/browser/browser-screencast-lifecycle.test.ts src/main/browser/browser-screencast-stream.test.ts src/main/runtime/orca-runtime-browser-screencast-fanout.test.ts", "ORCA_RUN_DOCKER_SSH_BROWSER_E2E=1 pnpm exec vitest run --config config/vitest.config.ts tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts", - "Manual iOS 26.5 simulator: pair current mobile code to packaged Orca 1.4.184; create Browser; navigate to https://example.com; require one visible Example Domain tab on the server-hosted surface" + "Manual iOS 26.5 simulator: pair current mobile code to packaged Orca 1.4.184; create Browser; navigate to https://example.com; require one visible Example Domain tab on the server-hosted surface", + "ORCA_RUN_DOCKER_SSH_BROWSER_E2E=1 node_modules/.bin/vitest run --config config/vitest.config.ts tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts" ], "testFiles": [ "tests/e2e/paired-client-hosted-browser.spec.ts", @@ -5197,6 +5340,15 @@ "result": "passed", "durationSeconds": 1.2, "summary": "22 screencast lifecycle, stream, and shared-fanout tests passed; the suite confirms one physical CDP stream fans out independently to multiple viewers, preserves viewport ownership, and cleans up without cross-viewer eviction." + }, + { + "date": "2026-09-06", + "runner": "ci", + "platform": "linux", + "command": "ORCA_RUN_DOCKER_SSH_BROWSER_E2E=1 node_modules/.bin/vitest run --config config/vitest.config.ts tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts", + "result": "passed", + "durationSeconds": 28.29, + "summary": "Previously excluded Docker SSH2 and system-OpenSSH remote-only domain journeys: 2/2 baseline cases (34043512327) plus 6/6 across three independent CI jobs (34043659504), zero skips/retries. Dedicated ssh-browser-network-route job now executes them for full E2E runs and matched source/test edits; preserves all original route and authority assertions." } ], "runtimeBudget": { @@ -18035,19 +18187,24 @@ ], "platforms": ["macos", "linux", "windows"], "providers": ["ssh"], - "coveredPlatforms": ["macos"], + "coveredPlatforms": ["macos", "linux"], "coveredProviders": ["ssh"], - "coverageNotes": "A macOS Electron client drives a Linux Docker SSH execution host. The six-spec suite passed ten enabled cases with clean worker exit (5.2m). The formerly skipped frozen-host input case now waits for recovered authority before sending input and passed four separate executions (one initial and three repetitions). The existing flooded-shell fixme remains an explicitly reproduced application gap.", + "coverageNotes": "A macOS Electron client drives a Linux Docker SSH execution host. The six-spec suite passed ten enabled cases with clean worker exit (5.2m). The formerly skipped frozen-host input case now waits for recovered authority before sending input and passed four separate executions (one initial and three repetitions). The existing flooded-shell fixme remains an explicitly reproduced application gap. The bulk-open freeze reproduction runs in Linux headed CI with SwiftShader on Xvfb: headless Linux schedules idle animation frames about 1s apart, invalidating the foreground interaction measurement. Original uninstrumented five-pane workload passed all ten repetitions with zero retries/skips in 6.6m; bulk-open lag 79.3–147.8ms and interaction 127.1–155.9ms, unchanged 2500ms/5000ms budgets. Run 34037669843, head f25eab3fd7d723509ced026633f80b193a139b76, excludes unmerged replay-input application fix #19075.", "motivatingLinks": [ "https://github.com/stablyai/orca/issues/18018", "https://github.com/stablyai/orca/pull/18546", - "https://github.com/stablyai/orca/issues/12547" + "https://github.com/stablyai/orca/issues/12547", + "https://github.com/stablyai/orca/issues/16764", + "https://github.com/stablyai/orca/actions/runs/34037450427", + "https://github.com/stablyai/orca/actions/runs/34037669843" ], "invariant": "Transport loss and frozen-host silence must preserve the remote session; host relay loss may rebind a pane without accumulating reattachable leases. Reconnects must preserve usable terminal content, bounded PTYs/fds/processes, complete large listings, and independently recoverable watcher processes. Electron test shutdown must release inherited pipes after confirmed root exit without closing live-process pipes.", "oracle": "Poll a changed connected SSH authority after injected faults, then require terminal output and appropriate PTY identity. Read remote process/fd state, listFiles replies, and rendered explorer rows. Resolve Playwright cleanup only after the root process exits and its inherited pipes close; live-process pipes remain untouched.", "commands": [ "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-transport-drop-recovery.spec.ts tests/e2e/ssh-docker-half-open-link.spec.ts tests/e2e/ssh-docker-quick-open-large-listing.spec.ts tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts tests/e2e/ssh-docker-resource-accumulation.spec.ts tests/e2e/ssh-docker-watcher-isolation.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1", - "pnpm exec vitest run --config config/vitest.config.ts tests/e2e/helpers/electron-process-shutdown.unit.test.ts" + "pnpm exec vitest run --config config/vitest.config.ts tests/e2e/helpers/electron-process-shutdown.unit.test.ts", + "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts --config tests/playwright.config.ts --project=electron-headful --workers=1", + "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts --config tests/playwright.config.ts --project=electron-headful --workers=1 --repeat-each=10" ], "testFiles": [ "tests/e2e/ssh-docker-transport-drop-recovery.spec.ts", @@ -18056,7 +18213,8 @@ "tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts", "tests/e2e/ssh-docker-resource-accumulation.spec.ts", "tests/e2e/ssh-docker-watcher-isolation.spec.ts", - "tests/e2e/helpers/electron-process-shutdown.unit.test.ts" + "tests/e2e/helpers/electron-process-shutdown.unit.test.ts", + "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts" ], "assertionRefs": [ { @@ -18101,6 +18259,12 @@ "releases inherited pipes after confirmed exit, including prior exit", "retains live-process pipes on shutdown timeout" ] + }, + { + "file": "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts", + "assertions": [ + "five flooding SSH panes remain below unchanged 2500ms soft and 5000ms hard freeze budgets during bulk reopen and two double-animation-frame view changes" + ] } ], "evidenceRuns": [ @@ -18121,6 +18285,15 @@ "command": "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-transport-drop-recovery.spec.ts tests/e2e/ssh-docker-half-open-link.spec.ts tests/e2e/ssh-docker-quick-open-large-listing.spec.ts tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts tests/e2e/ssh-docker-resource-accumulation.spec.ts tests/e2e/ssh-docker-watcher-isolation.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1", "durationSeconds": 312, "summary": "Six specs: ten passed, two existing fixme skipped, clean worker shutdown. Baseline same enabled suite: ten passed but worker teardown timed out (7.3m)." + }, + { + "date": "2026-09-06", + "runner": "ci", + "platform": "linux", + "result": "passed", + "command": "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts --config tests/playwright.config.ts --project=electron-headful --workers=1 --repeat-each=10", + "durationSeconds": 396, + "summary": "Original uninstrumented five-pane workload passed all ten repetitions with zero retries/skips in 6.6m; bulk-open lag 79.3–147.8ms and interaction 127.1–155.9ms, unchanged 2500ms/5000ms budgets. Run 34037669843, head f25eab3fd7d723509ced026633f80b193a139b76, excludes unmerged replay-input application fix #19075." } ], "runtimeBudget": { @@ -18146,11 +18319,96 @@ ], "knownGaps": [ "The disconnected 48MB flood still loses its relay channel: original post-flood input marker failed in 60s, and waiting for the finite producer completion marker failed in 120s. It remains an explicit #18018 fixme reproduction; frozen-host input is re-enabled after four successful runs.", - "Linux and Windows desktop clients, WSL, folder workspaces, paired runtimes and live agent CLIs are not exercised by these Docker specs.", + "Linux headed CI covers the bulk-open freeze reproduction; Windows clients, WSL, folder workspaces, paired runtimes and live agent CLIs are not covered by that result.", "Some legacy assertions inspect terminal serialization or backing state rather than rendered DOM; no blanket visual coverage claim.", - "No p95 CI history or full product mutation proof." + "No p95 CI history or full product mutation proof.", + "One headless bulk-open probe reached 6478.6ms in run 34035957303; animation-frame scheduling explains the consistent interaction failures, but does not directly explain that isolated timer-lag outlier. Long-term headed CI soak remains outstanding." ], "demotionRule": "Keep experimental while any recovery reproduction fails or any teardown, identity, resource-count, or rendered oracle flakes; never promote by extending sleeps or retries." + }, + { + "id": "terminal.windows-wsl-launch-and-paste", + "title": "Real WSL terminal agent launch and paste ownership", + "maturity": "experimental", + "protection": "partial", + "owner": "terminal-runtime", + "layer": "electron-windows-wsl", + "surfaces": ["agent tab launch", "keyboard paste", "terminal runtime retention"], + "platforms": ["windows"], + "providers": ["wsl1", "wsl2"], + "coveredPlatforms": ["windows"], + "coveredProviders": ["wsl1"], + "coverageNotes": "Real WSL1 coverage: three scenarios each passed three times with no skips or retries; exact JSON report verified. Latest PR routing and installer-checksum follow-ups await CI. WSL2 remains untested.", + "motivatingLinks": ["https://github.com/stablyai/orca/actions/runs/34030832614"], + "invariant": "An agent launched into WSL runs in the guest; keyboard paste reaches exactly one owning PTY and preserves Linux content even after the default shell changes.", + "oracle": "Run the existing real WSL launch and two paste cases three times; require nine passes and zero skipped, unexpected, or flaky results in the Playwright JSON report.", + "commands": [ + "gh workflow run windows-wsl-e2e.yml", + "pnpm exec playwright test tests/e2e/golden-tab-bar-agent-launch.spec.ts tests/e2e/terminal-windows-shell-paste-ownership.spec.ts --config tests/playwright.config.ts --project=electron-headless --workers=1", + "node_modules/.bin/vitest run --config config/vitest.config.ts config/scripts/wsl-e2e-lane-contract.test.mjs config/scripts/verify-wsl-e2e-participation.test.mjs", + "gh run view 34031806291 --log" + ], + "testFiles": [ + "tests/e2e/golden-tab-bar-agent-launch.spec.ts", + "tests/e2e/terminal-windows-shell-paste-ownership.spec.ts", + "config/scripts/wsl-e2e-lane-contract.test.mjs", + "config/scripts/verify-wsl-e2e-participation.test.mjs" + ], + "assertionRefs": [ + { + "file": "tests/e2e/golden-tab-bar-agent-launch.spec.ts", + "assertions": ["requires a distro-only marker from the launched agent"] + }, + { + "file": "tests/e2e/terminal-windows-shell-paste-ownership.spec.ts", + "assertions": [ + "requires exact Linux pasted content and exactly one PTY write", + "retains WSL paste ownership after changing the default shell" + ] + }, + { + "file": "config/scripts/verify-wsl-e2e-participation.test.mjs", + "assertions": ["rejects skipped, missing, substituted and retried scenarios"] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-06", + "runner": "ci", + "platform": "windows", + "result": "passed", + "command": "gh run view 34031806291 --log", + "durationSeconds": 210, + "summary": "Immutable run34031806291 at92fc5152: WSL1 launch3 and paste6 passed after reader-readiness correction; named-scenario verifier accepted actual JSON report with0skips0retries. Command retrieves recorded evidence; workflow_dispatch command above reruns current coverage." + } + ], + "runtimeBudget": { + "p95Seconds": 1800, + "scope": "CI job timeout; measured p95 is not established" + }, + "flakeHistory": { + "status": "soaking", + "evidence": "Initial permanent-lane diagnostic8passed1failed on missing PTY before changing settings. After requiring guest-reader readiness before mutation, run34031806291 passed9/9. Two earlier setup validations also passed9/9. Long-term CI history remains missing." + }, + "redGreenEvidence": { + "status": "partial", + "evidence": "Verifier rejects actual8pass1fail CI report and accepts actual9pass report. Unit contracts reject skips, missing or substituted scenarios and retried passes. No full application fault-mutation proof." + }, + "performanceBudget": { + "required": false, + "evidence": "CI-only provisioning and routing; no application runtime changes." + }, + "promotionCriteria": [ + "Require all nine real WSL executions on the final workflow head.", + "Demonstrate missing or skipped WSL execution fails participation.", + "Collect repeated CI history before adding this experimental lane to required verification." + ], + "knownGaps": [ + "WSL2 is not provisioned.", + "No SSH, folder-only workspace, packaged mixed-version, or live-service claim.", + "The new PR lane is outside verify until reliability is established." + ], + "demotionRule": "Keep experimental if provisioning or an execution flakes; never promote by skipping a case, raising timeouts, or retrying until green." } ] } diff --git a/config/scripts/electron-builder-runtime-resources.test.mjs b/config/scripts/electron-builder-runtime-resources.test.mjs index 453d5702cb0..5a93ec12c25 100644 --- a/config/scripts/electron-builder-runtime-resources.test.mjs +++ b/config/scripts/electron-builder-runtime-resources.test.mjs @@ -44,6 +44,135 @@ describe('packaged runtime resources', () => { } }) + it('verifies literal dynamic imports from the packaged main bundle', async () => { + const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-runtime-dynamic-imports-')) + try { + await writeFile(join(resourcesDir, 'app.asar'), '', 'utf8') + + // The first is the exact shape oxc emits for the memoized SDK import in a + // shipped build; the second is the spaced variant the pattern also accepts. + const sources = new Map([ + [ + 'out/main/index.js', + 'let p=null;function q(){return p??=import(`@anthropic-ai/claude-agent-sdk`),p}' + ], + [ + 'out/main/agent-hooks/managed-agent-hook-controls.js', + 'import (`@anthropic-ai/claude-agent-sdk`)' + ] + ]) + const asar = { + listPackage: () => [...sources.keys()].map((entry) => `/${entry}`), + extractFile: (_asarPath, internalPath) => Buffer.from(sources.get(internalPath), 'utf8') + } + + expect(() => verifyPackagedMainRuntimeDeps(resourcesDir, asar)).toThrow( + /@anthropic-ai\/claude-agent-sdk/ + ) + + await mkdir(join(resourcesDir, 'node_modules', '@anthropic-ai', 'claude-agent-sdk'), { + recursive: true + }) + expect(() => verifyPackagedMainRuntimeDeps(resourcesDir, asar)).not.toThrow() + } finally { + await rm(resourcesDir, { recursive: true, force: true }) + } + }) + + it('still fails when a required packaged main entry is missing entirely', async () => { + const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-runtime-missing-entry-')) + try { + await writeFile(join(resourcesDir, 'app.asar'), '', 'utf8') + + const asar = { + listPackage: () => ['/out/main/index.js'], + extractFile: () => Buffer.from('', 'utf8') + } + + expect(() => verifyPackagedMainRuntimeDeps(resourcesDir, asar)).toThrow( + /managed-agent-hook-controls\.js was not found/ + ) + } finally { + await rm(resourcesDir, { recursive: true, force: true }) + } + }) + + it('verifies bare imports that rolldown hoisted into a shared main chunk', async () => { + const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-runtime-chunk-imports-')) + try { + await writeFile(join(resourcesDir, 'app.asar'), '', 'utf8') + + // The entry points themselves carry no specifier; only the shared chunk does. + const sources = new Map([ + ['out/main/index.js', ''], + ['out/main/agent-hooks/managed-agent-hook-controls.js', ''], + ['out/main/chunks/managed-agent-hook-controls-CWf8D-KR.js', 'require(`jsonc-parser`)'] + ]) + // Real listPackage emits directory nodes too, and extractFile throws on them, + // so the `.js` anchor is load-bearing -- keep the mock able to catch that. + const directories = ['/out', '/out/main', '/out/main/chunks'] + const asar = { + listPackage: () => [...directories, ...[...sources.keys()].map((entry) => `/${entry}`)], + extractFile: (_asarPath, internalPath) => { + const source = sources.get(internalPath) + if (source === undefined) { + throw new Error(`Expected to find file at: ${internalPath} but found a directory`) + } + return Buffer.from(source, 'utf8') + } + } + + expect(() => verifyPackagedMainRuntimeDeps(resourcesDir, asar)).toThrow(/jsonc-parser/) + + await mkdir(join(resourcesDir, 'node_modules', 'jsonc-parser'), { recursive: true }) + expect(() => verifyPackagedMainRuntimeDeps(resourcesDir, asar)).not.toThrow() + } finally { + await rm(resourcesDir, { recursive: true, force: true }) + } + }) + + it('reads a spread require, whose leading dots are not member access', async () => { + const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-runtime-spread-require-')) + try { + await writeFile(join(resourcesDir, 'app.asar'), '', 'utf8') + + const sources = new Map([ + ['out/main/index.js', 'const all=[...require("jsonc-parser")]'], + ['out/main/agent-hooks/managed-agent-hook-controls.js', ''] + ]) + const asar = { + listPackage: () => [...sources.keys()].map((entry) => `/${entry}`), + extractFile: (_asarPath, internalPath) => Buffer.from(sources.get(internalPath), 'utf8') + } + + expect(() => verifyPackagedMainRuntimeDeps(resourcesDir, asar)).toThrow(/jsonc-parser/) + } finally { + await rm(resourcesDir, { recursive: true, force: true }) + } + }) + + it('ignores member calls onto Orca methods that are themselves named require', async () => { + const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-runtime-member-require-')) + try { + await writeFile(join(resourcesDir, 'app.asar'), '', 'utf8') + + // electron-sidecar-tab-registry and browser-execution-host-grant-registry both + // expose require(key); a literal key must never read as a packaged specifier. + const sources = new Map([ + ['out/main/index.js', 'registry.require("public-a");grants.require(`host-key`)'], + ['out/main/agent-hooks/managed-agent-hook-controls.js', 'state.import("android-sdk")'] + ]) + const asar = { + listPackage: () => [...sources.keys()].map((entry) => `/${entry}`), + extractFile: (_asarPath, internalPath) => Buffer.from(sources.get(internalPath), 'utf8') + } + + expect(() => verifyPackagedMainRuntimeDeps(resourcesDir, asar)).not.toThrow() + } finally { + await rm(resourcesDir, { recursive: true, force: true }) + } + }) + it('normalizes host-specific asar entry separators', () => { expect(findAsarEntry(['\\out\\main\\index.js'], 'out/main/index.js')).toBe( '\\out\\main\\index.js' @@ -134,6 +263,15 @@ describe('packaged runtime resources', () => { expect(packagedTargets).toContain(join('node_modules', 'proper-lockfile')) }) + it('includes the Claude agent SDK in every desktop package plan', () => { + for (const platform of ['darwin', 'linux', 'win32']) { + const packagedTargets = createPackagedRuntimeNodeModuleResources(platform).map( + (resource) => resource.to + ) + expect(packagedTargets).toContain(join('node_modules', '@anthropic-ai', 'claude-agent-sdk')) + } + }) + it('prunes non-target @parcel/watcher architecture subpackages', async () => { const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-parcel-watcher-prune-')) try { diff --git a/config/scripts/pr-e2e-gate-contract.test.mjs b/config/scripts/pr-e2e-gate-contract.test.mjs index 41f9338ab75..f5295faf1ad 100644 --- a/config/scripts/pr-e2e-gate-contract.test.mjs +++ b/config/scripts/pr-e2e-gate-contract.test.mjs @@ -168,6 +168,7 @@ describe('PR E2E gate contract', () => { expect(changedRun.env.TEST_FILES_JSON).toBe('${{ inputs.test_files }}') expect(changedRun.run).toContain('. != "tests/e2e/ssh-startup-exec-readiness.spec.ts"') expect(changedRun.run).toContain('. != "tests/e2e/paired-startup-exec-readiness.spec.ts"') + expect(changedRun.run).toContain('. != "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts"') expect(changedRun.run).toContain('if [ "${#TEST_FILES[@]}" -eq 0 ]') expect(changedRun.run).toContain('grep -l \'@headful\' "${TEST_FILES[@]}"') expect(changedRun.run).toContain('E2E_PROJECT_ARGS+=(--project=electron-headful)') @@ -379,8 +380,7 @@ describe('PR E2E gate contract', () => { // run-ssh-docker-e2e.mjs so the gap stays legible rather than looking like coverage. const unreachableSpecs = new Set([ 'tests/e2e/ssh-docker-relay-perf.spec.ts', - 'tests/e2e/ssh-codex-display-artifacts-repro.spec.ts', - 'tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts' + 'tests/e2e/ssh-codex-display-artifacts-repro.spec.ts' ]) // Why comments are stripped: this file's own runner lists the two exempt specs by name in a // prose comment. A substring scan over raw text would count any spec merely *discussed* in a diff --git a/config/scripts/pr-e2e-source-routing.mjs b/config/scripts/pr-e2e-source-routing.mjs index 5b698fb0b42..18c6b032788 100644 --- a/config/scripts/pr-e2e-source-routing.mjs +++ b/config/scripts/pr-e2e-source-routing.mjs @@ -13,6 +13,38 @@ const NATIVE_IME_HARNESS = /^(?:config\/scripts\/(?:run-terminal-ibus-hangul-e2e|terminal-ime-engagement-receipt)\.mjs$|tests\/e2e\/terminal-ime-(?:boundary-probe|byte-reader|engagement-receipt)\.ts$|tests\/e2e\/terminal-(?:ibus-hangul|hangul-terminating-digit|macos-2set-korean)-native\.spec\.ts$)/ export const PR_E2E_SOURCE_ROUTES = [ + { + id: 'ssh.localhost-agent-hooks', + specs: ['tests/e2e/ssh-localhost.spec.ts'], + matches: (file) => + isProductSource(file) && + /^src\/(?:relay\/(?:agent-hook|relay-agent-hook-runtime|plugin-overlay)|main\/(?:agent-hooks\/|ssh\/ssh-relay-session\.ts$)|shared\/agent-hook)/.test( + file + ) + }, + { + id: 'browser-network.ssh-docker-route', + specs: ['tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts'], + matches: (file) => + file === 'tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts' || + /^tests\/e2e\/helpers\/docker-ssh-relay-(?:image|target)\.ts$/.test(file) || + (isProductSource(file) && + /^src\/main\/(?:browser\/(?:ssh-browser-network-execution-route|browser-network-deferred-socket|browser-network-execution-route|system-ssh-socks-client-socket)|ssh\/system-ssh-dynamic-forward-process)\.ts$/.test( + file + )) + }, + { + id: 'terminal.windows-wsl-launch-and-paste', + specs: [ + 'tests/e2e/golden-tab-bar-agent-launch.spec.ts', + 'tests/e2e/terminal-windows-shell-paste-ownership.spec.ts' + ], + matches: (file) => + isProductSource(file) && + /^(?:config\/scripts\/verify-wsl-e2e-participation\.mjs$|src\/main\/(?:wsl[/-]|pty\/.*wsl|providers\/wsl)|src\/shared\/(?:wsl-|windows-terminal-shell)|src\/renderer\/src\/.*(?:terminal-paste|pty-paste)|tests\/e2e\/(?:golden-tab-bar-agent-launch\.spec|terminal-windows-shell-paste-ownership\.spec|helpers\/(?:wsl-golden-stub-agent|golden-stub-agent))|\.github\/(?:actions\/setup-wsl-test-runtime\/|workflows\/windows-wsl-e2e\.yml))/.test( + file + ) + }, { id: 'ephemeral-vm-runtime.rollback-readable-sidecar', specs: ['tests/e2e/ephemeral-vm-provisioned-root.spec.ts'], @@ -227,6 +259,13 @@ export function shouldRunReusablePrE2e(changedPaths) { ) } +export function hasWslSourceChange(changedPaths) { + const route = PR_E2E_SOURCE_ROUTES.find( + (candidate) => candidate.id === 'terminal.windows-wsl-launch-and-paste' + ) + return changedPaths.some(route.matches) +} + if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { let input = '' process.stdin.setEncoding('utf8') @@ -238,6 +277,8 @@ if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) process.stdout.write(`${hasSshSourceChange(changedPaths)}\n`) } else if (process.argv.includes('--reusable-workflow')) { process.stdout.write(`${shouldRunReusablePrE2e(changedPaths)}\n`) + } else if (process.argv.includes('--wsl-source')) { + process.stdout.write(`${hasWslSourceChange(changedPaths)}\n`) } else if (process.argv.includes('--native-ime-source')) { process.stdout.write(`${hasNativeImeSourceChange(changedPaths)}\n`) } else { diff --git a/config/scripts/release-cut-token-permissions.test.mjs b/config/scripts/release-cut-token-permissions.test.mjs index f2f544a8f27..0fc1e5f8448 100644 --- a/config/scripts/release-cut-token-permissions.test.mjs +++ b/config/scripts/release-cut-token-permissions.test.mjs @@ -12,6 +12,8 @@ const EXPECTED_MATRIX = { '.github/workflows/e2e.yml#changed-e2e': { contents: 'read' }, '.github/workflows/e2e.yml#e2e': { contents: 'read' }, '.github/workflows/e2e.yml#prepare-native-cache': { contents: 'read' }, + '.github/workflows/e2e.yml#ssh-browser-network-route': { contents: 'read' }, + '.github/workflows/e2e.yml#ssh-localhost': { contents: 'read' }, '.github/workflows/e2e.yml#ssh-docker-watcher-isolation': { contents: 'read' }, '.github/workflows/homebrew-bump.yml#bump-cask': { contents: 'read' }, '.github/workflows/release-mac-build.yml#build-mac': { contents: 'write' }, diff --git a/config/scripts/run-ssh-docker-bulk-open-freeze-e2e.mjs b/config/scripts/run-ssh-docker-bulk-open-freeze-e2e.mjs index 153f3fb5bc6..294bf7e2c7b 100644 --- a/config/scripts/run-ssh-docker-bulk-open-freeze-e2e.mjs +++ b/config/scripts/run-ssh-docker-bulk-open-freeze-e2e.mjs @@ -29,7 +29,7 @@ const result = spawnSync( '--config', 'tests/playwright.config.ts', '--project', - 'electron-headless', + 'electron-headful', '--workers=1', ...extraArgs ], diff --git a/config/scripts/run-ssh-docker-e2e.mjs b/config/scripts/run-ssh-docker-e2e.mjs index 9ab44b8457e..b88bde609bb 100644 --- a/config/scripts/run-ssh-docker-e2e.mjs +++ b/config/scripts/run-ssh-docker-e2e.mjs @@ -6,6 +6,8 @@ const pnpm = process.platform === 'win32' ? 'pnpm.cmd' : 'pnpm' const env = { ...process.env, ORCA_E2E_SSH_DOCKER: '1', + ORCA_E2E_LOCAL_SSH_BROWSER: '1', + ORCA_E2E_SSH_CLIENT_HOSTED_BROWSER: '1', ORCA_E2E_WEB_CLIENT: '1' } @@ -33,31 +35,8 @@ if (runtime.status !== 0) { // all. Recorded as a real gap, not as coverage living somewhere else. // ssh-codex-display-artifacts-repro.spec.ts — installs a real remote codex binary that CI // runners do not have (observed as `spawn codex ENOENT`). Runs in no CI lane at all. -// ssh-docker-bulk-open-freeze-repro.spec.ts — un-rotted and now measurable, and marked -// `test.fixme` because its oracle cannot gate. Absent from this list AND skipped, so the -// two cannot drift: it is also reachable from the changed-specs lane whenever the spec -// itself is edited, and a wall-clock oracle that fails there is worth no more than one -// that fails here. -// The rot (#16764) is fixed: the stale call sites are repaired, it connects after session -// restore instead of before, and readiness keys on the repeating flood marker rather than -// a one-shot READY line the flood buries within ~16ms. It runs end to end and prints a -// measurement instead of dying on a call site. -// What it is NOT is portable. Three runs of the same measurement path: -// developer workstation: hiddenFlood 2.1ms bulkOpen 41.5ms interaction 53.6ms -// GitHub ubuntu runner A: hiddenFlood 1.5ms bulkOpen 2575.6ms interaction 3464.2ms -// GitHub ubuntu runner B: hiddenFlood 0.2ms bulkOpen 397.4ms interaction 3386.7ms -// bulkOpen swings 6.5x between two CI runs of the same code, so a fixed threshold on it is -// a coin flip; interaction sits stably ~64x over the workstation figure because it times a -// view remount, not the renderer freeze the issue reports, and only shares the budget -// constant because both are milliseconds. Every failure so far is the soft budget; hard -// has never tripped, and the relay was still streaming each time — the budget failed, not -// the product. Same rule as ssh-docker-relay-perf above. Gating needs a distribution -// first, then a host-relative oracle; a bigger constant, or a ratio picked from three -// samples, is the same arbitrary number in different clothes. -// COVERAGE GAP, recorded as such: 5 simultaneously flooding SSH panes exercise writer -// saturation, ACK/credit accounting and per-pane polling together, and nothing else covers -// that combination. Flip `test.fixme` back to `test` to run it. Tracked in -// stablyai/orca#16764. +// The bulk-open frame probe runs headed: headless Linux compositing schedules idle RAFs +// roughly 1s apart, so it cannot measure foreground interaction against the same budget. // // Why both projects: ssh-port-forward-lifecycle is @headful, which the headless project // grep-inverts away. @@ -69,24 +48,23 @@ if (runtime.status !== 0) { // - E2E does not gate merges: `verify.needs` in pr.yml omits `e2e` while the suite is red on // main. Nothing in this lane blocks a PR yet. pr.yml's Require-successful-checks comment // has the exact wiring to flip it, and the gate contract asserts the current state. -// - Five specs and one unit test are gated on env vars no workflow sets, so they run nowhere +// - Two specs are gated on env vars no workflow sets, so they run nowhere // and are not Docker-gated, which puts them outside this file's contract: -// local-ssh-browser-routing (ORCA_E2E_LOCAL_SSH_BROWSER) -// ssh-client-hosted-browser-drop-reconnect (ORCA_E2E_SSH_CLIENT_HOSTED_BROWSER) // nested-runtime-ssh-lifecycle, nested-runtime-ssh-routing (ORCA_E2E_NESTED_RUNTIME_SSH) -// ssh-localhost (ORCA_E2E_SSH_LOCALHOST) -// ssh-browser-network-execution-route.docker.unit.test.ts (ORCA_RUN_DOCKER_SSH_BROWSER_E2E) -// Runner scripts for the first four sit unused in package.json; no workflow calls them. +// The nested-runtime runner remains unused by CI. const result = spawnSync( pnpm, [ 'exec', 'playwright', 'test', + 'tests/e2e/local-ssh-browser-routing.spec.ts', + 'tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts', 'tests/e2e/pty-input-write-queue-ssh.spec.ts', 'tests/e2e/ssh-ai-vault-session-history.spec.ts', 'tests/e2e/ssh-cold-activation-restore.spec.ts', 'tests/e2e/ssh-cold-hydration-gap-tab-seeding.spec.ts', + 'tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts', 'tests/e2e/ssh-docker-half-open-link.spec.ts', 'tests/e2e/ssh-docker-quick-open-large-listing.spec.ts', 'tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts', diff --git a/config/scripts/ssh-browser-e2e-routing.test.mjs b/config/scripts/ssh-browser-e2e-routing.test.mjs new file mode 100644 index 00000000000..0b46131cf14 --- /dev/null +++ b/config/scripts/ssh-browser-e2e-routing.test.mjs @@ -0,0 +1,64 @@ +import { readFileSync } from 'node:fs' +import { join, resolve } from 'node:path' +import { parse } from 'yaml' +import { expect, it } from 'vitest' +import { selectPrE2eSpecs } from './pr-e2e-source-routing.mjs' + +const root = resolve(import.meta.dirname, '../..') +const workflow = parse(readFileSync(join(root, '.github/workflows/e2e.yml'), 'utf8')) +const runner = readFileSync(join(root, 'config/scripts/run-ssh-docker-e2e.mjs'), 'utf8') + +it('routes SSH browser specs to a lane that enables their opt-ins', () => { + const changedRun = workflow.jobs['changed-e2e'].steps.find( + (step) => step.name === 'Run changed E2E specs' + ) + for (const [spec, flag] of [ + ['tests/e2e/local-ssh-browser-routing.spec.ts', 'ORCA_E2E_LOCAL_SSH_BROWSER'], + [ + 'tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts', + 'ORCA_E2E_SSH_CLIENT_HOSTED_BROWSER' + ] + ]) { + expect(runner).toContain(`'${spec}'`) + expect(runner).toContain(`${flag}: '1'`) + expect(workflow.jobs['ssh-docker-watcher-isolation'].if).toContain(spec) + expect(changedRun.run).toContain(`. != "${spec}"`) + } +}) + +it('executes both Docker network routes in a Node job with their opt-in enabled', () => { + const spec = 'tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts' + const job = workflow.jobs['ssh-browser-network-route'] + const install = job.steps.find( + (step) => step.uses === './.github/actions/install-node-dependencies' + ) + const run = job.steps.find( + (step) => step.name === 'Run Docker SSH browser network route journeys' + ) + expect(job['runs-on']).toBe('ubuntu-latest') + expect(job.if).toContain("inputs.test_files == ''") + expect(job.if).toContain(spec) + expect(install.with['native-runtime']).toBe('node') + expect(run.env.ORCA_RUN_DOCKER_SSH_BROWSER_E2E).toBe('1') + expect(run.run).toContain(`vitest run --config config/vitest.config.ts ${spec}`) + expect(run['continue-on-error']).toBeUndefined() + expect( + workflow.jobs['changed-e2e'].steps.find((step) => step.name === 'Run changed E2E specs').run + ).toContain(`. != "${spec}"`) + for (const changed of [ + spec, + 'src/main/browser/ssh-browser-network-execution-route.ts', + 'src/main/browser/browser-network-deferred-socket.ts', + 'src/main/browser/browser-network-execution-route.ts', + 'src/main/browser/system-ssh-socks-client-socket.ts', + 'src/main/ssh/system-ssh-dynamic-forward-process.ts', + 'tests/e2e/helpers/docker-ssh-relay-target.ts', + 'tests/e2e/helpers/docker-ssh-relay-image.ts' + ]) { + expect(selectPrE2eSpecs([changed])).toContain(spec) + } + expect(selectPrE2eSpecs(['src/renderer/src/components/Unrelated.tsx'])).not.toContain(spec) + expect(selectPrE2eSpecs(['tests/e2e/helpers/docker-ssh-relay-terminal-tabs.ts'])).not.toContain( + spec + ) +}) diff --git a/config/scripts/ssh-localhost-e2e-routing.test.mjs b/config/scripts/ssh-localhost-e2e-routing.test.mjs new file mode 100644 index 00000000000..b400e86153c --- /dev/null +++ b/config/scripts/ssh-localhost-e2e-routing.test.mjs @@ -0,0 +1,52 @@ +import { existsSync, readFileSync } from 'node:fs' +import { resolve } from 'node:path' +import { parse } from 'yaml' +import { expect, it } from 'vitest' +import { selectPrE2eSpecs } from './pr-e2e-source-routing.mjs' + +const workflow = parse( + readFileSync(resolve(import.meta.dirname, '../../.github/workflows/e2e.yml'), 'utf8') +) + +it('gives the localhost SSH journey its same-filesystem server and agent prerequisite', () => { + const spec = 'tests/e2e/ssh-localhost.spec.ts' + const job = workflow.jobs['ssh-localhost'] + expect(job.if).toContain("inputs.test_files == ''") + expect(job.if).toContain(spec) + expect(job['runs-on']).toBe('ubuntu-latest') + expect(job.needs).toEqual(['build', 'prepare-native-cache']) + const setup = job.steps.find((step) => step.name === 'Start isolated localhost SSH server') + expect(setup.run).toContain('ListenAddress 127.0.0.1') + expect(setup.run).toContain('PasswordAuthentication no') + expect(setup.run).toContain('UsePAM yes') + expect(setup.run).toContain('mkdir -p "$HOME/.pi/agent"') + for (const key of ['ORCA_E2E_SSH_PORT', 'ORCA_E2E_SSH_USER', 'ORCA_E2E_SSH_IDENTITY_FILE']) { + expect(setup.run).toContain(key) + } + const run = job.steps.find((step) => step.name === 'Run localhost SSH terminal and hook journey') + expect(run.env.ORCA_E2E_SSH_LOCALHOST).toBe('1') + expect(run.env.ORCA_FEATURE_REMOTE_AGENT_HOOKS).toBe('1') + expect(run.run).toContain(spec) + expect(run.run).toContain('--project=electron-headless') + expect(run.run).not.toContain('--retries') + expect(run['continue-on-error']).toBeUndefined() + expect( + workflow.jobs['changed-e2e'].steps.find((step) => step.name === 'Run changed E2E specs').run + ).toContain(`. != "${spec}"`) +}) + +it('selects the localhost journey for its remote hook authorities', () => { + const spec = 'tests/e2e/ssh-localhost.spec.ts' + for (const file of [ + 'src/relay/relay-agent-hook-runtime.ts', + 'src/relay/agent-hook-server.ts', + 'src/relay/plugin-overlay.ts', + 'src/main/agent-hooks/server.ts', + 'src/main/ssh/ssh-relay-session.ts', + 'src/shared/agent-hook-relay.ts' + ]) { + expect(existsSync(resolve(import.meta.dirname, '../..', file)), file).toBe(true) + expect(selectPrE2eSpecs([file])).toContain(spec) + } + expect(selectPrE2eSpecs(['src/renderer/src/components/Unrelated.tsx'])).not.toContain(spec) +}) diff --git a/config/scripts/verify-wsl-e2e-participation.mjs b/config/scripts/verify-wsl-e2e-participation.mjs new file mode 100644 index 00000000000..21570ef7689 --- /dev/null +++ b/config/scripts/verify-wsl-e2e-participation.mjs @@ -0,0 +1,54 @@ +import { readFileSync } from 'node:fs' +import { pathToFileURL } from 'node:url' + +export const WSL_TEST_TITLES = [ + 'tab-bar + menu launches an agent inside WSL @tab-bar-agent-launch-golden', + 'WSL terminal keyboard paste preserves Linux shell content with one PTY owner', + 'existing WSL terminal keeps paste runtime after default shell changes' +] + +export function verifyWslParticipation(report) { + const stats = report?.stats + if ( + !stats || + stats.expected !== 9 || + stats.skipped !== 0 || + stats.unexpected !== 0 || + stats.flaky !== 0 || + report.errors?.length + ) { + throw new Error(`WSL participation failed: ${JSON.stringify(stats)}`) + } + const counts = new Map(WSL_TEST_TITLES.map((title) => [title, 0])) + const visit = (suites) => { + for (const suite of suites ?? []) { + for (const spec of suite.specs ?? []) { + if (!counts.has(spec.title)) { + throw new Error(`Unexpected WSL scenario: ${spec.title}`) + } + for (const test of spec.tests ?? []) { + if ( + test.expectedStatus !== 'passed' || + test.results?.length !== 1 || + test.results[0].status !== 'passed' + ) { + throw new Error(`WSL scenario did not pass without retries: ${spec.title}`) + } + counts.set(spec.title, counts.get(spec.title) + 1) + } + } + visit(suite.suites) + } + } + visit(report.suites) + for (const [title, count] of counts) { + if (count !== 3) { + throw new Error(`WSL scenario requires three executions: ${title} (${count})`) + } + } +} + +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { + verifyWslParticipation(JSON.parse(readFileSync(process.argv[2], 'utf8'))) + console.log('All three WSL scenarios passed three times without skips or retries.') +} diff --git a/config/scripts/verify-wsl-e2e-participation.test.mjs b/config/scripts/verify-wsl-e2e-participation.test.mjs new file mode 100644 index 00000000000..ae2f0935879 --- /dev/null +++ b/config/scripts/verify-wsl-e2e-participation.test.mjs @@ -0,0 +1,52 @@ +import { describe, expect, it } from 'vitest' +import { verifyWslParticipation, WSL_TEST_TITLES } from './verify-wsl-e2e-participation.mjs' + +function report() { + return { + stats: { expected: 9, skipped: 0, unexpected: 0, flaky: 0 }, + suites: [ + { + suites: [ + { + specs: WSL_TEST_TITLES.map((title) => ({ + title, + tests: Array.from({ length: 3 }, () => ({ + expectedStatus: 'passed', + results: [{ status: 'passed' }] + })) + })) + } + ] + } + ] + } +} + +describe('WSL participation', () => { + it('accepts all three named scenarios executed three times', () => { + expect(() => verifyWslParticipation(report())).not.toThrow() + }) + it.each(['skipped', 'unexpected', 'flaky'])('rejects a nonzero %s result', (key) => { + const value = report() + value.stats[key] = 1 + expect(() => verifyWslParticipation(value)).toThrow('participation failed') + }) + it('rejects missing scenarios even when aggregate counts claim nine passes', () => { + const value = report() + value.suites[0].suites[0].specs.pop() + expect(() => verifyWslParticipation(value)).toThrow('requires three executions') + }) + it('rejects an unrelated scenario substituted for an expected scenario', () => { + const value = report() + value.suites[0].suites[0].specs[0].title = 'native shell passes' + expect(() => verifyWslParticipation(value)).toThrow('Unexpected WSL scenario') + }) + it('rejects a pass obtained after a failed attempt', () => { + const value = report() + value.suites[0].suites[0].specs[0].tests[0].results.unshift({ status: 'failed' }) + expect(() => verifyWslParticipation(value)).toThrow('without retries') + }) + it('rejects missing report content', () => { + expect(() => verifyWslParticipation({})).toThrow('participation failed') + }) +}) diff --git a/config/scripts/wsl-e2e-lane-contract.test.mjs b/config/scripts/wsl-e2e-lane-contract.test.mjs new file mode 100644 index 00000000000..0369eb7c0c4 --- /dev/null +++ b/config/scripts/wsl-e2e-lane-contract.test.mjs @@ -0,0 +1,69 @@ +import { readFileSync } from 'node:fs' +import { describe, expect, it } from 'vitest' +import { parse } from 'yaml' +import { hasWslSourceChange, selectPrE2eSpecs } from './pr-e2e-source-routing.mjs' + +const read = (path) => readFileSync(new URL(`../../${path}`, import.meta.url), 'utf8') + +describe('real WSL terminal lane', () => { + it.each([ + 'config/scripts/verify-wsl-e2e-participation.mjs', + 'src/main/wsl-availability.ts', + 'src/main/wsl/wsl-runner.ts', + 'src/main/pty/wsl-orca-env.ts', + 'src/shared/wsl-login-shell-command.ts', + 'src/shared/windows-terminal-shell.ts', + 'tests/e2e/helpers/wsl-golden-stub-agent.ts', + 'tests/e2e/golden-tab-bar-agent-launch.spec.ts', + 'tests/e2e/terminal-windows-shell-paste-ownership.spec.ts', + '.github/actions/setup-wsl-test-runtime/setup.ps1', + '.github/workflows/windows-wsl-e2e.yml' + ])('routes %s to both WSL sentinels', (path) => { + expect(hasWslSourceChange([path])).toBe(true) + expect(selectPrE2eSpecs([path])).toEqual( + expect.arrayContaining([ + 'tests/e2e/golden-tab-bar-agent-launch.spec.ts', + 'tests/e2e/terminal-windows-shell-paste-ownership.spec.ts' + ]) + ) + }) + + it.each([ + 'docs/reference/wsl-command-execution.md', + 'src/main/wsl-availability.test.ts', + 'src/main/ssh/connection.ts' + ])('excludes unrelated or unit-only change %s', (path) => { + expect(hasWslSourceChange([path])).toBe(false) + }) + + it('runs the reusable lane at the immutable PR head', () => { + const pr = parse(read('.github/workflows/pr.yml')) + expect(pr.jobs.windows_wsl.if).toBe("needs.code_paths.outputs.wsl_source_changed == 'true'") + expect(pr.jobs.windows_wsl.with.ref).toBe('${{ github.event.pull_request.head.sha }}') + const detector = pr.jobs['code_paths'].steps.find( + (step) => step.name === 'Filter changed E2E specs' + ) + expect(detector.run).toContain( + 'WSL_CHANGED="$(git diff --name-only --no-renames --diff-filter=ACDMR' + ) + expect(detector.run).toContain( + '"$WSL_CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --wsl-source' + ) + const workflow = parse(read('.github/workflows/windows-wsl-e2e.yml')) + const steps = workflow.jobs['wsl-terminal'].steps + expect(steps[0].with.ref).toBe('${{ inputs.ref || github.sha }}') + expect(steps.some((step) => step.uses === './.github/actions/setup-wsl-test-runtime')).toBe( + true + ) + const exercise = steps.find((step) => step.name === 'Exercise real WSL launch and paste') + expect(exercise.run.split(/\s+/).filter((arg) => arg.startsWith('--repeat-each='))).toEqual([ + '--repeat-each=3' + ]) + expect(exercise.run).toContain('--grep "WSL"') + const receipt = steps.find((step) => step.name === 'Require all nine WSL executions') + expect(receipt.if).toBe('always()') + expect(receipt.run).toBe( + 'node config/scripts/verify-wsl-e2e-participation.mjs test-results/wsl-results.json' + ) + }) +}) diff --git a/mobile/app.json b/mobile/app.json index 131e3899396..fc36687d74f 100644 --- a/mobile/app.json +++ b/mobile/app.json @@ -75,7 +75,7 @@ "allowBackup": false, "permissions": ["RECORD_AUDIO", "MODIFY_AUDIO_SETTINGS"], "package": "com.stably.orca.mobile", - "versionCode": 15 + "versionCode": 16 }, "plugins": [ "expo-router", diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-acquisition-options.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-acquisition-options.test.ts index 6340e12a265..8afbdedae8d 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-acquisition-options.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-acquisition-options.test.ts @@ -28,7 +28,8 @@ afterEach(async () => { function attachParams( operationId: string, - expectedRuntimeFence: number | null + expectedRuntimeFence: number | null, + options?: Readonly> ): AgentSessionAttachParams { const params: AgentSessionAttachParams = { envelope: { @@ -47,6 +48,7 @@ function attachParams( agent: 'codex', accountHome: { variable: 'CODEX_HOME', path: '/home/dev/.codex' }, runtimeKind: 'native', + ...(options ? { options } : {}), providerHandle: { kind: 'codex', threadId: 'legacy-thread' } } return { @@ -101,6 +103,70 @@ function expectSettledAttachLease(record: AgentSessionRecord | null): void { } describe('structured session acquisition options', () => { + it('persists create defaults before the first provider acquisition', async () => { + root = await mkdtemp(join(tmpdir(), 'orca-create-options-')) + const store = await AgentSessionRecordStore.open({ + directory: join(root, 'store'), + hostId: 'local' + }) + const sessionAdapter = adapter({ origin: 'created' }) + const options = { model: 'gpt-5.6-sol', effort: 'medium' } + + const created = await performAttach({ + store, + adapter: sessionAdapter, + journalRoot: root, + authority: { + spawnToken: 'spawn-a', + claimKeyId: 'key-1', + handoffOperationId: CREATE_OPERATION, + probe: { outcome: 'reservation-unused' } + }, + callerKey: 'client-1', + params: attachParams(CREATE_OPERATION, null, options), + now: () => NOW, + onAttached: () => {} + }) + + expect(created).toMatchObject({ ok: true }) + expect(sessionAdapter.acquire).toHaveBeenCalledWith(expect.objectContaining({ options })) + expect(store.getRecord(SESSION)?.options).toEqual(options) + }) + + it('replays a create retried after the host re-resolved different options', async () => { + root = await mkdtemp(join(tmpdir(), 'orca-create-retry-')) + const store = await AgentSessionRecordStore.open({ + directory: join(root, 'store'), + hostId: 'local' + }) + const sessionAdapter = adapter({ origin: 'created' }) + const attempt = async (options: Readonly>, spawnToken: string) => + performAttach({ + store, + adapter: sessionAdapter, + journalRoot: root!, + authority: { + spawnToken, + claimKeyId: 'key-1', + handoffOperationId: CREATE_OPERATION, + probe: { outcome: 'reservation-unused' } + }, + callerKey: 'client-1', + params: attachParams(CREATE_OPERATION, null, options), + now: () => NOW, + onAttached: () => {} + }) + + const created = await attempt({ model: 'gpt-5.6-sol', effort: 'medium' }, 'spawn-a') + // Why: the user may reselect a model between an unknown-outcome create and the + // retry that reuses its operation id; the retry must replay, not conflict. + const retried = await attempt({ model: 'gpt-5.5', effort: 'high' }, 'spawn-b') + + expect(created).toMatchObject({ ok: true }) + expect(retried).toMatchObject({ ok: true }) + expect(store.getRecord(SESSION)?.options).toEqual({ model: 'gpt-5.6-sol', effort: 'medium' }) + }) + it('persists provider options before proving a resumed legacy record', async () => { root = await mkdtemp(join(tmpdir(), 'orca-acquisition-options-')) const storeDir = join(root, 'store') diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts index ce58e31b4ee..59a5bfe0b8e 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts @@ -54,6 +54,8 @@ export type AgentSessionAttachParams = { agent: AgentSessionHandleProvider accountHome: AgentSessionAccountHome runtimeKind: AgentSessionOwnerRuntimeKind + /** Host-resolved defaults for a create-by-intent; remote attach schemas do not accept them. */ + options?: Readonly> /** Omitted only for create-by-intent; the adapter proves the durable handle. */ providerHandle?: Exclude } @@ -70,7 +72,10 @@ export type AgentSessionAttachAuthority = { /** The fields that define WHICH session this call would attach to. Deliberately * excludes the spawn token and the probe: those differ between a first attempt - * and its retry, and a retry must replay rather than conflict. */ + * and its retry, and a retry must replay rather than conflict. `options` is + * excluded for the same reason — it is the session's initial state, not its + * identity, and the host re-resolves it from settings the user may have changed + * between an unknown-outcome attempt and its retry. */ export function attachFingerprintFields(params: AgentSessionAttachParams): Record { return { location: params.location, @@ -191,6 +196,7 @@ export function reserveRequestFor(input: { location: params.location, provider: params.provider, accountHome: params.accountHome, + ...(params.options ? { options: params.options } : {}), ...(authority.launchArgs ? { launchArgs: authority.launchArgs } : {}), ...(authority.launchEnv ? { launchEnv: authority.launchEnv } : {}), runtimeKind: params.runtimeKind, diff --git a/src/main/runtime/agent-session-reservation-admission.ts b/src/main/runtime/agent-session-reservation-admission.ts index 1735d67e62c..f1be94a2c0e 100644 --- a/src/main/runtime/agent-session-reservation-admission.ts +++ b/src/main/runtime/agent-session-reservation-admission.ts @@ -21,6 +21,7 @@ import { agentSessionExecutionLocationsEqual, isAgentSessionLaunchArgs, isAgentSessionLaunchEnv, + isAgentSessionOptions, type AgentSessionAccountHome, type AgentSessionExecutionLocation, type AgentSessionLaunchArgs, @@ -43,6 +44,8 @@ export type AgentSessionReserveRequest = { launchArgs?: AgentSessionLaunchArgs /** Current launch input validated here but never written to the durable record. */ launchEnv?: AgentSessionLaunchEnv + /** Initial provider options persisted before the first process is acquired. */ + options?: Readonly> runtimeKind: AgentSessionReservation['runtimeKind'] /** Null when the session does not exist yet; otherwise the fence the caller last observed. */ expectedFence: number | null @@ -131,6 +134,9 @@ export function applyAgentSessionReservation( if (request.launchArgs && !isAgentSessionLaunchArgs(request.launchArgs)) { throw new Error('agent_session_launch_args_invalid') } + if (request.options && !isAgentSessionOptions(request.options)) { + throw new Error('agent_session_options_invalid') + } const reservation: AgentSessionReservation = { runtimeKind: request.runtimeKind, spawnToken: @@ -186,6 +192,7 @@ function createAgentSessionRecord( provider: request.provider, providerHandleChain: [], accountHome: request.accountHome, + ...(request.options ? { options: { ...request.options } } : {}), ...(request.launchArgs ? { launchArgs: [...request.launchArgs] } : {}), createdAt: request.now, updatedAt: request.now, diff --git a/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts b/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts index aef04bde6bc..497d5b381e7 100644 --- a/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts +++ b/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts @@ -13,6 +13,7 @@ import { getLocalProjectWorktreeGitOptions } from '../project-runtime-git-option import type { AgentSessionAttachParams } from '../native-chat/agent-session-wire/structured-agent-session-attach' import { getSystemCodexHomePath } from '../codex/codex-home-paths' import { resolveTuiAgentLaunchEnv } from '../../shared/tui-agent-launch-defaults' +import { resolveStructuredLaunchSeedOptions } from '../../shared/native-chat-session-option-defaults' import { hasPersistedStructuredAgentSessionStore as hasPersistedStructuredAgentSessionStoreOnDisk } from './structured-agent-session-runtime' import { getProfileUserDataPath } from '../orca-profiles/profile-storage-paths' import { homedir } from 'node:os' @@ -161,6 +162,10 @@ export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends Orca } const settings = this.requireStore().getSettings() const launchEnv = resolveTuiAgentLaunchEnv(input.agent, settings.agentDefaultEnv) + const options = resolveStructuredLaunchSeedOptions( + settings.nativeChatSessionOptions, + input.agent + ) const location = await this.resolveStructuredAgentSessionLocation(input.worktree) const workspacePath = (await this.resolveRuntimeFileTarget(input.worktree)).worktree.path return { @@ -177,6 +182,7 @@ export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends Orca variable: input.agent === 'claude' ? 'CLAUDE_CONFIG_DIR' : 'CODEX_HOME', path: await resolveAccountHomePath({ workspacePath, launchEnv, location }) }, + ...(options ? { options } : {}), runtimeKind: 'native' } } diff --git a/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts b/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts index af1fbc20372..9d2c6589508 100644 --- a/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts +++ b/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts @@ -7,7 +7,15 @@ describe('structured agent-session create intent', () => { const runtime = new OrcaRuntimeService( { getSettings: () => ({ - agentDefaultEnv: { codex: { CODEX_HOME: '/configured/home' } } + agentDefaultEnv: { codex: { CODEX_HOME: '/configured/home' } }, + nativeChatSessionOptions: { + codex: { + model: 'gpt-5.6-sol', + valuesByModel: { + 'gpt-5.6-sol': { effort: 'medium', fastMode: true, personality: 'concise' } + } + } + } }) } as never, undefined, @@ -51,6 +59,7 @@ describe('structured agent-session create intent', () => { variable: 'CODEX_HOME', path: '/accounts/selected/home' }) + expect(intent.options).toEqual({ model: 'gpt-5.6-sol', effort: 'medium' }) }) it('pins the configured Claude launch home without Codex launch preparation', async () => { @@ -60,6 +69,12 @@ describe('structured agent-session create intent', () => { getSettings: () => ({ agentDefaultEnv: { claude: { CLAUDE_CONFIG_DIR: '/configured/claude-home' } + }, + nativeChatSessionOptions: { + claude: { + model: 'opus', + valuesByModel: { opus: { effort: 'high', fastMode: true } } + } } }) } as never, @@ -101,6 +116,7 @@ describe('structured agent-session create intent', () => { variable: 'CLAUDE_CONFIG_DIR', path: '/configured/claude-home' }) + expect(intent.options).toEqual({ model: 'opus', effort: 'high' }) }) it('uses the managed Claude launch home before falling back to ~/.claude', async () => { diff --git a/src/main/runtime/rpc/methods/orchestration-worker-start-mode.ts b/src/main/runtime/rpc/methods/orchestration-worker-start-mode.ts index 46808474613..7c02c2a688f 100644 --- a/src/main/runtime/rpc/methods/orchestration-worker-start-mode.ts +++ b/src/main/runtime/rpc/methods/orchestration-worker-start-mode.ts @@ -86,7 +86,6 @@ const BLOCKER_REASON: Record< 'draft-prompt': 'structured_unsupported_on_host', 'floating-workspace': 'structured_unsupported_on_host', 'tui-launch-customization': 'tui_launch_customization', - 'initial-session-options': 'structured_unsupported_on_host', 'remote-execution-host': 'remote_execution_host', 'codex-on-windows': 'codex_on_windows', 'project-runtime': 'wsl_execution_runtime', diff --git a/src/main/runtime/rpc/methods/structured-agent-session-create.ts b/src/main/runtime/rpc/methods/structured-agent-session-create.ts index 19d15b1dd13..75a13ba6af9 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-create.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-create.ts @@ -18,7 +18,10 @@ import type { AgentSessionMutationEnvelope, AgentSessionMutationResult } from '../../../../shared/agent-session-wire' -import type { AgentSessionAttachParams } from '../../../native-chat/agent-session-wire/structured-agent-session-attach' +import { + attachFingerprintFields, + type AgentSessionAttachParams +} from '../../../native-chat/agent-session-wire/structured-agent-session-attach' import type { StructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-host' import type { StructuredAgentSessionCaller } from '../../../native-chat/agent-session-wire/structured-agent-session-host-types' import type { OrcaRuntimeService } from '../../orca-runtime' @@ -52,14 +55,7 @@ export async function prepareStructuredAgentSessionCreateForWorktree(args: { const hostFingerprint = computeAgentSessionPayloadFingerprint({ method: 'agentSession.attach', sessionId: args.envelope.sessionId, - fields: { - location: resolved.location, - provider: resolved.provider, - agent: resolved.agent, - accountHome: resolved.accountHome, - runtimeKind: resolved.runtimeKind, - expectedRuntimeFence: null - } + fields: attachFingerprintFields({ ...resolved, envelope: args.envelope }) }) const host = await args.ensureHost() const { agent: _resolvedAgent, provider: _resolvedProvider, ...resolvedAttach } = resolved diff --git a/src/main/runtime/rpc/methods/structured-agent-session.test.ts b/src/main/runtime/rpc/methods/structured-agent-session.test.ts index 4af7f165d0f..c43c82d03ee 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.test.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.test.ts @@ -200,6 +200,10 @@ function dispatcher(runtimeOverrides: Record = {}): RpcDispatch variable: params.agent === 'claude' ? 'CLAUDE_CONFIG_DIR' : 'CODEX_HOME', path: params.agent === 'claude' ? '/host/.claude' : '/host/.codex' }, + options: + params.agent === 'claude' + ? { model: 'opus', effort: 'high' } + : { model: 'gpt-5.6-sol', effort: 'medium' }, runtimeKind: 'native' })), publishStructuredAgentSessionTab: vi.fn() @@ -482,7 +486,8 @@ describe('method routing', () => { expect(hostCalls.attach).toHaveBeenCalledWith( expect.anything(), expect.objectContaining({ - accountHome: { variable: 'CODEX_HOME', path: '/host/.codex' } + accountHome: { variable: 'CODEX_HOME', path: '/host/.codex' }, + options: { model: 'gpt-5.6-sol', effort: 'medium' } }) ) expect(hostCalls.attach.mock.calls[0]?.[1]).not.toHaveProperty('providerHandle') diff --git a/src/renderer/src/components/right-sidebar/source-control/commit/commit-notices.tsx b/src/renderer/src/components/right-sidebar/source-control/commit/commit-notices.tsx index e46b1f6b3ea..666181a649a 100644 --- a/src/renderer/src/components/right-sidebar/source-control/commit/commit-notices.tsx +++ b/src/renderer/src/components/right-sidebar/source-control/commit/commit-notices.tsx @@ -113,20 +113,20 @@ export function CommitNotices({ role={createPrIntentNotice.tone === 'destructive' ? 'alert' : 'status'} aria-live="polite" className={cn( - 'mt-1 flex min-w-0 items-center gap-1.5 text-[11px]', + 'mt-1 flex min-w-0 flex-col items-start gap-1 text-[11px]', createPrIntentNotice.tone === 'destructive' ? 'text-destructive' : 'text-muted-foreground' )} > {/* Why: Create Review blockers carry recovery steps; truncating hides the action the user needs in a narrow sidebar. */} - + {createPrIntentNotice.message} {createPrIntentNotice.action === 'settings' && onOpenSourceControlAiSettings ? ( + ) } diff --git a/src/renderer/src/lib/agent-launch-routing.test.ts b/src/renderer/src/lib/agent-launch-routing.test.ts index dd33a8357d9..45c86bb1517 100644 --- a/src/renderer/src/lib/agent-launch-routing.test.ts +++ b/src/renderer/src/lib/agent-launch-routing.test.ts @@ -31,6 +31,9 @@ describe('resolveAgentLaunchRoute', () => { 'routes a supported local %s launch to structured native chat', (agent) => { expect(route({ agent })).toBe('structured-native-chat') + expect(route({ agent, initialSessionOptions: { model: 'gpt-5.6-sol' } })).toBe( + 'structured-native-chat' + ) expect( route({ agent, launchText: 'explain this change', promptDelivery: 'auto-submit' }) ).toBe('structured-native-chat') @@ -118,7 +121,6 @@ describe('resolveAgentLaunchRoute', () => { expect(route({ agent: 'openclaude' })).toBe('legacy-native-chat') expect(route({ agent: 'grok' })).toBe('legacy-native-chat') expect(route({ requiresTuiLaunchCustomization: true })).toBe('legacy-native-chat') - expect(route({ initialSessionOptions: { model: 'gpt-5.6-sol' } })).toBe('legacy-native-chat') }) it.each([ diff --git a/src/renderer/src/lib/agent-launch-routing.ts b/src/renderer/src/lib/agent-launch-routing.ts index 4cb773e28fc..17cb95a43d0 100644 --- a/src/renderer/src/lib/agent-launch-routing.ts +++ b/src/renderer/src/lib/agent-launch-routing.ts @@ -64,8 +64,7 @@ export function resolveAgentLaunchRoute(input: AgentLaunchRoutingInput): AgentLa workspaceKind: input.workspaceKind, projectRuntime: input.projectRuntime, isDraftPrompt: input.promptDelivery === 'draft', - requiresTuiLaunchCustomization: input.requiresTuiLaunchCustomization, - initialSessionOptions: input.initialSessionOptions + requiresTuiLaunchCustomization: input.requiresTuiLaunchCustomization }).supported ? 'structured-native-chat' : 'legacy-native-chat' diff --git a/src/renderer/src/lib/launch-agent-structured-chat-guard.test.ts b/src/renderer/src/lib/launch-agent-structured-chat-guard.test.ts index 5753651c051..179f6203d0b 100644 --- a/src/renderer/src/lib/launch-agent-structured-chat-guard.test.ts +++ b/src/renderer/src/lib/launch-agent-structured-chat-guard.test.ts @@ -43,7 +43,13 @@ const store = { activeRuntimeEnvironmentId: null, experimentalNativeChat: true, experimentalStructuredNativeChat: true, - openAgentTabsInChatByDefault: true + openAgentTabsInChatByDefault: true, + nativeChatSessionOptions: undefined as + | Record< + string, + { model?: string; valuesByModel?: Record> } + > + | undefined }, projects: [{ id: 'repo-1', localWindowsRuntimePreference: { kind: 'inherit-global' as const } }], repos: [{ id: 'repo-1', connectionId: null as string | null, path: '/repo' }], @@ -153,6 +159,7 @@ describe('structured chat adoption guard on the launch path', () => { mockToastError.mockReset() hostCapabilities = STRUCTURED_HOST_CAPABILITIES store.settings.openAgentTabsInChatByDefault = true + store.settings.nativeChatSessionOptions = undefined }) it('takes the structured path when the chat-default view is selected', async () => { @@ -175,6 +182,23 @@ describe('structured chat adoption guard on the launch path', () => { expect(mockWaitForAgentReady).not.toHaveBeenCalled() }) + // Routing only: the host seeds the saved values, so preservation is pinned there. + it('takes the structured path when a Codex model and effort are already saved', async () => { + store.settings.nativeChatSessionOptions = { + codex: { + model: 'gpt-5.6-sol', + valuesByModel: { 'gpt-5.6-sol': { effort: 'medium' } } + } + } + const { launchAgentInNewTab } = await import('./launch-agent-in-new-tab') + + const result = launchAgentInNewTab({ agent: 'codex', worktreeId: 'wt-1' }) + + expect(result).toMatchObject({ tabId: null, focusAfterMenuClose: 'structured-session' }) + expect(mockCreateStructuredCodexSessionLaunchIntent).toHaveBeenCalledWith('wt-1', 'codex') + expect(mockCreateTab).not.toHaveBeenCalled() + }) + it('takes the structured path for Claude, naming Claude as the create provider', async () => { const { launchAgentInNewTab } = await import('./launch-agent-in-new-tab') diff --git a/src/shared/agent-session-record.ts b/src/shared/agent-session-record.ts index 9a27ec1afb7..71369cffaed 100644 --- a/src/shared/agent-session-record.ts +++ b/src/shared/agent-session-record.ts @@ -221,7 +221,7 @@ function isAgentSessionAccountHome(value: unknown): value is AgentSessionAccount ) } -function isAgentSessionOptions(value: unknown): value is Record { +export function isAgentSessionOptions(value: unknown): value is Record { if (typeof value !== 'object' || value === null || Array.isArray(value)) { return false } diff --git a/src/shared/native-chat-session-option-defaults.test.ts b/src/shared/native-chat-session-option-defaults.test.ts index aed767d67ac..f982df9b7c1 100644 --- a/src/shared/native-chat-session-option-defaults.test.ts +++ b/src/shared/native-chat-session-option-defaults.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from 'vitest' import { clearNativeChatSessionOptionModel, resolveNativeChatSessionOptionDefaults, + resolveStructuredLaunchSeedOptions, updateNativeChatSessionOptionDefaults } from './native-chat-session-option-defaults' import type { PersistedNativeChatSessionOptions } from './native-chat-session-options' @@ -74,3 +75,66 @@ describe('resolveNativeChatSessionOptionDefaults', () => { }) }) }) + +describe('resolveStructuredLaunchSeedOptions', () => { + const persistedCodex = ( + valuesByModel: Record> + ): PersistedNativeChatSessionOptions => + ({ codex: { model: 'gpt-5.6-sol', valuesByModel } }) as PersistedNativeChatSessionOptions + + it('seeds the saved model and effort a structured create must apply', () => { + expect( + resolveStructuredLaunchSeedOptions( + persistedCodex({ 'gpt-5.6-sol': { effort: 'medium' } }), + 'codex' + ) + ).toEqual({ model: 'gpt-5.6-sol', effort: 'medium' }) + }) + + it('drops ids the providers only accept mid-session', () => { + // `fastMode` is a boolean and `personality` is settable only mid-session; + // neither belongs in the reservation's Record. + expect( + resolveStructuredLaunchSeedOptions( + persistedCodex({ + 'gpt-5.6-sol': { effort: 'high', fastMode: true, personality: 'concise' } + }), + 'codex' + ) + ).toEqual({ model: 'gpt-5.6-sol', effort: 'high' }) + }) + + it('drops a seeded id whose persisted value is not a usable string', () => { + // settings.json is user-writable, so a non-string `effort` must not reach a + // record typed Record and be emitted as a turn option. + expect( + resolveStructuredLaunchSeedOptions( + persistedCodex({ 'gpt-5.6-sol': { effort: true } }), + 'codex' + ) + ).toEqual({ model: 'gpt-5.6-sol' }) + expect( + resolveStructuredLaunchSeedOptions( + persistedCodex({ 'gpt-5.6-sol': { effort: ' ' } }), + 'codex' + ) + ).toEqual({ model: 'gpt-5.6-sol' }) + }) + + it('seeds nothing when the stored values empty the model out', () => { + // `valuesByModel` is merged over the resolved model, so a stored `model` key + // can blank it. Emitting `{ model: '' }` fails the record's bounded-string + // guard, and that throw is not a wire refusal code — it escapes as a raw + // error the client reads as unknown, stranding the launch with no fallback. + expect( + resolveStructuredLaunchSeedOptions(persistedCodex({ 'gpt-5.6-sol': { model: '' } }), 'codex') + ).toBeUndefined() + }) + + it('seeds nothing until a model is picked, so the CLI default survives', () => { + expect(resolveStructuredLaunchSeedOptions(undefined, 'codex')).toBeUndefined() + expect( + resolveStructuredLaunchSeedOptions({ codex: { valuesByModel: {} } }, 'codex') + ).toBeUndefined() + }) +}) diff --git a/src/shared/native-chat-session-option-defaults.ts b/src/shared/native-chat-session-option-defaults.ts index 238005ccd62..41f607ca12a 100644 --- a/src/shared/native-chat-session-option-defaults.ts +++ b/src/shared/native-chat-session-option-defaults.ts @@ -28,6 +28,33 @@ export function resolveNativeChatSessionOptionDefaults( return values } +/** Why only these two: they are the only ids the picker persists into + * `nativeChatSessionOptions` that both structured providers also accept as + * strings. Claude's `fastMode` is a boolean the durable `Record` + * record cannot carry, and the providers' remaining keys are settable only + * mid-session, never seeded at launch. */ +const STRUCTURED_LAUNCH_SEED_OPTION_IDS = ['model', 'effort'] as const + +/** The saved selection a structured create seeds into its reservation, narrowed + * to the wire-safe string subset the durable record and both providers accept. */ +export function resolveStructuredLaunchSeedOptions( + persisted: PersistedNativeChatSessionOptions | null | undefined, + agent: AgentType +): Record | undefined { + const defaults = resolveNativeChatSessionOptionDefaults(persisted, agent) + if (!defaults) { + return undefined + } + const seeded: Record = {} + for (const id of STRUCTURED_LAUNCH_SEED_OPTION_IDS) { + const value = defaults[id] + if (typeof value === 'string' && value.trim()) { + seeded[id] = value + } + } + return Object.keys(seeded).length > 0 ? seeded : undefined +} + /** Why: an authoritative probe proved this id gone, and a stale `model` is emitted * verbatim as a launch flag — grok exits fatally on an unknown one. Dropping only * `model` keeps the per-model option values for a later reselect. */ diff --git a/src/shared/structured-native-chat-launch-route.test.ts b/src/shared/structured-native-chat-launch-route.test.ts index d2ca360dbc2..48cb117fdf5 100644 --- a/src/shared/structured-native-chat-launch-route.test.ts +++ b/src/shared/structured-native-chat-launch-route.test.ts @@ -63,7 +63,6 @@ describe('per-launch structured feasibility', () => { ['a draft prompt', { isDraftPrompt: true }, 'draft-prompt'], ['a floating workspace', { workspaceKind: 'floating' }, 'floating-workspace'], ['a custom TUI launch', { requiresTuiLaunchCustomization: true }, 'tui-launch-customization'], - ['session options', { initialSessionOptions: { model: 'x' } }, 'initial-session-options'], ['an SSH host', { executionHostId: 'ssh:host-a' }, 'remote-execution-host'], ['Codex on Windows', { agent: 'codex', platform: 'win32' }, 'codex-on-windows'], ['a missing capability', { hostCapabilities: [] }, 'runtime-capability'] diff --git a/src/shared/structured-native-chat-launch-route.ts b/src/shared/structured-native-chat-launch-route.ts index 89744ec4d08..b97ffcc0dac 100644 --- a/src/shared/structured-native-chat-launch-route.ts +++ b/src/shared/structured-native-chat-launch-route.ts @@ -25,7 +25,6 @@ export type StructuredNativeChatBlocker = | 'draft-prompt' | 'floating-workspace' | 'tui-launch-customization' - | 'initial-session-options' | 'remote-execution-host' | 'codex-on-windows' | 'project-runtime' @@ -45,7 +44,6 @@ export type StructuredNativeChatSupportInput = { /** A draft stays terminal-backed: the composer, not a turn, owns unsent text. */ isDraftPrompt?: boolean requiresTuiLaunchCustomization?: boolean - initialSessionOptions?: Readonly> } /** The user's default for a new agent tab: native chat rather than the raw TUI. */ @@ -81,9 +79,6 @@ export function resolveStructuredNativeChatSupport( if (input.requiresTuiLaunchCustomization === true) { return { supported: false, blocker: 'tui-launch-customization' } } - if (input.initialSessionOptions && Object.keys(input.initialSessionOptions).length > 0) { - return { supported: false, blocker: 'initial-session-options' } - } if (input.executionHostId !== 'local') { return { supported: false, blocker: 'remote-execution-host' } } diff --git a/tests/e2e/fixtures/golden-stub-agent/golden-stub-agent.js b/tests/e2e/fixtures/golden-stub-agent/golden-stub-agent.js old mode 100755 new mode 100644 index 5c353ce2bc6..47810499db1 --- a/tests/e2e/fixtures/golden-stub-agent/golden-stub-agent.js +++ b/tests/e2e/fixtures/golden-stub-agent/golden-stub-agent.js @@ -3,8 +3,15 @@ const READY_MARKER = 'GOLDEN_STUB_AGENT_READY' const EXIT_MARKER = 'GOLDEN_STUB_AGENT_EXITED' +// The interactive fixture does not implement Codex's JSONL app-server API. +if (process.argv[2] === 'app-server') { + process.stderr.write("error: unrecognized subcommand 'app-server'\n") + process.exit(2) +} + const ESC = '\x1b' const keyboardProtocolMode = process.argv.includes('--keyboard-protocol') +const keyboardProtocolAgent = process.argv.includes('--grok') ? 'Grok' : 'Codex' // Both match the bytes after ESC, so the control character stays out of the // pattern: a CSI/SS3 introducer still missing its final byte, and a complete // CSI/SS3 sequence. Shift+Enter is matched before either is consulted. @@ -20,7 +27,7 @@ function render() { const lines = composer.split('\n') const renderedComposer = lines.map((line, index) => `${index === 0 ? '> ' : ' '}${line}`) process.stdout.write( - `${keyboardProtocolMode ? '\x1b]0;\u280b Codex is thinking\x07\x1b[>1u' : '\x1b]0;Golden Stub Agent\x07'}${[ + `${keyboardProtocolMode ? `\x1b]0;\u280b ${keyboardProtocolAgent} is thinking\x07\x1b[>1u` : '\x1b]0;Golden Stub Agent\x07'}${[ '\x1b[H\x1b[2JGolden Stub Agent', `[${READY_MARKER}]`, '', @@ -40,7 +47,7 @@ function exitCleanly() { if (process.stdin.isTTY) { process.stdin.setRawMode(false) } - const idleTitle = keyboardProtocolMode ? '\x1b]0;Codex\x07' : '' + const idleTitle = keyboardProtocolMode ? `\x1b]0;${keyboardProtocolAgent}\x07` : '' process.stdout.write(`${idleTitle}\x1b[?1049l[${EXIT_MARKER}]\r\n`, () => process.exit(0)) } diff --git a/tests/e2e/global-teardown.ts b/tests/e2e/global-teardown.ts index 63248cd36c8..0961eb0ee5c 100644 --- a/tests/e2e/global-teardown.ts +++ b/tests/e2e/global-teardown.ts @@ -10,7 +10,7 @@ import { readFileSync, existsSync, realpathSync, rmSync } from 'node:fs' import { TEST_REPO_PATH_FILE } from './global-setup' export function linkedWorktreePaths(testRepoDir: string): string[] { - const root = realpathSync(testRepoDir) + const root = realpathSync.native(testRepoDir) const output = execFileSync('git', ['-C', testRepoDir, 'worktree', 'list', '--porcelain'], { encoding: 'utf8' }) @@ -23,7 +23,7 @@ export function linkedWorktreePaths(testRepoDir: string): string[] { if (!existsSync(recordedPath)) { continue } - const canonicalPath = realpathSync(recordedPath) + const canonicalPath = realpathSync.native(recordedPath) if (canonicalPath !== root) { linked.add(canonicalPath) } @@ -32,7 +32,7 @@ export function linkedWorktreePaths(testRepoDir: string): string[] { } export function cleanupTestRepository(testRepoDir: string): void { - const root = realpathSync(testRepoDir) + const root = realpathSync.native(testRepoDir) let worktreePaths: string[] = [] try { worktreePaths = linkedWorktreePaths(root) diff --git a/tests/e2e/global-teardown.unit.test.ts b/tests/e2e/global-teardown.unit.test.ts index b5fabfe3de8..fde77199434 100644 --- a/tests/e2e/global-teardown.unit.test.ts +++ b/tests/e2e/global-teardown.unit.test.ts @@ -39,7 +39,7 @@ describe('E2E global teardown ownership', () => { git(repoPath, ['worktree', 'add', '-b', 'second-owned', secondWorktreePath]) expect(new Set(linkedWorktreePaths(repoPath))).toEqual( - new Set([realpathSync(firstWorktreePath), realpathSync(secondWorktreePath)]) + new Set([realpathSync.native(firstWorktreePath), realpathSync.native(secondWorktreePath)]) ) cleanupTestRepository(repoPath) diff --git a/tests/e2e/golden-fresh-profile-terminal.spec.ts b/tests/e2e/golden-fresh-profile-terminal.spec.ts index 5194868adb7..e987b2a20de 100644 --- a/tests/e2e/golden-fresh-profile-terminal.spec.ts +++ b/tests/e2e/golden-fresh-profile-terminal.spec.ts @@ -16,7 +16,7 @@ import { test.use({ dismissOnboarding: false, seedTestRepo: false }) async function createGitRepo(): Promise { - const root = realpathSync(await mkdtemp(path.join(os.tmpdir(), 'orca-e2e-golden-fresh-'))) + const root = realpathSync.native(await mkdtemp(path.join(os.tmpdir(), 'orca-e2e-golden-fresh-'))) const repoPath = path.join(root, 'golden-fresh-project') mkdirSync(repoPath) execFileSync('git', ['init'], { cwd: repoPath, stdio: 'pipe' }) diff --git a/tests/e2e/helpers/docker-ssh-relay-connection.ts b/tests/e2e/helpers/docker-ssh-relay-connection.ts index 3e0c35f3c53..caf29d40ce5 100644 --- a/tests/e2e/helpers/docker-ssh-relay-connection.ts +++ b/tests/e2e/helpers/docker-ssh-relay-connection.ts @@ -1,3 +1,4 @@ +import { connectSshTestTarget } from './ssh-test-target-connection' import { expect, type Page } from '@stablyai/playwright-test' import { @@ -30,153 +31,26 @@ export async function connectDockerSshRelayTarget( target: DockerSshRelayTarget, options: DockerSshRelayConnectionOptions = {} ): Promise { - return page.evaluate( - async ({ target, remotePath, relayGracePeriodSeconds, viaProxyJump, seedInitialTab }) => { - const store = window.__store - if (!store) { - throw new Error('Store unavailable') - } - const credentialUnsub = window.api.ssh.onCredentialRequest((request) => { - void window.api.ssh.submitCredential({ requestId: request.requestId, value: null }) - }) - try { - const { target: createdTarget, repoReadoptions } = await window.api.ssh.addTarget({ - target: { - label: `${viaProxyJump ? 'Docker SSH ProxyJump' : 'Docker SSH Relay'} E2E ${Date.now()}`, - ...(viaProxyJump ? { configHost: 'orca-e2e-destination' } : {}), - host: target.host, - port: viaProxyJump ? 22 : target.port, - username: 'root', - identityFile: target.identityFile, - identitiesOnly: true, - ...(viaProxyJump ? { jumpHost: 'orca-e2e-jump' } : {}), - relayGracePeriodSeconds - } - }) - store.getState().recordSshRepoReadoptions(repoReadoptions) - const state = await window.api.ssh.connect({ targetId: createdTarget.id }) - if (!state || state.status !== 'connected') { - throw new Error(`SSH target did not connect: ${JSON.stringify(state)}`) - } - if ( - !state.providerEpoch || - !Number.isSafeInteger(state.connectionGeneration) || - state.connectionGeneration === undefined || - state.connectionGeneration < 0 - ) { - throw new Error(`SSH target returned incomplete authority: ${JSON.stringify(state)}`) - } - store.getState().setSshConnectionState(createdTarget.id, state) - const labels = new Map(store.getState().sshTargetLabels) - labels.set(createdTarget.id, createdTarget.label) - store.getState().setSshTargetLabels(labels) - const executionHostId = `ssh:${encodeURIComponent(createdTarget.id)}` as const - const authority = { - targetId: createdTarget.id, - providerEpoch: state.providerEpoch, - connectionGeneration: state.connectionGeneration - } - - const result = await window.api.repos.addRemote({ - connectionId: createdTarget.id, - remotePath, - displayName: viaProxyJump ? 'Docker SSH ProxyJump E2E' : 'Docker SSH Relay E2E' - }) - if ('error' in result) { - throw new Error(result.error) - } - const hasExpectedRepoOwner = (): boolean => - store - .getState() - .repos.some( - (repo) => - repo.id === result.repo.id && - repo.connectionId === createdTarget.id && - repo.executionHostId === executionHostId - ) - const waitForRepoOwner = async (): Promise => { - if (hasExpectedRepoOwner()) { - return - } - await new Promise((resolve, reject) => { - const timer = window.setTimeout(() => { - unsubscribe() - reject(new Error(`Remote repo owner did not hydrate for ${result.repo.path}`)) - }, 15_000) - const unsubscribe = store.subscribe((next) => { - if ( - !next.repos.some( - (repo) => - repo.id === result.repo.id && - repo.connectionId === createdTarget.id && - repo.executionHostId === executionHostId - ) - ) { - return - } - window.clearTimeout(timer) - unsubscribe() - resolve() - }) - }) - } - await store.getState().fetchRepos() - await waitForRepoOwner() - const currentState = store.getState().sshConnectionStates.get(createdTarget.id) - if ( - currentState?.providerEpoch !== authority.providerEpoch || - currentState.connectionGeneration !== authority.connectionGeneration - ) { - throw new Error(`SSH authority rotated before worktree hydration for ${result.repo.path}`) - } - const worktreeResult = await store.getState().fetchWorktrees(result.repo.id, { - executionHostId, - directSshAuthority: authority, - requireAuthoritative: true - }) - if ( - worktreeResult.status !== 'complete' || - worktreeResult.repoId !== result.repo.id || - worktreeResult.authority.kind !== 'direct-ssh' || - worktreeResult.authority.executionHostId !== executionHostId || - worktreeResult.authority.targetId !== authority.targetId || - worktreeResult.authority.providerEpoch !== authority.providerEpoch || - worktreeResult.authority.connectionGeneration !== authority.connectionGeneration - ) { - throw new Error( - `Remote worktree hydration was not authoritative: ${JSON.stringify(worktreeResult)}` - ) - } - const worktree = (store.getState().worktreesByRepo[result.repo.id] ?? []).find( - (candidate) => candidate.hostId === executionHostId - ) - if (!worktree) { - throw new Error(`No remote worktree found for ${result.repo.path}`) - } - store.getState().setActiveWorktree(worktree.id) - if (seedInitialTab && (store.getState().tabsByWorktree[worktree.id] ?? []).length === 0) { - store.getState().createTab(worktree.id) - } - store.getState().setActiveTabType('terminal') - return { - targetId: createdTarget.id, - repoId: result.repo.id, - worktreeId: worktree.id - } - } finally { - credentialUnsub() - } + const viaProxyJump = options.viaProxyJump ?? false + return connectSshTestTarget( + page, + { + label: `${viaProxyJump ? 'Docker SSH ProxyJump' : 'Docker SSH Relay'} E2E ${Date.now()}`, + ...(viaProxyJump ? { configHost: 'orca-e2e-destination' } : {}), + host: target.host, + port: viaProxyJump ? 22 : target.port, + username: 'root', + identityFile: target.identityFile, + identitiesOnly: true, + ...(viaProxyJump ? { jumpHost: 'orca-e2e-jump' } : {}), + relayGracePeriodSeconds: options.relayGracePeriodSeconds ?? 1 }, { - target, remotePath: options.remotePath ?? - (options.viaProxyJump - ? DOCKER_SSH_PROXY_JUMP_REMOTE_REPO_PATH - : DOCKER_SSH_RELAY_REMOTE_REPO_PATH), - viaProxyJump: options.viaProxyJump ?? false, - seedInitialTab: options.seedInitialTab ?? true, - relayGracePeriodSeconds: options.relayGracePeriodSeconds ?? 1 + (viaProxyJump ? DOCKER_SSH_PROXY_JUMP_REMOTE_REPO_PATH : DOCKER_SSH_RELAY_REMOTE_REPO_PATH), + displayName: viaProxyJump ? 'Docker SSH ProxyJump E2E' : 'Docker SSH Relay E2E', + seedInitialTab: options.seedInitialTab } ) } diff --git a/tests/e2e/helpers/golden-stub-agent.ts b/tests/e2e/helpers/golden-stub-agent.ts index 427a6a2560c..393b244cf8f 100644 --- a/tests/e2e/helpers/golden-stub-agent.ts +++ b/tests/e2e/helpers/golden-stub-agent.ts @@ -25,7 +25,7 @@ export function getGoldenStubAgentLaunchEnv(): NodeJS.ProcessEnv { export async function configureGoldenStubAgent( page: Page, options: { - agent?: (typeof GOLDEN_STUB_AGENTS)[number]['id'] + agent?: (typeof GOLDEN_STUB_AGENTS)[number]['id'] | 'grok' agentArgs?: string /** Windows default shell the launch command must survive; ignored elsewhere. */ windowsShell?: BuiltInWindowsTerminalShell diff --git a/tests/e2e/helpers/nested-runtime-same-id-pairing.ts b/tests/e2e/helpers/nested-runtime-same-id-pairing.ts index 5e13630de80..2d8e7d99d6d 100644 --- a/tests/e2e/helpers/nested-runtime-same-id-pairing.ts +++ b/tests/e2e/helpers/nested-runtime-same-id-pairing.ts @@ -28,6 +28,10 @@ export async function replaceRuntimePairingInPlace(args: { if (!store) { throw new Error('Paired desktop store is unavailable during same-ID re-pair') } + const connection = await window.api.runtimeEnvironments.connect({ selector }) + if (!connection.ok) { + throw new Error(`Same-ID re-pair reconnect failed: ${JSON.stringify(connection.error)}`) + } const environments = await window.api.runtimeEnvironments.list() store.getState().setRuntimeEnvironments(environments) if (!(await store.getState().refreshRuntimeEnvironmentStatus(selector))) { diff --git a/tests/e2e/helpers/paired-client-runtime-environment.ts b/tests/e2e/helpers/paired-client-runtime-environment.ts index 2bb64d02974..771499a3ed8 100644 --- a/tests/e2e/helpers/paired-client-runtime-environment.ts +++ b/tests/e2e/helpers/paired-client-runtime-environment.ts @@ -1,4 +1,6 @@ import type { Page } from '@stablyai/playwright-test' +import type { PairedElectronClient, RuntimeDesktopPairingOffer } from './paired-electron-client' +import { revealPairedClientWindow } from './paired-client-window-reveal' /** * Points a freshly launched paired desktop client at the HUB runtime and makes it the active @@ -35,3 +37,71 @@ export async function selectPairedRuntimeEnvironment( return environmentId }, args) } + +export async function rePairPairedElectronClient( + client: PairedElectronClient, + offer: RuntimeDesktopPairingOffer, + name: string +): Promise { + await client.captureDirectSshAttempts() + const environmentId = await client.page.evaluate( + async ({ currentEnvironmentId, name, pairingUrl }) => { + const store = window.__store + if (!store) { + throw new Error('Paired desktop store is unavailable') + } + if (!(await store.getState().setActiveRuntimeEnvironmentPreference(null))) { + throw new Error('Paired desktop could not select local before replacing the HUB') + } + await window.api.runtimeEnvironments.remove({ selector: currentEnvironmentId }) + const result = await window.api.runtimeEnvironments.addFromPairingCode({ + name, + pairingCode: pairingUrl + }) + store.getState().setRuntimeEnvironments(await window.api.runtimeEnvironments.list()) + if (!(await store.getState().refreshRuntimeEnvironmentStatus(result.environment.id))) { + throw new Error('Re-paired desktop could not reach the HUB runtime') + } + if (!(await store.getState().setActiveRuntimeEnvironmentPreference(result.environment.id))) { + throw new Error('Re-paired desktop could not select the HUB runtime') + } + return result.environment.id + }, + { + currentEnvironmentId: client.environmentId, + name, + pairingUrl: offer.pairingUrl + } + ) + client.environmentId = environmentId + // Why: removing and re-adding the same HUB changes the environment identity; remount so no pane keeps the retired transport wrapper. + await client.page.reload() + // Xvfb needs a mapped window to resume actionability frames after reload. + if ( + process.env.GITHUB_ACTIONS === 'true' && + process.platform === 'linux' && + process.env.DISPLAY && + process.env.ORCA_BACKGROUND_LAUNCH !== '1' + ) { + await revealPairedClientWindow(client) + } + await client.page.waitForFunction( + () => window.__store?.getState().workspaceSessionReady === true, + null, + { timeout: 30_000, polling: 100 } + ) + await client.installDirectSshAttemptProbe() + const reachable = await client.page.evaluate(async (nextEnvironmentId) => { + const store = window.__store + if (!store) { + throw new Error('Re-paired desktop store is unavailable after reload') + } + if (!(await store.getState().refreshRuntimeEnvironmentStatus(nextEnvironmentId))) { + return false + } + return store.getState().setActiveRuntimeEnvironmentPreference(nextEnvironmentId) + }, environmentId) + if (!reachable) { + throw new Error('Re-paired desktop could not reach the HUB after reload') + } +} diff --git a/tests/e2e/helpers/paired-client-runtime-environment.unit.test.ts b/tests/e2e/helpers/paired-client-runtime-environment.unit.test.ts new file mode 100644 index 00000000000..3851a9184c1 --- /dev/null +++ b/tests/e2e/helpers/paired-client-runtime-environment.unit.test.ts @@ -0,0 +1,74 @@ +import { afterEach, expect, it, vi } from 'vitest' +import { rePairPairedElectronClient } from './paired-client-runtime-environment' +import type { PairedElectronClient } from './paired-electron-client' + +afterEach(() => { + vi.unstubAllGlobals() + vi.unstubAllEnvs() +}) + +function fixture(canSelectLocal: boolean) { + let selected: string | null = 'old-hub' + const remove = vi.fn(async () => { + if (selected !== null) { + throw new Error('Cannot remove the selected runtime') + } + }) + const state = { + setActiveRuntimeEnvironmentPreference: vi.fn(async (id: string | null) => { + if (id === null && !canSelectLocal) { + return false + } + selected = id + return true + }), + setRuntimeEnvironments: vi.fn(), + refreshRuntimeEnvironmentStatus: vi.fn(async () => true) + } + vi.stubGlobal('window', { + __store: { getState: () => state }, + api: { + runtimeEnvironments: { + remove, + addFromPairingCode: vi.fn(async () => ({ environment: { id: 'new-hub' } })), + list: vi.fn(async () => [{ id: 'new-hub' }]) + } + } + }) + const nativeEvaluate = vi.fn() + const reload = vi.fn(async () => undefined) + const client = { + environmentId: 'old-hub', + captureDirectSshAttempts: vi.fn(async () => undefined), + installDirectSshAttemptProbe: vi.fn(async () => undefined), + app: { evaluate: nativeEvaluate }, + page: { + evaluate: async (callback: (args: unknown) => unknown, args: unknown) => callback(args), + reload, + waitForFunction: vi.fn(async () => undefined) + } + } as unknown as PairedElectronClient + return { client, remove, reload, nativeEvaluate } +} + +it('keeps the old pairing when selecting local fails', async () => { + const { client, remove, reload } = fixture(false) + await expect(rePairPairedElectronClient(client, { pairingUrl: 'code' }, 'HUB')).rejects.toThrow( + 'could not select local' + ) + expect(remove).not.toHaveBeenCalled() + expect(reload).not.toHaveBeenCalled() + expect(client.environmentId).toBe('old-hub') +}) + +it('replaces the active pairing without touching native windows in background mode', async () => { + vi.stubEnv('ORCA_BACKGROUND_LAUNCH', '1') + vi.stubEnv('GITHUB_ACTIONS', 'true') + vi.stubEnv('DISPLAY', ':99') + const { client, remove, reload, nativeEvaluate } = fixture(true) + await rePairPairedElectronClient(client, { pairingUrl: 'code' }, 'HUB') + expect(remove).toHaveBeenCalledWith({ selector: 'old-hub' }) + expect(client.environmentId).toBe('new-hub') + expect(reload).toHaveBeenCalledOnce() + expect(nativeEvaluate).not.toHaveBeenCalled() +}) diff --git a/tests/e2e/helpers/paired-electron-client.ts b/tests/e2e/helpers/paired-electron-client.ts index d08aaebe5f9..a5947bc1228 100644 --- a/tests/e2e/helpers/paired-electron-client.ts +++ b/tests/e2e/helpers/paired-electron-client.ts @@ -25,6 +25,8 @@ import { import { createPairedWebClientUrl, type PairedWebClientOptions } from './paired-web-client-url' import { selectPairedRuntimeEnvironment } from './paired-client-runtime-environment' +export { rePairPairedElectronClient } from './paired-client-runtime-environment' + export type { SameIdPairingReplacement } from './nested-runtime-same-id-pairing' export type PairedElectronClient = { @@ -222,13 +224,13 @@ export async function launchPairedElectronClient( replacementOffer: RuntimeDesktopPairingOffer ): Promise => replaceRuntimePairingInPlace({ - environmentId, + environmentId: client.environmentId, page, pairingUrl: replacementOffer.pairingUrl, userDataDir }) - return { + const client: PairedElectronClient = { app, page, environmentId, @@ -247,6 +249,7 @@ export async function launchPairedElectronClient( replacePairingInPlace, userDataDir } + return client } catch (error) { await closeElectronAppForE2E(app) await cleanupE2EDaemons(userDataDir) @@ -254,59 +257,3 @@ export async function launchPairedElectronClient( throw error } } - -export async function rePairPairedElectronClient( - client: PairedElectronClient, - offer: RuntimeDesktopPairingOffer, - name: string -): Promise { - await client.captureDirectSshAttempts() - const environmentId = await client.page.evaluate( - async ({ currentEnvironmentId, name, pairingUrl }) => { - const store = window.__store - if (!store) { - throw new Error('Paired desktop store is unavailable') - } - await window.api.runtimeEnvironments.remove({ selector: currentEnvironmentId }) - const result = await window.api.runtimeEnvironments.addFromPairingCode({ - name, - pairingCode: pairingUrl - }) - store.getState().setRuntimeEnvironments(await window.api.runtimeEnvironments.list()) - if (!(await store.getState().refreshRuntimeEnvironmentStatus(result.environment.id))) { - throw new Error('Re-paired desktop could not reach the HUB runtime') - } - if (!(await store.getState().setActiveRuntimeEnvironmentPreference(result.environment.id))) { - throw new Error('Re-paired desktop could not select the HUB runtime') - } - return result.environment.id - }, - { - currentEnvironmentId: client.environmentId, - name, - pairingUrl: offer.pairingUrl - } - ) - client.environmentId = environmentId - // Why: removing and re-adding the same HUB changes the environment identity; remount so no pane keeps the retired transport wrapper. - await client.page.reload() - await client.page.waitForFunction( - () => window.__store?.getState().workspaceSessionReady === true, - null, - { timeout: 30_000 } - ) - await client.installDirectSshAttemptProbe() - const reachable = await client.page.evaluate(async (nextEnvironmentId) => { - const store = window.__store - if (!store) { - throw new Error('Re-paired desktop store is unavailable after reload') - } - if (!(await store.getState().refreshRuntimeEnvironmentStatus(nextEnvironmentId))) { - return false - } - return store.getState().setActiveRuntimeEnvironmentPreference(nextEnvironmentId) - }, environmentId) - if (!reachable) { - throw new Error('Re-paired desktop could not reach the HUB after reload') - } -} diff --git a/tests/e2e/helpers/ssh-test-target-connection.ts b/tests/e2e/helpers/ssh-test-target-connection.ts new file mode 100644 index 00000000000..2108b2de96b --- /dev/null +++ b/tests/e2e/helpers/ssh-test-target-connection.ts @@ -0,0 +1,155 @@ +import type { Page } from '@stablyai/playwright-test' +import type { SshTargetCreateInput } from '../../../src/shared/ssh-types' + +export type ConnectedSshTestTarget = { + targetId: string + repoId: string + worktreeId: string +} + +type SshTestConnectionOptions = { + remotePath: string + displayName: string + seedInitialTab?: boolean +} + +export async function connectSshTestTarget( + page: Page, + target: SshTargetCreateInput, + options: SshTestConnectionOptions +): Promise { + return page.evaluate( + async ({ target, remotePath, displayName, seedInitialTab }) => { + const store = window.__store + if (!store) { + throw new Error('Store unavailable') + } + const credentialUnsub = window.api.ssh.onCredentialRequest((request) => { + void window.api.ssh.submitCredential({ requestId: request.requestId, value: null }) + }) + try { + const { target: createdTarget, repoReadoptions } = await window.api.ssh.addTarget({ + target + }) + store.getState().recordSshRepoReadoptions(repoReadoptions) + const state = await window.api.ssh.connect({ targetId: createdTarget.id }) + if (!state || state.status !== 'connected') { + throw new Error(`SSH target did not connect: ${JSON.stringify(state)}`) + } + if ( + !state.providerEpoch || + !Number.isSafeInteger(state.connectionGeneration) || + state.connectionGeneration === undefined || + state.connectionGeneration < 0 + ) { + throw new Error(`SSH target returned incomplete authority: ${JSON.stringify(state)}`) + } + store.getState().setSshConnectionState(createdTarget.id, state) + const labels = new Map(store.getState().sshTargetLabels) + labels.set(createdTarget.id, createdTarget.label) + store.getState().setSshTargetLabels(labels) + const executionHostId = `ssh:${encodeURIComponent(createdTarget.id)}` as const + const authority = { + targetId: createdTarget.id, + providerEpoch: state.providerEpoch, + connectionGeneration: state.connectionGeneration + } + + const result = await window.api.repos.addRemote({ + connectionId: createdTarget.id, + remotePath, + displayName + }) + if ('error' in result) { + throw new Error(result.error) + } + const hasExpectedRepoOwner = (): boolean => + store + .getState() + .repos.some( + (repo) => + repo.id === result.repo.id && + repo.connectionId === createdTarget.id && + repo.executionHostId === executionHostId + ) + const waitForRepoOwner = async (): Promise => { + if (hasExpectedRepoOwner()) { + return + } + await new Promise((resolve, reject) => { + const timer = window.setTimeout(() => { + unsubscribe() + reject(new Error(`Remote repo owner did not hydrate for ${result.repo.path}`)) + }, 15_000) + const unsubscribe = store.subscribe((next) => { + if ( + !next.repos.some( + (repo) => + repo.id === result.repo.id && + repo.connectionId === createdTarget.id && + repo.executionHostId === executionHostId + ) + ) { + return + } + window.clearTimeout(timer) + unsubscribe() + resolve() + }) + }) + } + await store.getState().fetchRepos() + await waitForRepoOwner() + const currentState = store.getState().sshConnectionStates.get(createdTarget.id) + if ( + currentState?.providerEpoch !== authority.providerEpoch || + currentState.connectionGeneration !== authority.connectionGeneration + ) { + throw new Error(`SSH authority rotated before worktree hydration for ${result.repo.path}`) + } + const worktreeResult = await store.getState().fetchWorktrees(result.repo.id, { + executionHostId, + directSshAuthority: authority, + requireAuthoritative: true + }) + if ( + worktreeResult.status !== 'complete' || + worktreeResult.repoId !== result.repo.id || + worktreeResult.authority.kind !== 'direct-ssh' || + worktreeResult.authority.executionHostId !== executionHostId || + worktreeResult.authority.targetId !== authority.targetId || + worktreeResult.authority.providerEpoch !== authority.providerEpoch || + worktreeResult.authority.connectionGeneration !== authority.connectionGeneration + ) { + throw new Error( + `Remote worktree hydration was not authoritative: ${JSON.stringify(worktreeResult)}` + ) + } + const worktree = (store.getState().worktreesByRepo[result.repo.id] ?? []).find( + (candidate) => candidate.hostId === executionHostId + ) + if (!worktree) { + throw new Error(`No remote worktree found for ${result.repo.path}`) + } + store.getState().setActiveWorktree(worktree.id) + if (seedInitialTab && (store.getState().tabsByWorktree[worktree.id] ?? []).length === 0) { + store.getState().createTab(worktree.id) + } + store.getState().setActiveTabType('terminal') + return { + targetId: createdTarget.id, + repoId: result.repo.id, + worktreeId: worktree.id + } + } finally { + credentialUnsub() + } + }, + { + target, + remotePath: options.remotePath, + displayName: options.displayName, + seedInitialTab: options.seedInitialTab ?? true + } + ) +} diff --git a/tests/e2e/helpers/wsl-golden-stub-agent.ts b/tests/e2e/helpers/wsl-golden-stub-agent.ts index 0b09644c14b..94ee5727d93 100644 --- a/tests/e2e/helpers/wsl-golden-stub-agent.ts +++ b/tests/e2e/helpers/wsl-golden-stub-agent.ts @@ -41,7 +41,7 @@ const BACKUP_EXISTING_STUB_SCRIPT = // The marker is written first so stale-lock recovery only removes a stub this helper wrote. const STAGE_SCRIPT = `mkdir -p /usr/local/bin && : > ${WSL_STUB_STAGED_MARKER} && ` + - `printf '#!/bin/sh\\necho GOLDEN_STUB_AGENT_READY\\nexec sleep 3600\\n' > ${WSL_STUB_PATH} && ` + + `printf '#!/bin/sh\\nif [ "$1" = app-server ]; then echo "error: unrecognized subcommand app-server" >&2; exit 2; fi\\necho GOLDEN_STUB_AGENT_READY\\nexec sleep 3600\\n' > ${WSL_STUB_PATH} && ` + `chmod 0755 ${WSL_STUB_PATH}` // The marker is written before the link so a crashed run over-reports rather than leaks a link. diff --git a/tests/e2e/nested-runtime-ssh-lifecycle.spec.ts b/tests/e2e/nested-runtime-ssh-lifecycle.spec.ts index 4b4eb7f21c6..2680d4c3204 100644 --- a/tests/e2e/nested-runtime-ssh-lifecycle.spec.ts +++ b/tests/e2e/nested-runtime-ssh-lifecycle.spec.ts @@ -722,7 +722,7 @@ test('restores a paired nested SSH route after the HUB restarts', async ({ if (!(await store.getState().refreshRuntimeEnvironmentStatus(environmentId))) { return false } - return store.getState().switchRuntimeEnvironment(environmentId) + return store.getState().setActiveRuntimeEnvironmentPreference(environmentId) }, preRestartEnvironmentId) expect(existingPairingRecovered).toBe(true) await reconnectDisconnectedDockerSshRelayTarget(hubLaunch.page, remote.targetId) diff --git a/tests/e2e/source-control-create-pr-intent-notice-layout.spec.ts b/tests/e2e/source-control-create-pr-intent-notice-layout.spec.ts new file mode 100644 index 00000000000..61f3416ec1d --- /dev/null +++ b/tests/e2e/source-control-create-pr-intent-notice-layout.spec.ts @@ -0,0 +1,91 @@ +/** + * The Create PR intent notice pairs a wrapping sentence with a "Source Control + * AI settings" link. In a minimum-width sidebar the link must not share the + * message's row, or it squeezes the sentence into a one-word-per-line column. + */ +import { test, expect } from './helpers/orca-app' +import { waitForActiveWorktree, waitForSessionReady } from './helpers/store' +import { + createStagedCommitMessageChange, + openSourceControl, + seedCreatePrComposer +} from './helpers/source-control-ai-generation' +import { RIGHT_SIDEBAR_MIN_WIDTH } from '../../src/renderer/src/components/right-sidebar/right-sidebar-width' + +test.describe('Source Control Create PR intent notice layout', () => { + test('keeps the settings link off the message row at the minimum sidebar width', async ({ + orcaPage + }) => { + await waitForSessionReady(orcaPage) + await waitForActiveWorktree(orcaPage) + const { prWorktreeId, prWorktreePath, primaryBranch } = await seedCreatePrComposer(orcaPage) + // A real staged change with no commit draft is what routes the intent run + // into the "configure Source Control AI" notice, which carries the link. + createStagedCommitMessageChange(prWorktreePath) + + await orcaPage.evaluate( + ({ prWorktreeId, primaryBranch }) => { + const store = window.__store + if (!store) { + throw new Error('window.__store is not available') + } + store.setState((current) => ({ + // Unconfigured Source Control AI is what routes the run into the + // "configure Source Control AI" notice rather than a failure notice. + settings: current.settings + ? { ...current.settings, sourceControlAi: undefined, commitMessageAi: undefined } + : current.settings, + repos: current.repos.map((repo) => ({ ...repo, sourceControlAi: undefined })), + // Blocked-on-push keeps "Create PR" running the intent flow instead of + // opening the composer form. + getHostedReviewCreationEligibility: async () => ({ + provider: 'github' as const, + review: null, + reviewLookupOutcome: 'not_found' as const, + canCreate: false, + blockedReason: 'needs_push' as const, + nextAction: 'push' as const, + defaultBaseRef: primaryBranch, + head: 'e2e-secondary' + }), + fetchHostedReviewForBranch: async () => null, + fetchPRForBranch: async () => null, + pushBranch: async (worktreeId: string) => { + if (worktreeId !== prWorktreeId) { + throw new Error(`Create PR intent pushed unexpected worktree ${worktreeId}`) + } + } + })) + }, + { prWorktreeId, primaryBranch } + ) + + await openSourceControl(orcaPage, prWorktreeId) + await orcaPage.evaluate((minWidth) => { + window.__store?.getState().setRightSidebarWidth(minWidth) + }, RIGHT_SIDEBAR_MIN_WIDTH) + + const createPr = orcaPage.getByRole('button', { name: 'Create PR' }).first() + await expect(createPr).toBeVisible({ timeout: 10_000 }) + await expect(createPr).toBeEnabled() + await createPr.click() + + const notice = orcaPage.locator('#commit-area-create-pr-intent') + const settingsLink = notice.getByRole('button', { name: 'Source Control AI settings' }) + await expect(settingsLink).toBeVisible({ timeout: 20_000 }) + + if (process.env.ORCA_PR_INTENT_NOTICE_SCREENSHOT_PATH) { + await orcaPage.evaluate(() => document.documentElement.classList.add('dark')) + await notice.screenshot({ path: process.env.ORCA_PR_INTENT_NOTICE_SCREENSHOT_PATH }) + } + + // The layout contract: the link starts below the message's last line. + const messageBox = await notice.locator('span').first().boundingBox() + const linkBox = await settingsLink.boundingBox() + expect(messageBox).not.toBeNull() + expect(linkBox).not.toBeNull() + expect(linkBox!.y).toBeGreaterThanOrEqual(messageBox!.y + messageBox!.height) + // A squeezed message wraps far taller than the ~4 lines this sentence needs. + expect(messageBox!.height).toBeLessThan(70) + }) +}) diff --git a/tests/e2e/source-control-large-file-count.spec.ts b/tests/e2e/source-control-large-file-count.spec.ts index 85f3710b308..c8c699b2bd8 100644 --- a/tests/e2e/source-control-large-file-count.spec.ts +++ b/tests/e2e/source-control-large-file-count.spec.ts @@ -31,6 +31,7 @@ import { removeLargeFileCountUntrackedTree } from './large-file-count-fixtures' import { DEFAULT_GIT_STATUS_LIMIT } from '../../src/shared/git-status-limit' +import { RIGHT_SIDEBAR_MIN_WIDTH } from '../../src/renderer/src/components/right-sidebar/right-sidebar-width' // Matches the large-diff freeze budget: a blocking stall past 1s is the // "UI becomes unresponsive" symptom reported in #8013. @@ -416,12 +417,17 @@ test.describe('Source Control large file count (#8013)', () => { rendererWorkingSetMb: { before: workingSetBeforeMb, after: workingSetAfterMb } }) - const tooManyChangesBanner = orcaPage.getByText('Too many changes detected.', { - exact: false - }) + const tooManyChangesBanner = orcaPage.getByTestId('too-many-changes-banner') await expect(tooManyChangesBanner).toBeVisible() if (process.env.ORCA_LARGE_FILE_SCREENSHOT_PATH) { - await orcaPage.screenshot({ path: process.env.ORCA_LARGE_FILE_SCREENSHOT_PATH }) + // Narrowest supported sidebar is where the banner layout is worst. + await orcaPage.evaluate((minWidth) => { + window.__store?.getState().setRightSidebarWidth(minWidth) + document.documentElement.classList.add('dark') + }, RIGHT_SIDEBAR_MIN_WIDTH) + await tooManyChangesBanner.screenshot({ + path: process.env.ORCA_LARGE_FILE_SCREENSHOT_PATH + }) } expect(measurement.didHitLimit).toBe(true) @@ -439,7 +445,7 @@ test.describe('Source Control large file count (#8013)', () => { ) expect(hugeState).not.toBeNull() - const retryButton = tooManyChangesBanner.locator('..').getByRole('button', { name: 'Retry' }) + const retryButton = tooManyChangesBanner.getByRole('button', { name: 'Retry' }) await expect(retryButton).toBeVisible() // Keep automatic refreshes from removing Retry before its real request starts. await installGitStatusRetryBarrier(electronApp, fixture.repoPath) diff --git a/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts b/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts index 2de70c199d3..4f51b346b91 100644 --- a/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts +++ b/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts @@ -53,32 +53,9 @@ function continuousFloodCommand(runId: string, index: number): string { test.describe('R2 Docker SSH bulk-open freeze', () => { test.skip(!RUN_DOCKER_SSH, 'Set ORCA_E2E_SSH_DOCKER=1 to run Docker SSH freeze repro') - // Fixme: un-rotted and measurable, but its oracle is wall-clock and does not survive a change of - // host, so it cannot gate. Three runs of the same measurement path: - // - // host hiddenFlood bulkOpen interaction - // developer workstation 2.1ms 41.5ms 53.6ms - // GitHub ubuntu runner A 1.5ms 2575.6ms 3464.2ms - // GitHub ubuntu runner B 0.2ms 397.4ms 3386.7ms - // - // Two separate problems, and neither is the product. `bulkOpenMaxLagMs` swings 6.5x between two - // CI runs of the same code, so a fixed threshold on it is a coin flip; `interactionProbeMs` sits - // stably ~64x over the workstation figure, because it times two `setActiveView` round trips - // through a double rAF — a view remount cost, not the renderer freeze #16764 reports. It shares - // SOFT/HARD_FREEZE_LAG_MS with the lag probe only because both are milliseconds. `hardFreeze` - // has never tripped on any host; the failure is always the soft budget. - // - // Not converted to a ratio against a calibration run: with a 6.5x within-host swing on the very - // quantity that would be normalized, a threshold picked from three samples is the same arbitrary - // constant in dimensionless clothing. Gating needs a distribution first. - // - // Kept executable rather than deleted: flip `test.fixme` back to `test` to run it, which is how - // the numbers above were taken. Tracked in stablyai/orca#16764. - // - // The cost is real and is recorded in run-ssh-docker-e2e.mjs: 5 simultaneously flooding SSH panes - // exercise writer saturation, ACK/credit accounting and per-pane polling together, and nothing - // else covers that combination. It is a gap, not coverage living somewhere else. - test.fixme('bulk-open many flooding SSH terminals and measure renderer lag @freeze-repro', async ({ + // Headless Linux disables compositing and schedules idle RAFs ~1s apart; use headed CI. + // Headed SwiftShader restores ~16ms frames without changing the freeze budgets. + test('bulk-open many flooding SSH terminals and measure renderer lag @freeze-repro @headful', async ({ orcaPage, registerPostElectronShutdownCleanup }, testInfo) => { diff --git a/tests/e2e/ssh-localhost.spec.ts b/tests/e2e/ssh-localhost.spec.ts index 1117fcb9409..00d2d461967 100644 --- a/tests/e2e/ssh-localhost.spec.ts +++ b/tests/e2e/ssh-localhost.spec.ts @@ -1,4 +1,7 @@ +import { connectSshTestTarget } from './helpers/ssh-test-target-connection' import os from 'node:os' +import { createSeededTestRepo } from './helpers/seeded-test-repo' +import { cleanupTestRepository } from './global-teardown' import type { Page } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' @@ -151,91 +154,28 @@ test.describe('Localhost SSH', () => { test('routes a terminal and agent-hook status over localhost SSH', async ({ orcaPage, - testRepoPath + registerPostElectronShutdownCleanup }) => { test.slow() + // The relay persists workspace sessions by path across fresh client profiles. + const testRepoPath = createSeededTestRepo({ publishPath: false }) + registerPostElectronShutdownCleanup(async () => cleanupTestRepository(testRepoPath)) await waitForSessionReady(orcaPage) await waitForActiveWorktree(orcaPage) const target = readLocalhostSshTarget() - const remote = await orcaPage.evaluate( - async ({ remotePath, target }) => { - const store = window.__store - if (!store) { - throw new Error('Store unavailable') - } - - const credentialUnsub = window.api.ssh.onCredentialRequest((request) => { - void window.api.ssh.submitCredential({ requestId: request.requestId, value: null }) - }) - - try { - const { target: createdTarget, repoReadoptions } = await window.api.ssh.addTarget({ - target: { - ...target, - // Why: local-only E2E should not leave a long-lived relay process - // behind if the Electron app is killed between cleanup hooks. - relayGracePeriodSeconds: 1 - } - }) - store.getState().recordSshRepoReadoptions(repoReadoptions) - - let state - try { - state = await window.api.ssh.connect({ targetId: createdTarget.id }) - } catch (err) { - const message = err instanceof Error ? err.message : String(err) - throw new Error( - `Failed to connect to localhost SSH target ${target.username}@${target.host || target.configHost}:${target.port}. ` + - `Ensure sshd is running and key/agent auth is non-interactive. ${message}` - ) - } - - if (!state || state.status !== 'connected') { - throw new Error(`SSH target did not reach connected state: ${JSON.stringify(state)}`) - } - - store.getState().setSshConnectionState(createdTarget.id, state) - const labels = new Map(store.getState().sshTargetLabels) - labels.set(createdTarget.id, createdTarget.label) - store.getState().setSshTargetLabels(labels) - - const result = await window.api.repos.addRemote({ - connectionId: createdTarget.id, - remotePath, - displayName: 'Localhost SSH E2E' - }) - if ('error' in result) { - throw new Error(result.error) - } - - await store.getState().fetchRepos() - await store.getState().fetchWorktrees(result.repo.id) - - const worktrees = store.getState().worktreesByRepo[result.repo.id] ?? [] - const worktree = - worktrees.find((candidate) => candidate.path === result.repo.path) ?? worktrees[0] - if (!worktree) { - throw new Error(`No remote worktree found for ${result.repo.path}`) - } - - store.getState().setActiveWorktree(worktree.id) - if ((store.getState().tabsByWorktree[worktree.id] ?? []).length === 0) { - store.getState().createTab(worktree.id) - } - store.getState().setActiveTabType('terminal') - - return { - targetId: createdTarget.id, - repoId: result.repo.id, - worktreeId: worktree.id - } - } finally { - credentialUnsub() - } - }, - { remotePath: testRepoPath, target } - ) + const remote = await connectSshTestTarget( + orcaPage, + // Limit orphan relay lifetime if the test app exits before cleanup. + { ...target, relayGracePeriodSeconds: 1 }, + { remotePath: testRepoPath, displayName: 'Localhost SSH E2E' } + ).catch((error: unknown) => { + throw new Error( + `Failed to prepare localhost SSH target ${target.username}@${target.host || target.configHost}:${target.port}. ` + + `Ensure sshd is running and key/agent auth is non-interactive. ${String(error)}`, + { cause: error } + ) + }) await expect(remote.targetId).toBeTruthy() await ensureTerminalVisible(orcaPage, 30_000) diff --git a/tests/e2e/terminal-windows-codex-multiline-paste.spec.ts b/tests/e2e/terminal-windows-codex-multiline-paste.spec.ts index 543f6072247..70d11602adc 100644 --- a/tests/e2e/terminal-windows-codex-multiline-paste.spec.ts +++ b/tests/e2e/terminal-windows-codex-multiline-paste.spec.ts @@ -2,6 +2,7 @@ import { createHash, randomUUID } from 'node:crypto' import { rmSync, writeFileSync } from 'node:fs' import path from 'node:path' import { test, expect } from './helpers/orca-app' +import { attachRepoAndOpenTerminal } from './helpers/orca-restart' import { focusActiveTerminalInput, getTerminalContent, @@ -54,33 +55,8 @@ async function activateTestRepository( page: Parameters[0], repoPath: string ): Promise { - await page.evaluate(async (targetRepoPath) => { - const normalizePath = (value: string): string => value.replaceAll('\\', '/').toLowerCase() - await window.api.repos.add({ path: targetRepoPath }) - const store = window.__store - if (!store) { - throw new Error('Orca store unavailable') - } - await store.getState().fetchRepos() - const repo = store - .getState() - .repos.find((candidate) => normalizePath(candidate.path) === normalizePath(targetRepoPath)) - if (!repo) { - throw new Error('Seeded repository unavailable') - } - await store.getState().updateRepo(repo.id, { externalWorktreeVisibility: 'show' }) - await store.getState().fetchWorktrees(repo.id) - const worktree = store - .getState() - .worktreesByRepo[repo.id]?.find( - (candidate) => normalizePath(candidate.path) === normalizePath(targetRepoPath) - ) - if (!worktree) { - throw new Error('Seeded worktree unavailable') - } - store.getState().setActiveWorktree(worktree.id) - store.getState().createTab(worktree.id) - }, repoPath) + const worktreeId = await attachRepoAndOpenTerminal(page, repoPath) + await page.evaluate((id) => window.__store!.getState().createTab(id), worktreeId) } function pasteCollectorScript( diff --git a/tests/e2e/terminal-windows-conpty-keyboard-reset.spec.ts b/tests/e2e/terminal-windows-conpty-keyboard-reset.spec.ts index 5766c1e2d49..6c384fd6d2c 100644 --- a/tests/e2e/terminal-windows-conpty-keyboard-reset.spec.ts +++ b/tests/e2e/terminal-windows-conpty-keyboard-reset.spec.ts @@ -54,13 +54,19 @@ test('resets standard keyboard bytes after a protocol-mode agent exits on ConPTY await waitForSessionReady(orcaPage) await waitForActiveWorktree(orcaPage) await ensureTerminalVisible(orcaPage) - await configureGoldenStubAgent(orcaPage, { agentArgs: '--keyboard-protocol' }) - await launchGoldenStubAgentFromNewTab(orcaPage) + // Grok is the supported native ConPTY exception to Kitty protocol withholding. + await configureGoldenStubAgent(orcaPage, { + agent: 'grok', + agentArgs: '--keyboard-protocol --grok' + }) + await launchGoldenStubAgentFromNewTab(orcaPage, /^Grok(?:\s|$)/i) const ptyId = await waitForActivePanePtyId(orcaPage) await expect.poll(() => getKittyKeyboardFlags(orcaPage), { timeout: 10_000 }).toBe(1) await clearTerminalPtyWriteLog(electronApp) + // Kitty flag 1 preserves plain Enter; modified Enter proves CSI-u input. + await orcaPage.keyboard.press('Shift+Enter') await orcaPage.keyboard.type('exit') await orcaPage.keyboard.press('Enter') await waitForTerminalOutput(orcaPage, GOLDEN_STUB_EXIT_MARKER, 15_000) @@ -68,7 +74,8 @@ test('resets standard keyboard bytes after a protocol-mode agent exits on ConPTY .filter((entry) => entry.id === ptyId) .map((entry) => entry.data) .join('') - expect(protocolWrites.includes('\x1b[13u') || protocolWrites.includes('\x1b[13;1u')).toBe(true) + expect(protocolWrites).toContain('\x1b[13;2u') + expect(protocolWrites).toContain('\r') await expect.poll(() => getKittyKeyboardFlags(orcaPage), { timeout: 10_000 }).toBe(0) await clearTerminalPtyWriteLog(electronApp) diff --git a/tests/e2e/terminal-windows-shell-paste-ownership.spec.ts b/tests/e2e/terminal-windows-shell-paste-ownership.spec.ts index 43fad9122a2..ecaf5baf8bf 100644 --- a/tests/e2e/terminal-windows-shell-paste-ownership.spec.ts +++ b/tests/e2e/terminal-windows-shell-paste-ownership.spec.ts @@ -221,7 +221,8 @@ test.describe('Windows terminal shell paste ownership', () => { `mixed-newline-before\r\nlf-line\ncrlf-line\r\n${sentinel}` ].join('\n') const scriptPath = path.join(testRepoPath, `.orca-paste-powershell-shell-${runId}.mjs`) - writeFileSync(scriptPath, pasteCollectScript(runId, sentinel, payload)) + const expectedText = payload.replace(/\r?\n/g, '\r') + writeFileSync(scriptPath, pasteCollectScript(runId, sentinel, expectedText)) let scriptStarted = false try { @@ -237,7 +238,7 @@ test.describe('Windows terminal shell paste ownership', () => { await waitForTerminalOutput(orcaPage, `PASTE_COMPLETE_${runId}:MATCH`, 10_000, 12_000) const writes = (await readTerminalPtyWrites(electronApp)).join('') - expect(countOccurrences(writes, payload), 'PowerShell payload PTY write count').toBe(1) + expect(countOccurrences(writes, expectedText), 'PowerShell payload PTY write count').toBe(1) } finally { if (scriptStarted) { await sendToTerminal(orcaPage, ptyId, '\x03').catch(() => undefined) @@ -271,7 +272,8 @@ test.describe('Windows terminal shell paste ownership', () => { `mixed-newline-before\r\nlf-line\ncrlf-line\r\n${sentinel}` ].join('\n') const scriptPath = path.join(testRepoPath, `.orca-paste-cmd-shell-${runId}.mjs`) - writeFileSync(scriptPath, pasteCollectScript(runId, sentinel, payload)) + const expectedText = payload.replace(/\r?\n/g, '\r') + writeFileSync(scriptPath, pasteCollectScript(runId, sentinel, expectedText)) let scriptStarted = false try { @@ -287,7 +289,7 @@ test.describe('Windows terminal shell paste ownership', () => { await waitForTerminalOutput(orcaPage, `PASTE_COMPLETE_${runId}:MATCH`, 10_000, 12_000) const writes = (await readTerminalPtyWrites(electronApp)).join('') - expect(countOccurrences(writes, payload), 'cmd.exe payload PTY write count').toBe(1) + expect(countOccurrences(writes, expectedText), 'cmd.exe payload PTY write count').toBe(1) } finally { if (scriptStarted) { await sendToTerminal(orcaPage, ptyId, '\x03').catch(() => undefined) @@ -322,7 +324,8 @@ test.describe('Windows terminal shell paste ownership', () => { `mixed-newline-before\r\nlf-line\ncrlf-line\r\n${sentinel}` ].join('\n') const scriptPath = path.join(testRepoPath, `.orca-paste-git-bash-shell-${runId}.mjs`) - writeFileSync(scriptPath, pasteCollectScript(runId, sentinel, payload)) + const expectedText = payload.replace(/\r?\n/g, '\r') + writeFileSync(scriptPath, pasteCollectScript(runId, sentinel, expectedText)) let scriptStarted = false try { @@ -338,7 +341,7 @@ test.describe('Windows terminal shell paste ownership', () => { await waitForTerminalOutput(orcaPage, `PASTE_COMPLETE_${runId}:MATCH`, 10_000, 12_000) const writes = (await readTerminalPtyWrites(electronApp)).join('') - expect(countOccurrences(writes, payload), 'Git Bash payload PTY write count').toBe(1) + expect(countOccurrences(writes, expectedText), 'Git Bash payload PTY write count').toBe(1) } finally { if (scriptStarted) { await sendToTerminal(orcaPage, ptyId, '\x03').catch(() => undefined) @@ -376,7 +379,8 @@ test.describe('Windows terminal shell paste ownership', () => { `mixed-newline-before\r\nlf-line\ncrlf-line\r\n${sentinel}` ].join('\n') const scriptPath = path.join(testRepoPath, `.orca-paste-wsl-shell-${runId}.mjs`) - writeFileSync(scriptPath, pasteCollectScript(runId, sentinel, payload)) + const expectedText = payload.replace(/\r?\n/g, '\r') + writeFileSync(scriptPath, pasteCollectScript(runId, sentinel, expectedText)) let scriptStarted = false try { @@ -396,7 +400,7 @@ test.describe('Windows terminal shell paste ownership', () => { await waitForTerminalOutput(orcaPage, `PASTE_COMPLETE_${runId}:MATCH`, 10_000, 12_000) const writes = (await readTerminalPtyWrites(electronApp)).join('') - expect(countOccurrences(writes, payload), 'WSL payload PTY write count').toBe(1) + expect(countOccurrences(writes, expectedText), 'WSL payload PTY write count').toBe(1) } finally { if (scriptStarted) { await sendToTerminal(orcaPage, ptyId, '\x03').catch(() => undefined) @@ -419,10 +423,6 @@ test.describe('Windows terminal shell paste ownership', () => { const wslDistro = await configureActiveProjectWslRuntime(orcaPage) test.skip(!wslDistro, 'No WSL distro is available on this Windows host') const tabId = await createWindowsProjectRuntimeTerminalTab(orcaPage, 'wsl.exe') - await updateWindowsDefaultShellSetting(orcaPage, 'cmd.exe') - await expect( - orcaPage.locator(`[data-testid="sortable-tab"][data-tab-id="${tabId}"] [data-shell-icon]`) - ).toHaveAttribute('data-shell-icon', 'wsl.exe') await waitForActiveTerminalManager(orcaPage, 30_000) await installTerminalPtyWriteSpy(electronApp) @@ -437,7 +437,8 @@ test.describe('Windows terminal shell paste ownership', () => { `mixed-newline-before\r\nlf-line\ncrlf-line\r\n${sentinel}` ].join('\n') const scriptPath = path.join(testRepoPath, `.orca-paste-wsl-retention-${runId}.mjs`) - writeFileSync(scriptPath, pasteCollectScript(runId, sentinel, payload)) + const expectedText = payload.replace(/\r?\n/g, '\r') + writeFileSync(scriptPath, pasteCollectScript(runId, sentinel, expectedText)) let scriptStarted = false try { @@ -449,6 +450,13 @@ test.describe('Windows terminal shell paste ownership', () => { scriptStarted = true await waitForTerminalOutput(orcaPage, `PASTE_READY_${runId}`, 10_000) + // Exercise a live WSL process across the settings change. + await updateWindowsDefaultShellSetting(orcaPage, 'cmd.exe') + await expect( + orcaPage.locator(`[data-testid="sortable-tab"][data-tab-id="${tabId}"] [data-shell-icon]`) + ).toHaveAttribute('data-shell-icon', 'wsl.exe') + expect(await waitForActivePanePtyId(orcaPage)).toBe(ptyId) + await clearTerminalPtyWriteLog(electronApp) await orcaPage.evaluate((text) => window.api.ui.writeClipboardText(text), payload) await focusActiveTerminalInput(orcaPage) @@ -457,7 +465,7 @@ test.describe('Windows terminal shell paste ownership', () => { await waitForTerminalOutput(orcaPage, `PASTE_COMPLETE_${runId}:MATCH`, 10_000, 12_000) const writes = (await readTerminalPtyWrites(electronApp)).join('') - expect(countOccurrences(writes, payload), 'retained WSL payload PTY write count').toBe(1) + expect(countOccurrences(writes, expectedText), 'retained WSL payload PTY write count').toBe(1) } finally { if (scriptStarted) { await sendToTerminal(orcaPage, ptyId, '\x03').catch(() => undefined) diff --git a/tests/e2e/windows-terminal-env-icons.spec.ts b/tests/e2e/windows-terminal-env-icons.spec.ts index 85080c3162a..c6d41776d3a 100644 --- a/tests/e2e/windows-terminal-env-icons.spec.ts +++ b/tests/e2e/windows-terminal-env-icons.spec.ts @@ -1,4 +1,5 @@ import { test, expect } from './helpers/orca-app' +import { getFirstWslDistro, useWslRuntimeForActiveProject } from './helpers/wsl-golden-stub-agent' import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' import { execInTerminal, @@ -27,7 +28,9 @@ test.describe('Windows terminal env and shell identity', () => { await waitForTerminalOutput(orcaPage, marker, 15_000) }) - test('Windows tab icons stay pinned to the shell used at tab creation', async ({ orcaPage }) => { + test('native Windows tab icons stay pinned to the effective shell at tab creation', async ({ + orcaPage + }) => { test.skip(process.platform !== 'win32', 'Windows shell icons only render on Windows') const tabIds = await orcaPage.evaluate(() => { @@ -41,10 +44,11 @@ test.describe('Windows terminal env and shell identity', () => { throw new Error('No active worktree') } + // Native project ownership makes a global WSL shell fall back to PowerShell. store.setState({ settings: { ...state.settings!, terminalWindowsShell: 'wsl.exe' } }) - const wslTab = store.getState().createTab(worktreeId, undefined, undefined, { + const fallbackTab = store.getState().createTab(worktreeId, undefined, undefined, { activate: false }) @@ -55,33 +59,68 @@ test.describe('Windows terminal env and shell identity', () => { activate: false }) - return { wslTabId: wslTab.id, cmdTabId: cmdTab.id } + return { fallbackTabId: fallbackTab.id, cmdTabId: cmdTab.id } }) - const tabSnapshot = await orcaPage.evaluate(({ wslTabId, cmdTabId }) => { + const tabSnapshot = await orcaPage.evaluate(({ fallbackTabId, cmdTabId }) => { const state = window.__store!.getState() const tabs = Object.values(state.tabsByWorktree).flat() return { - wslShell: tabs.find((tab) => tab.id === wslTabId)?.shellOverride, + fallbackShell: tabs.find((tab) => tab.id === fallbackTabId)?.shellOverride, cmdShell: tabs.find((tab) => tab.id === cmdTabId)?.shellOverride } }, tabIds) expect(tabSnapshot).toEqual({ - wslShell: 'wsl.exe', + fallbackShell: 'powershell.exe', cmdShell: 'cmd.exe' }) - const wslTab = orcaPage.locator( - `[data-testid="sortable-tab"][data-tab-id="${tabIds.wslTabId}"]` + const fallbackTab = orcaPage.locator( + `[data-testid="sortable-tab"][data-tab-id="${tabIds.fallbackTabId}"]` ) const cmdTab = orcaPage.locator( `[data-testid="sortable-tab"][data-tab-id="${tabIds.cmdTabId}"]` ) - await expect(wslTab).toBeVisible() + await expect(fallbackTab).toBeVisible() await expect(cmdTab).toBeVisible() - await expect(wslTab.locator('[data-shell-icon]')).toHaveAttribute('data-shell-icon', 'wsl.exe') + await expect(fallbackTab.locator('[data-shell-icon]')).toHaveAttribute( + 'data-shell-icon', + 'powershell.exe' + ) await expect(cmdTab.locator('[data-shell-icon]')).toHaveAttribute('data-shell-icon', 'cmd.exe') }) + + test('WSL project tab icons retain runtime ownership across global shell changes', async ({ + orcaPage + }) => { + test.skip(process.platform !== 'win32', 'WSL shell icons require Windows') + const distro = await getFirstWslDistro(orcaPage) + test.skip(!distro, 'WSL icon coverage requires an installed distro') + await useWslRuntimeForActiveProject(orcaPage, distro!) + + const tabIds = await orcaPage.evaluate(async () => { + const store = window.__store! + const worktreeId = store.getState().activeWorktreeId! + const ids: string[] = [] + for (const shell of ['powershell.exe', 'cmd.exe'] as const) { + await store.getState().updateSettings({ terminalWindowsShell: shell }) + ids.push( + store.getState().createTab(worktreeId, undefined, undefined, { activate: false }).id + ) + } + return ids + }) + const shells = await orcaPage.evaluate((ids) => { + const tabs = Object.values(window.__store!.getState().tabsByWorktree).flat() + return ids.map((id) => tabs.find((tab) => tab.id === id)?.shellOverride) + }, tabIds) + expect(shells).toEqual(['wsl.exe', 'wsl.exe']) + for (const id of tabIds) { + const tab = orcaPage.locator(`[data-testid="sortable-tab"][data-tab-id="${id}"]`) + await expect(tab).toBeVisible() + await expect(tab.locator('[data-shell-icon]')).toHaveAttribute('data-shell-icon', 'wsl.exe') + } + }) })