diff --git a/.gitattributes b/.gitattributes index 2a99890023b..8f4f884295d 100644 --- a/.gitattributes +++ b/.gitattributes @@ -8,7 +8,19 @@ /src/cli/bundled-skill-guides.ts text eol=lf # Bundled plugin trees are byte-hashed; CRLF checkout would break the pinned hash. /resources/plugins/** text eol=lf -# pnpm hashes every patch byte-for-byte, so a CRLF checkout breaks the install. +# Relay assets are copied verbatim into the bundle and hashed byte-for-byte into +# .version, which names the immutable remote install dir. A CRLF checkout makes a +# Windows-built client disagree with a mac/Linux-built one on the same release, +# so one host ends up with two relay trees (#17886 review). +/config/relay-assets/** text eol=lf +# Pin the bytes so a patch reads and diffs identically on every host. It is NOT +# what makes the hash right: pnpm hashes a patch LF-normalized, so a CRLF checkout +# cannot change it. Believing otherwise put a hand-computed raw digest in the +# lockfile twice and broke every install (#17886). +# These files are stored LF, which is not always the encoding they were written +# against -- @vscode/windows-process-tree ships CRLF sources -- so any code that +# runs `git apply` on one must force `-c core.autocrlf=input` rather than trust +# the host's setting. See config/scripts/windows-process-tree-gyp-rebuild.mjs. /config/patches/*.patch -text # The xterm bundle hunks also make a diff nobody can read; review the hand-written # source patch under xterm-src/ instead. The sibling patches stay diffable. diff --git a/.github/actions/install-node-dependencies/action.yml b/.github/actions/install-node-dependencies/action.yml index e36ec4c65d8..7695d2bec9b 100644 --- a/.github/actions/install-node-dependencies/action.yml +++ b/.github/actions/install-node-dependencies/action.yml @@ -77,14 +77,6 @@ runs: ;; esac - # pnpm's bundled gyp_main.py is not executable on fresh Linux runners. - - name: Use external node-gyp - if: runner.os == 'Linux' && inputs.native-runtime != 'none' - shell: bash - run: | - npm install -g node-gyp@11.5.0 - echo "npm_config_node_gyp=$(npm root -g)/node-gyp/bin/node-gyp.js" >> "$GITHUB_ENV" - - name: Prepare dependency install shell: bash run: | @@ -175,6 +167,22 @@ runs: node_modules/.pnpm/@vscode+windows-process-tree@*/node_modules/@vscode/windows-process-tree/build key: native-modules-${{ runner.os }}-${{ steps.native-cache-scope.outputs.scope }}-${{ runner.arch }}-${{ inputs.native-runtime }}-node${{ steps.requested-node.outputs.node-version || steps.default-node.outputs.node-version }}-${{ hashFiles('pnpm-lock.yaml', '.github/actions/install-node-dependencies/action.yml', 'config/scripts/ensure-native-runtime.mjs', 'config/scripts/rebuild-native-deps.mjs', 'config/patches/node-pty@1.1.0.patch', 'config/patches/@vscode__windows-process-tree@0.8.0.patch') }} + # pnpm's bundled gyp_main.py is not executable on fresh Linux runners. + - name: Use external node-gyp + if: runner.os == 'Linux' && inputs.native-runtime != 'none' + shell: bash + env: + NATIVE_RUNTIME: ${{ inputs.native-runtime }} + NATIVE_CACHE_HIT: ${{ steps.native-cache-restore.outputs.cache-hit || steps.native-cache-restore-only.outputs.cache-hit }} + run: | + # A cache hit can contain unusable addons; probe before skipping the rebuild toolchain. + if [ "$NATIVE_RUNTIME" = node ] && [ "$NATIVE_CACHE_HIT" = true ] && + node config/scripts/ensure-native-runtime.mjs --check-only; then + exit 0 + fi + npm install -g node-gyp@11.5.0 + echo "npm_config_node_gyp=$(npm root -g)/node-gyp/bin/node-gyp.js" >> "$GITHUB_ENV" + - name: Prepare native runtime if: inputs.native-runtime != 'none' shell: bash diff --git a/.github/scripts/e2e-with-window-manager.sh b/.github/scripts/e2e-with-window-manager.sh new file mode 100644 index 00000000000..d431a039809 --- /dev/null +++ b/.github/scripts/e2e-with-window-manager.sh @@ -0,0 +1,26 @@ +#!/usr/bin/env bash +set -euo pipefail +openbox --sm-disable > /tmp/orca-e2e-window-manager.log 2>&1 & +wm_pid=$! +cleanup() { + kill "$wm_pid" 2>/dev/null || true + wait "$wm_pid" 2>/dev/null || true +} +trap cleanup EXIT +ready=false +for attempt in {1..100}; do + if xprop -root _NET_SUPPORTING_WM_CHECK 2>/dev/null | rg -q 'window id # 0x[1-9a-fA-F]'; then + ready=true + break + fi + if ! kill -0 "$wm_pid" 2>/dev/null; then + cat /tmp/orca-e2e-window-manager.log + exit 1 + fi + sleep 0.1 +done +if [ "$ready" != true ]; then + echo 'Window manager did not acquire the Xvfb root window' >&2 + exit 1 +fi +"$@" diff --git a/.github/workflows/adhoc-mac-build.yml b/.github/workflows/adhoc-mac-build.yml index d7dd6d5ffb6..3e17eee9b68 100644 --- a/.github/workflows/adhoc-mac-build.yml +++ b/.github/workflows/adhoc-mac-build.yml @@ -127,9 +127,12 @@ jobs: esac # Bare: a work-tree repo refuses to fetch over its own checked-out # branch. tree:0 keeps the fetch to the commit graph — no trees, no - # blobs — so this stays cheap next to the build it fronts. + # blobs — so this stays cheap next to the build it fronts. reftable + # because this repo has branches that differ only in casing, and the + # files backend cannot store both on a case-insensitive runner disk — + # it fails the entire fetch, not just the one ref. scratch="$RUNNER_TEMP/vet-requested-ref" - git init -q --bare "$scratch" + git init -q --bare --ref-format=reftable "$scratch" git -C "$scratch" fetch -q --filter=tree:0 "$REPO_URL" '+refs/heads/*:refs/heads/*' '+refs/tags/*:refs/tags/*' # Branch first to keep actions/checkout's old tie-break: bare # rev-parse would prefer the tag when a branch shares its name. @@ -157,6 +160,9 @@ jobs: - name: Checkout the requested ref uses: actions/checkout@v6 + env: + # Full-history checkout must also preserve case-twin branch and tag names. + GIT_DEFAULT_REF_FORMAT: reftable with: # Why an input at all rather than just github.ref: the whole point is to # build code that has not landed, and the workflow definition itself diff --git a/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml b/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml index 6430c3f4793..8ef61507088 100644 --- a/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml +++ b/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml @@ -91,7 +91,11 @@ jobs: test -n "${CAPACITY_SERVICE_ACCOUNT}" test -n "${DIRECTOR_RUNTIME_SERVICE_ACCOUNT}" + # Full history: the monitor evidence this job verifies is sealed at an ancestor commit, + # and the provenance check fails closed on a commit a shallow clone left out. - uses: actions/checkout@v4 + with: + fetch-depth: 0 - uses: pnpm/action-setup@v4 with: { package_json_file: cloud/package.json } @@ -177,11 +181,12 @@ jobs: env: ORCA_RELAY_ADMIN_ID_TOKEN: ${{ steps.deploy-auth.outputs.id_token }} run: | - RETRY_ARGS=() - if test "${WAVE_INDEX}" != 0; then RETRY_ARGS=(--retry-freshness); fi + # Freshness-only failures are publish lag, not health, on every wave + # including the first; the CLI still caps the retry at the wave's + # evidence-age budget, so this cannot mutate on aged evidence. pnpm incident:relay-preflight -- \ --state-file "${OUTPUT_DIRECTORY}/relay-${MONITOR_RUN_ID}-dry-run.state.json" \ - --wave-index "${WAVE_INDEX}" "${RETRY_ARGS[@]}" + --wave-index "${WAVE_INDEX}" --retry-freshness - name: Require durable rehome disabled and exact selector env: @@ -271,10 +276,22 @@ jobs: env: ORCA_RELAY_ADMIN_ID_TOKEN: ${{ steps.deploy-auth.outputs.id_token }} run: | - CURRENT_RUNTIME="$(curl --fail-with-body --max-time 30 \ - --request POST "${CELL_ORIGIN}/v1/admin/runtime-status" \ - --header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \ - --header 'Content-Type: application/json' --data '{"v":1}')" + # A single transient 5xx (LB warm-up behind a fresh instance) must not + # fail a canary; 4xx (auth, generation mismatch) still fails fast. + admin_post() { + local out="${RUNNER_TEMP}/$1.json" + if ! curl --fail-with-body --max-time 30 \ + --retry 3 --retry-delay 2 --retry-connrefused --output "${out}" \ + --request POST "$2" \ + --header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \ + --header 'Content-Type: application/json' --data "$3"; then + cat "${out}" >&2 + return 1 + fi + cat "${out}" + } + CURRENT_RUNTIME="$(admin_post current-runtime \ + "${CELL_ORIGIN}/v1/admin/runtime-status" '{"v":1}')" # A rollback that failed between template apply and admission restore # leaves the cell already on the rollback image; resume from that # state instead of demanding the pre-rollback predecessor. @@ -370,11 +387,9 @@ jobs: if .regionalRehomeProtocol == null then "regionalRehomeProtocol" else empty end ] | if length > 0 then "runtime predecessor normalized legacy fields=" + join(",") else empty end' \ <<< "${CURRENT_RUNTIME}" - CURRENT_DIRECTOR_STATUS="$(curl --fail-with-body --max-time 30 \ - --request POST "${DIRECTOR_ORIGIN}/v1/admin/cell-status" \ - --header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \ - --header 'Content-Type: application/json' \ - --data "$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")" + CURRENT_DIRECTOR_STATUS="$(admin_post current-cell-status \ + "${DIRECTOR_ORIGIN}/v1/admin/cell-status" \ + "$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")" SOURCE_INCARNATION="$(jq -er '.status.runtime.cellIncarnation' \ <<< "${CURRENT_DIRECTOR_STATUS}")" if test "${ROLLBACK_RESUME}" = true && ! jq -e \ @@ -418,13 +433,13 @@ jobs: # result's generation is authoritative either way. ISOLATE_RESULT="$(node dev/scripts/prepare-relay-production-capacity-canary.mjs \ --director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \ - --cell-id "${TARGET_CELL_ID}" --mode isolate)" + --cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode isolate)" echo "${ISOLATE_RESULT}" ISOLATE_GENERATION="$(jq -er '.generation' <<< "${ISOLATE_RESULT}")" echo "SELECTOR_GENERATION_AFTER_ISOLATE=${ISOLATE_GENERATION}" >> "${GITHUB_ENV}" node dev/scripts/prepare-relay-production-capacity-canary.mjs \ --director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \ - --cell-id "${TARGET_CELL_ID}" --mode drain + --cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode drain node dev/scripts/verify-relay-capacity-transition.mjs \ --director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \ --cell-id "${TARGET_CELL_ID}" --hard-cap "${EXPECTED_HARD_CAP}" \ @@ -486,6 +501,7 @@ jobs: --rollback-image "${DESIRED_IMAGE}" \ --rehome-director-service-account "${DIRECTOR_RUNTIME_SERVICE_ACCOUNT}" \ --rehome-audience https://relay.onorca.dev/v1/admin/host-drain \ + --regional-rehome-protocol "${DESIRED_REHOME_PROTOCOL}" \ | jq -e '.changes == 2' >/dev/null fi gcloud compute instance-groups managed wait-until "${MIG_NAME}" --stable \ @@ -511,7 +527,8 @@ jobs: --unobserved-bound "${EXPECTED_UNOBSERVED_BOUND}" --image "${DESIRED_IMAGE}" \ --rollback-image "${IMAGE_REPOSITORY}@${CURRENT_IMAGE_DIGEST}" \ --rehome-director-service-account "${DIRECTOR_RUNTIME_SERVICE_ACCOUNT}" \ - --rehome-audience https://relay.onorca.dev/v1/admin/host-drain + --rehome-audience https://relay.onorca.dev/v1/admin/host-drain \ + --regional-rehome-protocol "${DESIRED_REHOME_PROTOCOL}" terraform -chdir=infra/terraform apply -auto-approve \ "${RUNNER_TEMP}/relay-same-cap.tfplan" gcloud compute instance-groups managed wait-until "${MIG_NAME}" --stable \ @@ -532,6 +549,20 @@ jobs: env: ORCA_RELAY_ADMIN_ID_TOKEN: ${{ steps.post-auth.outputs.id_token }} run: | + # A single transient 5xx (LB warm-up behind a fresh instance) must not + # fail a canary; 4xx (auth, generation mismatch) still fails fast. + admin_post() { + local out="${RUNNER_TEMP}/$1.json" + if ! curl --fail-with-body --max-time 30 \ + --retry 3 --retry-delay 2 --retry-connrefused --output "${out}" \ + --request POST "$2" \ + --header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \ + --header 'Content-Type: application/json' --data "$3"; then + cat "${out}" >&2 + return 1 + fi + cat "${out}" + } node dev/scripts/verify-relay-capacity-transition.mjs \ --director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \ --cell-id "${TARGET_CELL_ID}" --hard-cap "${EXPECTED_HARD_CAP}" \ @@ -539,19 +570,15 @@ jobs: --heartbeat fresh --admission migration-only --draining forbidden \ --activity allowed --expected-image-digests "${DESIRED_IMAGE_DIGEST}" \ --regional-rehome-protocol "${DESIRED_REHOME_PROTOCOL}" --timeout-ms 900000 - TARGET_RUNTIME="$(curl --fail-with-body --max-time 30 \ - --request POST "${CELL_ORIGIN}/v1/admin/runtime-status" \ - --header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \ - --header 'Content-Type: application/json' --data '{"v":1}')" + TARGET_RUNTIME="$(admin_post target-runtime \ + "${CELL_ORIGIN}/v1/admin/runtime-status" '{"v":1}')" jq -e --arg digest "${DESIRED_IMAGE_DIGEST}" \ --argjson protocol "${DESIRED_REHOME_PROTOCOL}" \ '.imageDigest == $digest and (.regionalRehomeProtocol // 0) == $protocol' \ <<< "${TARGET_RUNTIME}" >/dev/null - TARGET_DIRECTOR_STATUS="$(curl --fail-with-body --max-time 30 \ - --request POST "${DIRECTOR_ORIGIN}/v1/admin/cell-status" \ - --header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \ - --header 'Content-Type: application/json' \ - --data "$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")" + TARGET_DIRECTOR_STATUS="$(admin_post target-cell-status \ + "${DIRECTOR_ORIGIN}/v1/admin/cell-status" \ + "$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")" TARGET_INCARNATION="$(jq -er '.status.runtime.cellIncarnation' \ <<< "${TARGET_DIRECTOR_STATUS}")" if test "${ROLLBACK_RESUME}" = true; then @@ -588,7 +615,7 @@ jobs: echo "MUTATION_STARTED=true" >> "${GITHUB_ENV}" ACTIVATE_RESULT="$(node dev/scripts/prepare-relay-production-capacity-canary.mjs \ --director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \ - --cell-id "${TARGET_CELL_ID}" --mode activate)" + --cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode activate)" echo "${ACTIVATE_RESULT}" SELECTOR_GENERATION_AFTER_ACTIVATE="$(jq -er '.generation' \ <<< "${ACTIVATE_RESULT}")" @@ -627,7 +654,7 @@ jobs: test "${MUTATION_STARTED:-false}" = true || exit 0 ISOLATE_RESULT="$(node dev/scripts/prepare-relay-production-capacity-canary.mjs \ --director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \ - --cell-id "${TARGET_CELL_ID}" --mode isolate)" + --cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode isolate)" echo "${ISOLATE_RESULT}" # The isolate result carries the authoritative post-isolate generation; # fixed offsets are wrong whenever an earlier isolate was a no-op. diff --git a/.github/workflows/cloud-deploy-relay-production-same-cap.yml b/.github/workflows/cloud-deploy-relay-production-same-cap.yml index 1994966d083..fba5df0dcb9 100644 --- a/.github/workflows/cloud-deploy-relay-production-same-cap.yml +++ b/.github/workflows/cloud-deploy-relay-production-same-cap.yml @@ -87,13 +87,18 @@ jobs: gate: if: ${{ vars.ORCA_CLOUD_OPERATIONS_ENABLED == 'true' && (github.ref == 'refs/heads/main') }} runs-on: blacksmith-2vcpu-ubuntu-2204 - timeout-minutes: 10 + # Headroom for the full-history checkout the canary provenance check needs. + timeout-minutes: 15 environment: production outputs: cells: ${{ steps.wave.outputs.cells }} job-mode: ${{ steps.wave.outputs.job-mode }} steps: + # Full history: the canary authority a batch verifies is sealed at an ancestor commit, and + # the provenance check fails closed on a commit a shallow clone left out. - uses: actions/checkout@v4 + with: + fetch-depth: 0 - uses: actions/setup-node@v4 with: { node-version: 24 } diff --git a/.github/workflows/cloud-operate-relay-production-rehome-job.yml b/.github/workflows/cloud-operate-relay-production-rehome-job.yml index 682953af5e7..fdb1aca45e0 100644 --- a/.github/workflows/cloud-operate-relay-production-rehome-job.yml +++ b/.github/workflows/cloud-operate-relay-production-rehome-job.yml @@ -95,7 +95,11 @@ jobs: ;; esac + # Full history: the monitor evidence this job verifies is sealed at an ancestor commit, + # and the provenance check fails closed on a commit a shallow clone left out. - uses: actions/checkout@v4 + with: + fetch-depth: 0 - uses: actions/setup-node@v4 with: diff --git a/.github/workflows/cloud-verify.yml b/.github/workflows/cloud-verify.yml index f0cc2df2bad..e2ba9407ac4 100644 --- a/.github/workflows/cloud-verify.yml +++ b/.github/workflows/cloud-verify.yml @@ -25,9 +25,10 @@ defaults: working-directory: cloud jobs: + # Public-repository hosted runners preserve Blacksmith allowance for macOS. security: name: Secret scan - runs-on: blacksmith-2vcpu-ubuntu-2204 + runs-on: ubuntu-22.04 steps: - uses: actions/checkout@v4 with: @@ -53,7 +54,7 @@ jobs: # Compiles the workspace. No Postgres service: nothing here reaches a # database, and the service container costs ~13s of startup. build: - runs-on: blacksmith-4vcpu-ubuntu-2204 + runs-on: ubuntu-22.04 steps: - uses: actions/checkout@v4 @@ -73,7 +74,7 @@ jobs: # package it needs through the relay pretest hook, so it does not depend on # `pnpm build` having run. test: - runs-on: blacksmith-4vcpu-ubuntu-2204 + runs-on: ubuntu-22.04 services: postgres: image: postgres:16-alpine @@ -107,7 +108,7 @@ jobs: # Fork pull requests reach this job, so it never configures a backend, never plans, and never # holds a credential. Only the relay root ships here; foundation and apps stay private. terraform: - runs-on: blacksmith-2vcpu-ubuntu-2204 + runs-on: ubuntu-22.04 steps: - uses: actions/checkout@v4 diff --git a/.github/workflows/dev-channel-win-build.yml b/.github/workflows/dev-channel-win-build.yml index e16a50f1c3c..89fda2ebef9 100644 --- a/.github/workflows/dev-channel-win-build.yml +++ b/.github/workflows/dev-channel-win-build.yml @@ -149,9 +149,12 @@ jobs: fi # Reachability is the trust test: GitHub serves PR-only commits by SHA, # so resolving the object is not proof a branch or tag of this repo - # reaches it. Bare + tree:0 keeps this to the commit graph. + # reaches it. Bare + tree:0 keeps this to the commit graph; reftable + # because branches that differ only in casing cannot both be stored by + # the files backend on a case-insensitive runner disk, which fails the + # entire fetch rather than the one ref. scratch="$RUNNER_TEMP/vet-requested-ref" - git init -q --bare "$scratch" + git init -q --bare --ref-format=reftable "$scratch" git -C "$scratch" fetch -q --filter=tree:0 "$REPO_URL" '+refs/heads/*:refs/heads/*' '+refs/tags/*:refs/tags/*' if ! git -C "$scratch" rev-parse --verify --quiet "$REQUESTED_SHA^{commit}" >/dev/null; then echo "::error::Commit $REQUESTED_SHA is not in stablyai/orca." diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml index 3560d302a79..a94a7ea2ba5 100644 --- a/.github/workflows/e2e.yml +++ b/.github/workflows/e2e.yml @@ -27,6 +27,10 @@ on: description: Ref to check out (defaults to the workflow ref) required: false type: string + test_files: + description: JSON array of specs to run; empty runs the full suite + required: false + type: string schedule: # Why: GitHub cron uses UTC; these slots map to 10am and 3pm # America/Phoenix for the default-branch E2E run. @@ -146,7 +150,7 @@ jobs: # Native cache misses need the compiler, Electron needs Xvfb, and paired # Quick Open needs ripgrep. Install them in one apt transaction per shard. - name: Install native build and headless UI tools - run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk python3 ripgrep xvfb zsh + run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk python3 ripgrep xvfb zsh openbox x11-utils - uses: ./.github/actions/install-node-dependencies with: @@ -167,7 +171,7 @@ jobs: # ORCA_E2E_FORWARD_APP_LOGS keeps startup failures visible when Electron # launches but never creates a BrowserWindow. - name: Run E2E tests (${{ matrix.shard_name }}) - run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 ORCA_E2E_WEB_CLIENT=1 ORCA_RELAY_PATH="$GITHUB_WORKSPACE/out/relay" pnpm run test:e2e --shard=${{ matrix.shard }} + run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 ORCA_E2E_WEB_CLIENT=1 ORCA_RELAY_PATH="$GITHUB_WORKSPACE/out/relay" pnpm run test:e2e --shard=${{ matrix.shard }} # Why: Playwright retains traces/screenshots only on failure. Uploading # them as an artifact makes post-mortem debugging on CI possible without @@ -201,7 +205,7 @@ jobs: # unbounded inventory fallback; the paired fixture exercises that real boundary. # Why openssh-client: the Docker-SSH fixture shells out to ssh/ssh-keygen, and this # lane now receives those specs from pr.yml's SSH source mapping. - run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 ripgrep xvfb zsh + run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 ripgrep xvfb zsh openbox x11-utils - uses: ./.github/actions/install-node-dependencies with: @@ -241,7 +245,7 @@ jobs: if grep -l '@headful' "${TEST_FILES[@]}" >/dev/null; then E2E_PROJECT_ARGS+=(--project=electron-headful) fi - xvfb-run --auto-servernum env "${E2E_ENV[@]}" \ + xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env "${E2E_ENV[@]}" \ pnpm run test:e2e "${TEST_FILES[@]}" --workers=1 "${E2E_PROJECT_ARGS[@]}" - name: Upload Playwright traces @@ -278,7 +282,7 @@ jobs: ref: ${{ inputs.ref || github.ref }} - name: Install native build and headless UI tools - run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 xvfb zsh + run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 ripgrep xvfb zsh openbox x11-utils - uses: ./.github/actions/install-node-dependencies with: @@ -293,7 +297,7 @@ jobs: # Why: this is the release-path proof that the deployed Linux relay keeps # its PTY and explorer live across a real watcher SIGSEGV. - name: Run Docker SSH watcher isolation E2E - run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-watcher-isolation + run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-watcher-isolation # Why: Playwright empties test-results/ when it starts, so each step here used to # destroy the previous step's traces. Only the last lane's failure was ever @@ -310,7 +314,7 @@ jobs: # readiness across live SSH, headed paired, and headless serve topologies. - name: Run Docker SSH terminal parking + startup readiness E2E if: always() - run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-terminal-parking + run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-terminal-parking - name: Keep terminal-parking traces if: always() @@ -326,7 +330,7 @@ jobs: # legible as an SSH-named failure. - name: Run remaining Docker SSH E2E if: always() - run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker + run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker - name: Keep remaining-ssh-docker traces if: always() diff --git a/.github/workflows/golden-e2e-experiment.yml b/.github/workflows/golden-e2e-experiment.yml index d46c80033fa..11cfa866c67 100644 --- a/.github/workflows/golden-e2e-experiment.yml +++ b/.github/workflows/golden-e2e-experiment.yml @@ -98,12 +98,17 @@ jobs: $env:SKIP_BUILD = '1' $env:ORCA_E2E_FORWARD_APP_LOGS = '1' pnpm run --if-present test:e2e:workspace-session-golden + if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } pnpm run --if-present test:e2e:windows-fresh-startup-golden + if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } pnpm run --if-present test:e2e:tab-bar-agent-launch-golden + if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } if (Test-Path tests/e2e/golden-fresh-profile-terminal.spec.ts) { pnpm run test:e2e -- tests/e2e/golden-fresh-profile-terminal.spec.ts tests/e2e/golden-shell-command.spec.ts + if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } } pnpm run --if-present test:e2e:source-control-golden + if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } - name: Upload Playwright traces if: failure() diff --git a/.github/workflows/hourly-mac-build.yml b/.github/workflows/hourly-mac-build.yml index ac3af92a3bc..c300b2543b8 100644 --- a/.github/workflows/hourly-mac-build.yml +++ b/.github/workflows/hourly-mac-build.yml @@ -26,7 +26,7 @@ name: Hourly macOS Dev Build # HOURLY_RELEASE_APP_ID the App's numeric id # HOURLY_RELEASE_APP_PRIVATE_KEY the App's .pem private key # -# Installation tokens live one hour, which is why this mints twice. Install and +# Installation tokens live one hour, so the build job mints twice. Install and # build need no token at all, and notarization can hold the publish step for tens # of minutes; minting again once the build is done starts the clock at the first # call that actually uses it rather than burning a third of it on `pnpm install`. @@ -60,33 +60,15 @@ env: HOURLY_RETAIN_COUNT: 72 jobs: - build-hourly-mac: + # Avoid occupying the limited Mac pool when main has not moved. + preflight: if: github.repository == 'stablyai/orca' + runs-on: ubuntu-latest + timeout-minutes: 5 outputs: - tag: ${{ steps.release.outputs.tag }} - version: ${{ steps.hourly.outputs.version }} + should_build: ${{ steps.freshness.outputs.should_build }} head_sha: ${{ steps.freshness.outputs.head_sha }} - published: ${{ steps.publish_live.outcome == 'success' && 'true' || 'false' }} - runs-on: blacksmith-6vcpu-macos-15 - # Why 150: it must exceed the worst case the retry budgets below can produce - # (install 3x10 + publish 2x45 = 120, plus ~25 for checkout/build/verify), or - # the job is killed mid-retry and no cleanup step runs at all. A typical run - # is far shorter — this is the notary queue's tail, not its median. - timeout-minutes: 150 - env: - NODE_OPTIONS: --max-old-space-size=4096 steps: - - name: Checkout - uses: actions/checkout@v6 - with: - ref: main - fetch-depth: 0 - # Why: this job only reads stablyai/orca and never pushes; every write - # goes to the hourly repo through a minted App token passed by env. - # Not persisting the checkout credential shrinks the blast radius if a - # build step is compromised (zizmor: artipacked). - persist-credentials: false - - name: Mint hourly repo token id: app_token uses: actions/create-github-app-token@v2 @@ -95,18 +77,19 @@ jobs: private-key: ${{ secrets.HOURLY_RELEASE_APP_PRIVATE_KEY }} owner: stablyai repositories: orca-hourly + permission-contents: read - # Why: main is often idle overnight. Rebuilding an unchanged commit burns a - # runner hour and adds a redundant tag to the retention window. - name: Check whether main moved since the last hourly id: freshness shell: bash env: GH_TOKEN: ${{ steps.app_token.outputs.token }} + MAIN_REPO_TOKEN: ${{ github.token }} FORCED: ${{ github.event_name == 'workflow_dispatch' && inputs.force }} run: | set -euo pipefail - head_sha="$(git rev-parse HEAD)" + head_sha="$(GH_TOKEN="$MAIN_REPO_TOKEN" gh api "repos/$GITHUB_REPOSITORY/commits/main" --jq .sha)" + [[ "$head_sha" =~ ^[0-9a-f]{40}$ ]] || { echo "::error::Could not resolve main"; exit 1; } echo "head_sha=$head_sha" >>"$GITHUB_OUTPUT" if [[ "$FORCED" == "true" ]]; then echo "should_build=true" >>"$GITHUB_OUTPUT" @@ -133,21 +116,55 @@ jobs: echo "main moved to $head_sha (last hourly built $last_sha); building." fi + build-hourly-mac: + needs: preflight + if: needs.preflight.outputs.should_build == 'true' + outputs: + tag: ${{ steps.release.outputs.tag }} + version: ${{ steps.hourly.outputs.version }} + head_sha: ${{ needs.preflight.outputs.head_sha }} + published: ${{ steps.publish_live.outcome == 'success' && 'true' || 'false' }} + runs-on: blacksmith-6vcpu-macos-15 + # Why 150: it must exceed the worst case the retry budgets below can produce + # (install 3x10 + publish 2x45 = 120, plus ~25 for checkout/build/verify), or + # the job is killed mid-retry and no cleanup step runs at all. A typical run + # is far shorter — this is the notary queue's tail, not its median. + timeout-minutes: 150 + env: + NODE_OPTIONS: --max-old-space-size=4096 + steps: + - name: Checkout + uses: actions/checkout@v6 + with: + ref: ${{ needs.preflight.outputs.head_sha }} + fetch-depth: 0 + # Why: this job only reads stablyai/orca and never pushes; every write + # goes to the hourly repo through a minted App token passed by env. + # Not persisting the checkout credential shrinks the blast radius if a + # build step is compromised (zizmor: artipacked). + persist-credentials: false + + - name: Mint hourly repo token + id: app_token + uses: actions/create-github-app-token@v2 + with: + app-id: ${{ secrets.HOURLY_RELEASE_APP_ID }} + private-key: ${{ secrets.HOURLY_RELEASE_APP_PRIVATE_KEY }} + owner: stablyai + repositories: orca-hourly + - name: Setup pnpm - if: steps.freshness.outputs.should_build == 'true' uses: pnpm/setup@v2 with: install: false - name: Setup Node.js - if: steps.freshness.outputs.should_build == 'true' uses: actions/setup-node@v6 with: node-version-file: package.json cache: pnpm - name: Cache electron-builder downloads - if: steps.freshness.outputs.should_build == 'true' uses: actions/cache@v5 with: path: | @@ -158,7 +175,6 @@ jobs: electron-builder-mac- - name: Install dependencies - if: steps.freshness.outputs.should_build == 'true' uses: nick-fields/retry@v4 with: timeout_minutes: 10 @@ -169,7 +185,6 @@ jobs: # Why: signing is what makes an hourly installable over an existing Orca, so # a missing cert must fail here rather than after a 20-minute build. - name: Verify macOS signing environment - if: steps.freshness.outputs.should_build == 'true' run: node config/scripts/verify-macos-release-env.mjs env: CSC_LINK: ${{ secrets.MAC_CERTS }} @@ -180,7 +195,6 @@ jobs: - name: Compute hourly version id: hourly - if: steps.freshness.outputs.should_build == 'true' shell: bash env: GH_TOKEN: ${{ steps.app_token.outputs.token }} @@ -211,7 +225,7 @@ jobs: node config/scripts/hourly-build-version.mjs \ >"$RUNNER_TEMP/hourly-identity.txt" grep -E '^(version|build_number)=' "$RUNNER_TEMP/hourly-identity.txt" - # Why check rather than trust: the checkout above pins `ref: main`, but a + # Why check rather than trust: the checkout above pins the resolved main commit, but a # workflow_dispatch runs this file from whatever branch was dispatched. A # branch that edits this step while main still has the old script yields # an empty name and an untitled release — silent, and only visible once @@ -223,7 +237,6 @@ jobs: cat "$RUNNER_TEMP/hourly-identity.txt" >>"$GITHUB_OUTPUT" - name: Build app - if: steps.freshness.outputs.should_build == 'true' run: pnpm build:release env: NODE_OPTIONS: --max-old-space-size=4096 @@ -239,7 +252,6 @@ jobs: # part the full budget. - name: Re-mint hourly repo token for publish id: app_token_publish - if: steps.freshness.outputs.should_build == 'true' uses: actions/create-github-app-token@v2 with: app-id: ${{ secrets.HOURLY_RELEASE_APP_ID }} @@ -249,13 +261,12 @@ jobs: - name: Create hourly release id: release - if: steps.freshness.outputs.should_build == 'true' shell: bash env: GH_TOKEN: ${{ steps.app_token_publish.outputs.token }} TAG: v${{ steps.hourly.outputs.version }} NAME: ${{ steps.hourly.outputs.name }} - SHA: ${{ steps.freshness.outputs.head_sha }} + SHA: ${{ needs.preflight.outputs.head_sha }} run: | set -euo pipefail # Kept at 12 even though the title shows 7: the freshness check above @@ -291,7 +302,6 @@ jobs: echo "tag=$TAG" >>"$GITHUB_OUTPUT" - name: Publish hourly macOS artifacts - if: steps.freshness.outputs.should_build == 'true' uses: nick-fields/retry@v4 with: # Why 45 like the release pipeline: an attempt is pack + notarize + @@ -322,7 +332,6 @@ jobs: # release missing that manifest is a tag the picker offers and the download # 404s on, so fail loudly instead of leaving a broken entry. - name: Verify update manifest published - if: steps.freshness.outputs.should_build == 'true' shell: bash env: GH_TOKEN: ${{ steps.app_token_publish.outputs.token }} @@ -352,7 +361,6 @@ jobs: # means the picker can never offer a release whose assets are incomplete. - name: Publish the verified release id: publish_live - if: steps.freshness.outputs.should_build == 'true' shell: bash env: GH_TOKEN: ${{ steps.app_token_publish.outputs.token }} diff --git a/.github/workflows/mobile.yml b/.github/workflows/mobile.yml index 59f6cf20bf4..6dbfc02aa3c 100644 --- a/.github/workflows/mobile.yml +++ b/.github/workflows/mobile.yml @@ -15,8 +15,13 @@ on: # Why: this job holds the only checks that load the Fastfile, so edits to # it or to the release workflow it guards must re-run them. - '.github/workflows/mobile.yml' + - '.github/actions/install-node-dependencies/**' - '.github/workflows/mobile-ios-release.yml' +concurrency: + group: mobile-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + jobs: verify: runs-on: ubuntu-latest @@ -35,10 +40,7 @@ jobs: - name: Checkout uses: actions/checkout@v6 - - name: Setup Node.js - uses: actions/setup-node@v6 - with: - node-version-file: package.json + - uses: ./.github/actions/install-node-dependencies # bundler-cache installs mobile/Gemfile.lock, so this job is also what # proves the pinned fastlane the release workflow depends on still @@ -50,23 +52,6 @@ jobs: bundler-cache: true working-directory: mobile - - name: Setup pnpm - uses: pnpm/setup@v2 - with: - install: false - - # Why: the mobile typecheck imports shared types from ../src/shared, and - # some of those files import runtime deps (tweetnacl, ws) resolved from - # the repo-root node_modules. Without a root install, tsc fails with - # "Cannot find module 'tweetnacl'/'ws'". Mobile is a separate pnpm project - # (not in the root workspace), so this is a distinct install. - # --ignore-scripts skips the root postinstall (Electron native-module - # rebuild) which is irrelevant to a type-only check and would only add - # time and failure surface on this ubuntu mobile runner. - - name: Install root dependencies - working-directory: . - run: pnpm install --frozen-lockfile --ignore-scripts - - name: Install dependencies run: pnpm install --frozen-lockfile diff --git a/.github/workflows/performance-contracts.yml b/.github/workflows/performance-contracts.yml new file mode 100644 index 00000000000..d45d8b8f45a --- /dev/null +++ b/.github/workflows/performance-contracts.yml @@ -0,0 +1,63 @@ +name: Performance contracts + +on: + schedule: + - cron: '15 9 * * *' + workflow_dispatch: + pull_request: + paths: + - '.github/workflows/performance-contracts.yml' + - 'config/vitest.performance.config.ts' + - 'config/oxlint-performance-audit.json' + - 'config/oxlint-plugins/*performance.mjs' + - 'config/oxlint-plugins/quadratic-buffer-concat.mjs' + - 'config/scripts/*-plugin.test.mjs' + # Keep in sync with the contract list in config/vitest.performance.config.ts; + # without these a rename lands green and only breaks the next nightly. + - 'src/main/sqlite/sync-database.test.ts' + - 'src/main/runtime/orchestration/db/row-column-lists.test.ts' + - 'src/relay/fs-path-metadata-symlink-concurrency.test.ts' + - 'src/renderer/src/components/editor/rich-markdown-list-tokenizers.test.ts' + - 'src/renderer/src/components/editor/rich-markdown-lowlight-cache.test.ts' + - 'src/renderer/src/components/terminal-pane/agent-completion-coordinator-queued-inspection-disposal.test.ts' + - 'src/renderer/src/lib/pane-manager/pane-terminal-output-scheduler-queue-retention.test.ts' + +permissions: + contents: read + +concurrency: + group: performance-contracts-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + +jobs: + contracts: + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, macos-latest, windows-latest] + runs-on: ${{ matrix.os }} + timeout-minutes: 20 + steps: + - uses: actions/checkout@v6 + with: + persist-credentials: false + - uses: ./.github/actions/install-node-dependencies + - name: Run operation-count and retention contracts + run: pnpm test:perf:contracts --reporter=default --reporter=json --outputFile=performance-contracts.json + # Source-only scan: identical on every OS, so run it once. + - name: Audit production performance patterns + if: always() && matrix.os == 'ubuntu-latest' + shell: bash + run: pnpm --silent audit:perf > performance-audit.json + - uses: actions/upload-artifact@v7 + if: always() + with: + name: performance-contracts-${{ matrix.os }} + path: performance-contracts.json + if-no-files-found: error + - uses: actions/upload-artifact@v7 + if: always() && matrix.os == 'ubuntu-latest' + with: + name: performance-audit + path: performance-audit.json + if-no-files-found: error diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index ca1651325c9..bd655eb801a 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -41,6 +41,10 @@ jobs: managed_hook_node18: ${{ steps.filter.outputs.managed_hook_node18 }} package: ${{ steps.filter.outputs.package }} package_windows: ${{ steps.filter.outputs.package_windows }} + e2e_should_run: ${{ steps.e2e_filter.outputs.should_run }} + test_files: ${{ steps.e2e_filter.outputs.test_files }} + ssh_source_changed: ${{ steps.e2e_filter.outputs.ssh_source_changed }} + native_ime_source_changed: ${{ steps.e2e_filter.outputs.native_ime_source_changed }} steps: - name: Checkout uses: actions/checkout@v6 @@ -66,6 +70,38 @@ jobs: printf '%s\n' "$CHANGED" printf '%s\n' "$CHANGED" | node config/scripts/pr-code-change-scope.mjs | tee -a "$GITHUB_OUTPUT" + # Reuse the path-detector checkout instead of queuing another runner. + - name: Filter changed E2E specs + id: e2e_filter + if: github.event.pull_request.draft != true && steps.filter.outputs.should_run == 'true' + run: | + set -euo pipefail + BASE="${{ github.event.pull_request.base.sha }}" + HEAD="${{ github.event.pull_request.head.sha }}" + CHANGED="$(git diff --name-only --diff-filter=AMCR --merge-base "$BASE" "$HEAD")" + # Source routes are executable contracts so a test can prove exact + # authorities, exclusions, and sentinels without evaluating workflow shell. + TEST_FILES_JSON="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs)" + echo "test_files=$TEST_FILES_JSON" >> "$GITHUB_OUTPUT" + # Why a separate signal: the Docker-SSH lane must trigger on SSH source, not on a + # spec name surviving in a route's list. Same routes, so the two cannot drift. + SSH_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --ssh-source)" + echo "ssh_source_changed=$SSH_SOURCE_CHANGED" >> "$GITHUB_OUTPUT" + echo "SSH source changed: $SSH_SOURCE_CHANGED" + # Why its own signal: the real-IME lane is a whole ibus session, not a spec, so it must + # trigger on IME source rather than on a spec name in some route's list. + NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)" + echo "native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED" >> "$GITHUB_OUTPUT" + echo "Native IME source changed: $NATIVE_IME_SOURCE_CHANGED" + SHOULD_RUN="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --reusable-workflow)" + if [ "$SHOULD_RUN" = true ]; then + echo "should_run=true" >> "$GITHUB_OUTPUT" + echo "Changed E2E specs: $TEST_FILES_JSON" + else + echo "should_run=false" >> "$GITHUB_OUTPUT" + echo "No specs requiring the reusable E2E workflow" + fi + static_analysis: name: static analysis needs: [code_paths] @@ -712,7 +748,11 @@ jobs: - name: Package unpacked app env: ORCA_REUSE_PREPARED_NATIVE_RUNTIME: '1' - run: pnpm exec electron-builder --config config/electron-builder.config.cjs --linux AppImage deb rpm --x64 --publish never + # PR artifacts are only inspected locally; gzip avoids release-size xz compression. + run: >- + pnpm exec electron-builder --config config/electron-builder.config.cjs + --linux AppImage deb rpm --x64 --publish never + --config.deb.compression=gz --config.rpm.compression=gzip - name: Verify root-package marker payloads run: | @@ -792,10 +832,13 @@ jobs: node_modules/.pnpm/@vscode+windows-process-tree@*/node_modules/@vscode/windows-process-tree/build key: native-modules-${{ runner.os }}-${{ steps.deps.outputs.native-cache-scope }}-${{ runner.arch }}-node-node${{ steps.deps.outputs.node-version }}-${{ hashFiles('pnpm-lock.yaml', '.github/actions/install-node-dependencies/action.yml', 'config/scripts/ensure-native-runtime.mjs', 'config/scripts/rebuild-native-deps.mjs', 'config/patches/node-pty@1.1.0.patch', 'config/patches/@vscode__windows-process-tree@0.8.0.patch') }} + # vitest runs here directly rather than through `pnpm test`, so the addon + # assertions only hold once install-node-dependencies has rebuilt natives. - name: Test Windows-specific boundaries run: >- pnpm exec vitest run --config config/vitest.config.ts config/scripts/rebuild-native-deps.test.mjs + config/scripts/rebuild-native-deps-windows-process-tree.test.mjs src/main/browser/browser-client-page-renderer-lifecycle.electron.test.ts src/main/browser/browser-route-tcp-egress.electron.test.ts src/main/browser/browser-route-webrtc-egress.electron.test.ts @@ -804,9 +847,14 @@ jobs: src/main/providers/windows-conpty-wide-char-duplication.node-pty.test.ts src/main/providers/pty-repaint-wide-char-buffer.node-pty.test.ts src/shared/child-process/windows-command-line.win32.test.ts + src/shared/child-process/windows-cmd-shim-resolution.test.ts + src/shared/child-process/windows-cmd-shim-resolution.win32.test.ts src/main/agent-hooks/windows-hook-payload-delivery.test.ts + src/main/agent-hooks/windows-direct-cmd-hook-command.test.ts src/main/windows/windows-pty-job.win32.test.ts src/main/windows/windows-host-job.win32.test.ts + src/main/windows/windows-process-tree-command-line-patch.test.ts + src/main/windows-live-tree-kill.win32.test.ts src/main/wsl/wsl-runner.test.ts src/main/wsl/wsl-guest-environment.test.ts src/main/wsl/wsl-invocation-boundary.test.ts @@ -814,14 +862,18 @@ jobs: src/main/wsl/wsl-w1-w3-contract.test.ts src/shared/source-scan/source-tree-scan.test.ts src/main/cli/wsl-cli-powershell-boundary.test.ts + src/main/computer/desktop-script-runtime-host.win32.test.ts src/main/cursor/hook-service.test.ts src/main/orca-profiles/profile-index-store.test.ts src/main/startup/windows-install-dir-acl-repair.win32.test.ts src/main/runtime/repo-worktree-admin-fingerprint.test.ts src/main/runtime/worktree-scan-admin-fingerprint-gate.test.ts src/shared/secure-file-fsync-flags.test.ts + src/shared/secure-path-windows-acl.win32.test.ts + src/main/runtime/unreadable-secret-store-preservation.win32.test.ts src/main/ipc/pty-codex-account-attribution.test.ts src/main/ipc/pty-spawn-env-codex-resume-provenance.test.ts + src/relay/windows-port-scan.win32.test.ts # Why the :parallel variant: identical to build:release except the three # electron-vite targets overlap instead of running back to back. The Linux package @@ -860,65 +912,10 @@ jobs: - name: Smoke packaged CLI run: node config/scripts/smoke-packaged-cli.mjs --app-dir=dist/win-unpacked - # Why: PR E2E is advisory and only validates changed specs; scheduled and - # release runs retain full-suite coverage. - e2e-paths: - name: detect changed e2e specs - needs: [code_paths] - runs-on: ubuntu-latest - if: github.event.pull_request.draft != true && needs.code_paths.outputs.should_run == 'true' - # Why: detector only needs to read the checkout; do not inherit repo defaults. - permissions: - contents: read - outputs: - should_run: ${{ steps.filter.outputs.should_run }} - test_files: ${{ steps.filter.outputs.test_files }} - ssh_source_changed: ${{ steps.filter.outputs.ssh_source_changed }} - native_ime_source_changed: ${{ steps.filter.outputs.native_ime_source_changed }} - steps: - - name: Checkout - uses: actions/checkout@v6 - with: - # Why blob:none: full history is needed for the merge-base diff, but historical - # file contents are not. Blobs are ~89% of this repo's pack, and Git fetches the - # few this job actually reads on demand. - fetch-depth: 0 - filter: blob:none - persist-credentials: false - - - name: Filter changed E2E specs - id: filter - run: | - set -euo pipefail - BASE="${{ github.event.pull_request.base.sha }}" - HEAD="${{ github.event.pull_request.head.sha }}" - CHANGED="$(git diff --name-only --diff-filter=AMCR --merge-base "$BASE" "$HEAD")" - # Source routes are executable contracts so a test can prove exact - # authorities, exclusions, and sentinels without evaluating workflow shell. - TEST_FILES_JSON="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs)" - echo "test_files=$TEST_FILES_JSON" >> "$GITHUB_OUTPUT" - # Why a separate signal: the Docker-SSH lane must trigger on SSH source, not on a - # spec name surviving in a route's list. Same routes, so the two cannot drift. - SSH_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --ssh-source)" - echo "ssh_source_changed=$SSH_SOURCE_CHANGED" >> "$GITHUB_OUTPUT" - echo "SSH source changed: $SSH_SOURCE_CHANGED" - # Why its own signal: the real-IME lane is a whole ibus session, not a spec, so it must - # trigger on IME source rather than on a spec name in some route's list. - NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)" - echo "native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED" >> "$GITHUB_OUTPUT" - echo "Native IME source changed: $NATIVE_IME_SOURCE_CHANGED" - if [ "$TEST_FILES_JSON" != '[]' ]; then - echo "should_run=true" >> "$GITHUB_OUTPUT" - echo "Changed E2E specs: $TEST_FILES_JSON" - else - echo "should_run=false" >> "$GITHUB_OUTPUT" - echo "No changed E2E specs" - fi - e2e: name: e2e - needs: e2e-paths - if: needs.e2e-paths.outputs.should_run == 'true' + needs: code_paths + if: needs.code_paths.outputs.e2e_should_run == 'true' # Why: reusable e2e.yml only checkouts, builds, and uploads artifacts. permissions: contents: read @@ -927,8 +924,8 @@ jobs: # The synthetic pull-request merge ref can disappear while this reusable # workflow is queued. The head SHA is immutable and works for every PR. ref: ${{ github.event.pull_request.head.sha }} - test_files: ${{ needs.e2e-paths.outputs.test_files }} - ssh_source_changed: ${{ needs.e2e-paths.outputs.ssh_source_changed }} + test_files: ${{ needs.code_paths.outputs.test_files }} + ssh_source_changed: ${{ needs.code_paths.outputs.ssh_source_changed }} # Why this is not in verify's needs: it is the first PR-gate run of a harness whose reliability # is only known from nightly main runs (20/20 green, 2026-08-09..2026-08-29, p50 3m25s). It @@ -938,8 +935,8 @@ jobs: # require `success || skipped` outside the strict loop — see the note on `e2e`. terminal_ime_native: name: real IME - needs: e2e-paths - if: needs.e2e-paths.outputs.native_ime_source_changed == 'true' + needs: code_paths + if: needs.code_paths.outputs.native_ime_source_changed == 'true' # Why: the reusable workflow only checks out, builds, and uploads artifacts. permissions: contents: read diff --git a/.github/workflows/release-cut.yml b/.github/workflows/release-cut.yml index 6f888c3a512..c2124d12990 100644 --- a/.github/workflows/release-cut.yml +++ b/.github/workflows/release-cut.yml @@ -809,13 +809,7 @@ jobs: env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} TAG: ${{ needs.cut.outputs.tag }} - run: | - if gh release view "$TAG" --repo "$GITHUB_REPOSITORY" >/dev/null 2>&1; then - echo "Release $TAG already exists." - exit 0 - fi - - node config/scripts/create-draft-release.mjs "$TAG" + run: node config/scripts/create-draft-release.mjs "$TAG" terminal-rendering-golden: needs: cut @@ -858,16 +852,17 @@ jobs: if: runner.os == 'Linux' run: sudo apt-get update && sudo apt-get install -y build-essential python3 xvfb - - name: Setup Node.js - uses: actions/setup-node@v6 - with: - node-version-file: package.json - - name: Setup pnpm uses: pnpm/setup@v2 with: install: false + - name: Setup Node.js + uses: actions/setup-node@v6 + with: + node-version-file: package.json + cache: pnpm + # Why: Linux terminal golden E2E uses the same native install path as # release CI, which needs pnpm to bypass its non-executable gyp_main.py. - name: Use external node-gyp to avoid pnpm's bundled copy (Linux only) @@ -1074,16 +1069,17 @@ jobs: if: runner.os == 'Linux' run: sudo apt-get update && sudo apt-get install -y build-essential python3 xvfb - - name: Setup Node.js - uses: actions/setup-node@v6 - with: - node-version-file: package.json - - name: Setup pnpm uses: pnpm/setup@v2 with: install: false + - name: Setup Node.js + uses: actions/setup-node@v6 + with: + node-version-file: package.json + cache: pnpm + # Why: keep the non-blocking evidence lane on the same Linux native # install path as the blocking golden and release build jobs. - name: Use external node-gyp to avoid pnpm's bundled copy (Linux only) @@ -1425,6 +1421,17 @@ jobs: command: ${{ matrix.release_command }} env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + # Why: the NSIS uninstaller only exists inside electron-builder's + # uninstaller pass, which deletes it right after embedding it. The sign + # hook in config/scripts/windows-uninstaller-signing.cjs copies it out + # here so it can ride the inner-binaries SignPath request below. + # Why runner.temp and never the workspace: `files` in + # config/electron-builder.config.cjs is all-negation, so app-builder + # prepends `**/*` and packs whatever is left in the checkout root. This + # step retries up to 3 times; attempt 1 writes the file after packing, + # but attempts 2 and 3 would then pack the unsigned uninstaller into + # app.asar - the exact defect this chain exists to remove. + ORCA_WIN_UNINSTALLER_EXPORT_PATH: ${{ runner.temp }}\uninstaller-signing\unsigned\orca-uninstaller.exe - name: Verify Windows node-pty ConPTY runtime if: matrix.platform == 'win' && github.run_attempt == 1 @@ -1451,7 +1458,10 @@ jobs: # Why: SignPath cannot deep-sign inside NSIS installers, so inner PE # files (Orca.exe, node-pty *.node, DLLs) are signed via a separate zip # request, then the installer is rebuilt from the signed tree before the - # existing installer signing request below. Every step in this chain is + # existing installer signing request below. The NSIS uninstaller rides + # this same request (it is the MDE update cluster: old-uninstaller.exe / + # Uninstall Orca.exe), captured through electron-builder's sign hook and + # swapped back in during the rebuild — no third approval wait. Every step is # fail-open (continue-on-error + outcome gating): any failure ships the # original installer with unsigned inner binaries, exactly like releases # did before this chain existed. Rehearsed end to end in run 28988432001 @@ -1498,6 +1508,36 @@ jobs: Write-Host "Skipped $($skipped.Count) already-signed files:" $skipped | ForEach-Object { Write-Host " $_" } + # Why the uninstaller rides this request: it is the file MDE flagged in + # the whole update cluster (old-uninstaller.exe / Uninstall Orca.exe), + # and folding it in here costs no extra approval wait. Why it is kept + # out of inner-signing-list.txt: that list drives the copy-back into + # dist/win-unpacked, and the uninstaller does not live there — it is + # re-injected through the sign hook during the rebuild instead. + # Why this name and not "Uninstall Orca.exe": the restore loop below + # matches staged files by suffix (`-like "*$relative"`) and takes the + # first hit, so any staged path ending in "Orca.exe" is separated from + # the real Orca.exe only by Get-ChildItem's enumeration order. That + # order happens to favour the root file today, but it is not a + # documented guarantee; a name that cannot suffix-match is. + # Why the whole block is caught rather than just Test-Path'd: this + # step's outcome gates the upload of every inner binary, so a locked + # file or a full disk here would cost all of them their signatures - + # worse than shipping no uninstaller signature at all. + try { + $exportedUninstaller = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\unsigned\orca-uninstaller.exe' + if (Test-Path -LiteralPath $exportedUninstaller) { + $uninstallerStagePath = Join-Path $stage.FullName 'uninstaller\orca-uninstaller.exe' + New-Item -ItemType Directory -Force -Path (Split-Path $uninstallerStagePath) -ErrorAction Stop | Out-Null + Copy-Item -LiteralPath $exportedUninstaller -Destination $uninstallerStagePath -Force -ErrorAction Stop + Write-Host 'Staged the NSIS uninstaller for signing: uninstaller\orca-uninstaller.exe' + } else { + Write-Host "::warning::No exported NSIS uninstaller at $exportedUninstaller; this release ships an unsigned uninstaller (fail-open)." + } + } catch { + Write-Host "::warning::Could not stage the NSIS uninstaller ($_); this release ships an unsigned uninstaller (fail-open)." + } + - name: Upload unsigned inner binaries for SignPath id: upload-unsigned-inner if: matrix.platform == 'win' && github.run_attempt == 1 && steps.stage-inner.outcome == 'success' @@ -1642,6 +1682,31 @@ jobs: throw "Signed inner artifact did not round-trip cleanly ($($failures.Count) failures)." } + # Why gated separately from the inner restore above: if SignPath's + # windows-inner-binaries-zip artifact configuration does not (yet) cover the + # uninstaller/ directory, the uninstaller comes back missing. That must cost + # only the uninstaller signature — the rebuild below still runs and still + # ships the signed inner binaries, exactly as it does today. + - name: Restore signed uninstaller for the installer rebuild + id: restore-signed-uninstaller + if: matrix.platform == 'win' && github.run_attempt == 1 && steps.restore-signed-inner.outcome == 'success' + continue-on-error: true + shell: pwsh + run: | + $signed = Get-ChildItem -Path signed-inner -Recurse -File -Filter 'orca-uninstaller.exe' | + Select-Object -First 1 + if ($null -eq $signed) { + throw 'SignPath did not return uninstaller/orca-uninstaller.exe; check the windows-inner-binaries-zip artifact configuration covers it.' + } + $signature = Get-AuthenticodeSignature -FilePath $signed.FullName + if ($null -eq $signature.SignerCertificate) { + throw 'The returned NSIS uninstaller carries no signature.' + } + $signedDir = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\signed' + New-Item -ItemType Directory -Force -Path $signedDir | Out-Null + Copy-Item -LiteralPath $signed.FullName -Destination (Join-Path $signedDir 'orca-uninstaller.exe') -Force + Write-Host ("{0,-14} uninstaller <{1}>" -f $signature.Status, $signature.SignerCertificate.Subject) + # Why this step exists: electron-builder's CopyElevateHelper re-copies a # pristine elevate.exe from its download cache over resources\elevate.exe # on EVERY nsis pack — including the --prepackaged rebuild below — which @@ -1651,9 +1716,12 @@ jobs: # no-op. Known quirk: the cache persists across releases via actions/cache, # so later runs may see elevate.exe as already signed and skip staging it — # that is fine (the signature is timestamped) and the evidence gate checks - # elevate.exe in the shipped installer unconditionally. If this ever causes - # trouble, delete this step; the only effect is elevate.exe shipping - # unsigned again, which the evidence gate will flag. + # elevate.exe in the shipped installer unconditionally. + # + # The cache lookup lives in a script because the inline path this step used + # (`\nsis`) matches no app-builder-lib layout, and `SilentlyContinue` + # plus `exit 0` turned that miss into a green step — v1.4.193 and v1.4.194 + # shipped an unsigned elevate.exe that way. A miss now fails the step. - name: Replace cached elevate.exe with the signed copy id: sign-elevate-cache if: matrix.platform == 'win' && github.run_attempt == 1 && steps.restore-signed-inner.outcome == 'success' @@ -1665,20 +1733,26 @@ jobs: Write-Host '::warning::No elevate.exe in win-unpacked resources; nothing to protect from the rebuild clobber.' exit 0 } + # Why this guard stays: windows-signing-rehearsal.yml shares the + # electron-builder-win- cache key with this workflow, so a + # test-certificate elevate.exe must never be staged into a release cache. $signature = Get-AuthenticodeSignature -FilePath $signed $subject = if ($null -eq $signature.SignerCertificate) { '' } else { $signature.SignerCertificate.Subject } if ($signature.Status -ne 'Valid' -or $subject -notlike '*CN=SignPath Foundation*') { Write-Host "::warning::win-unpacked elevate.exe is not SignPath-signed ($($signature.Status), $subject); skipping cache swap." exit 0 } - $cached = @(Get-ChildItem "$env:LOCALAPPDATA\electron-builder\Cache\nsis" -Recurse -Filter elevate.exe -ErrorAction SilentlyContinue) - if ($cached.Count -eq 0) { - Write-Host '::warning::No cached elevate.exe found (electron-builder cache layout changed?); the rebuild will pack the unsigned copy and the evidence gate will flag it.' - exit 0 - } - foreach ($file in $cached) { - Copy-Item -Path $signed -Destination $file.FullName -Force - Write-Host "Replaced $($file.FullName) with the SignPath-signed copy." + node config/scripts/replace-cached-nsis-elevate.mjs $signed + if ($LASTEXITCODE -ne 0) { + $message = 'Cached elevate.exe swap found nothing to replace; the rebuilt installer ships an unsigned UAC elevation helper (issue #7785).' + if ($env:GITHUB_STEP_SUMMARY) { + try { + Add-Content -Path $env:GITHUB_STEP_SUMMARY -Value "**Windows elevate.exe cache swap:** FAILED — $message" -ErrorAction Stop + } catch { + Write-Host "::warning::Could not write the elevate.exe swap verdict to the job summary: $_" + } + } + throw $message } - name: Rebuild NSIS installer from signed unpacked app @@ -1686,6 +1760,11 @@ jobs: if: matrix.platform == 'win' && github.run_attempt == 1 && steps.restore-signed-inner.outcome == 'success' continue-on-error: true shell: pwsh + env: + # Why unconditional: the sign hook keys off the file existing, which it + # only does when the restore step above succeeded. A missing file logs a + # warning and embeds the freshly built unsigned uninstaller instead. + ORCA_WIN_UNINSTALLER_SIGNED_PATH: ${{ runner.temp }}\uninstaller-signing\signed\orca-uninstaller.exe run: | # Why: keep the pre-rebuild artifacts so a failed rebuild can fall # back to shipping them unchanged (fail-open). @@ -1716,6 +1795,7 @@ jobs: with: name: orca-windows-unsigned-${{ needs.cut.outputs.tag }} path: dist/orca-windows-setup.exe + compression-level: 0 if-no-files-found: error # Why: SignPath Foundation production certificates require manual review, @@ -1876,6 +1956,7 @@ jobs: env: ORCA_WINDOWS_INNER_SIGNATURE_REQUIRED: 'false' INNER_SIGNING_COMPLETED: ${{ steps.rebuild-nsis-signed.outcome == 'success' }} + UNINSTALLER_SIGNING_COMPLETED: ${{ steps.restore-signed-uninstaller.outcome == 'success' }} run: | $required = $env:ORCA_WINDOWS_INNER_SIGNATURE_REQUIRED -eq 'true' @@ -1956,6 +2037,39 @@ jobs: if ($targets -notcontains 'resources\elevate.exe') { $targets += 'resources\elevate.exe' } + # Why the uninstaller is not in $targets: NSIS embeds it in its own + # compressed data section (`File /oname=${UNINSTALL_FILENAME}` in + # app-builder-lib templates/nsis/include/installer.nsh), not in the + # app 7z payload extracted above - the bundled 7za cannot see it. + # What the receipt proves and does not: the digest comparison is + # equal by construction (the hook digests the bytes it copied from + # this same file), so the real signal is that the receipt exists at + # all - the import leg ran, and these are the bytes it embedded. The + # signature check below is the part with teeth. The shipped-artifact + # check lives in windows-signing-rehearsal.yml, which installs the + # installer and inspects the uninstaller it drops on disk. + if ($env:UNINSTALLER_SIGNING_COMPLETED -eq 'true') { + $signedUninstaller = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\signed\orca-uninstaller.exe' + $receipt = "$signedUninstaller.embedded-sha256" + if (-not (Test-Path -LiteralPath $receipt)) { + $failures.Add('the sign hook did not embed the signed uninstaller into the rebuilt installer') + } else { + $embedded = (Get-Content -LiteralPath $receipt -Raw).Trim() + $actual = (Get-FileHash -LiteralPath $signedUninstaller -Algorithm SHA256).Hash.ToLowerInvariant() + $signature = Get-AuthenticodeSignature -FilePath $signedUninstaller + $subject = if ($null -eq $signature.SignerCertificate) { '' } else { $signature.SignerCertificate.Subject } + $line = "{0,-14} {1} <{2}>" -f $signature.Status, 'Uninstall Orca.exe (embedded)', $subject + $report.Add($line) + Write-Host $line + if ($embedded -ne $actual) { + $failures.Add("the rebuilt installer embedded different uninstaller bytes than the signed one ($embedded vs $actual)") + } elseif ($signature.Status -ne 'Valid' -or $subject -notlike '*CN=SignPath Foundation*') { + $failures.Add("not signed by SignPath Foundation: Uninstall Orca.exe ($($signature.Status), $subject)") + } + } + } else { + Write-Host '::warning::The NSIS uninstaller was not signed on this run; it is excluded from the evidence gate (fail-open).' + } foreach ($relative in $targets) { $path = Join-Path $root $relative if (-not (Test-Path $path)) { @@ -1988,7 +2102,9 @@ jobs: Add-GateEvidence "VERDICT: FAILED — $message" Add-GateSummary "FAILED — $message" } else { - $ok = "All $($targets.Count) inner binaries in the shipped installer are signed by SignPath Foundation." + # $report, not $targets: the embedded uninstaller is reported but + # is not one of the extracted payload targets. + $ok = "All $($report.Count) checked binaries are signed by SignPath Foundation." Add-GateEvidence "VERDICT: PASSED — $ok" Add-GateSummary "PASSED — $ok" Write-Host $ok diff --git a/.github/workflows/release-ref-validation.yml b/.github/workflows/release-ref-validation.yml new file mode 100644 index 00000000000..6995f9db174 --- /dev/null +++ b/.github/workflows/release-ref-validation.yml @@ -0,0 +1,38 @@ +name: Release ref validation + +on: + pull_request: + paths: + - '.github/workflows/adhoc-mac-build.yml' + - '.github/workflows/dev-channel-win-build.yml' + - '.github/workflows/release-ref-validation.yml' + - 'config/scripts/workflow-ref-reachability.test.mjs' + - 'config/scripts/workflow-ref-mirror-case-safety.test.mjs' + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: release-ref-validation-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + +jobs: + validate: + strategy: + fail-fast: false + matrix: + os: [macos-15, windows-2022] + runs-on: ${{ matrix.os }} + timeout-minutes: 10 + steps: + - uses: actions/checkout@v6 + with: + persist-credentials: false + - uses: ./.github/actions/install-node-dependencies + - name: Verify case-twin refs and release trust boundary + run: >- + pnpm exec vitest run --config config/vitest.config.ts + config/scripts/workflow-ref-reachability.test.mjs + config/scripts/workflow-ref-mirror-case-safety.test.mjs + config/scripts/dev-channel-windows-workflow-contract.test.mjs diff --git a/.github/workflows/skill-update-roundtrip.yml b/.github/workflows/skill-update-roundtrip.yml index 71fcf264f69..96de1101275 100644 --- a/.github/workflows/skill-update-roundtrip.yml +++ b/.github/workflows/skill-update-roundtrip.yml @@ -22,6 +22,10 @@ on: - main paths: *skill-roundtrip-paths +concurrency: + group: skill-roundtrip-${{ github.event_name }}-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + jobs: roundtrip: strategy: @@ -41,7 +45,9 @@ jobs: steps: - uses: actions/checkout@v6 with: + # Historical skill snapshots need tags, but only their blobs are read. fetch-depth: 0 + filter: blob:none persist-credentials: false - uses: actions/setup-node@v6 with: diff --git a/.github/workflows/terminal-ime-e2e.yml b/.github/workflows/terminal-ime-e2e.yml index b9957b2daa9..1ab905d8783 100644 --- a/.github/workflows/terminal-ime-e2e.yml +++ b/.github/workflows/terminal-ime-e2e.yml @@ -38,23 +38,9 @@ jobs: xfwm4 xvfb - - name: Setup Node.js - uses: actions/setup-node@v6 + - uses: ./.github/actions/install-node-dependencies with: - node-version-file: package.json - - - name: Setup pnpm - uses: pnpm/setup@v2 - with: - install: false - - - name: Use external node-gyp to avoid pnpm bundled copy - run: | - npm install -g node-gyp@11.5.0 - echo "npm_config_node_gyp=$(npm root -g)/node-gyp/bin/node-gyp.js" >> "$GITHUB_ENV" - - - name: Install dependencies - run: pnpm install --frozen-lockfile + native-runtime: electron - name: Build Electron app for E2E run: pnpm exec electron-vite build --mode e2e diff --git a/.github/workflows/terminal-perf.yml b/.github/workflows/terminal-perf.yml index 38d0a25bbb7..72a0fec8992 100644 --- a/.github/workflows/terminal-perf.yml +++ b/.github/workflows/terminal-perf.yml @@ -67,16 +67,17 @@ jobs: - name: Install native build tools and xvfb run: sudo apt-get update && sudo apt-get install -y build-essential python3 xvfb zsh - - name: Setup Node.js - uses: actions/setup-node@v6 - with: - node-version-file: package.json - - name: Setup pnpm uses: pnpm/setup@v2 with: install: false + - name: Setup Node.js + uses: actions/setup-node@v6 + with: + node-version-file: package.json + cache: pnpm + # Why: this scheduled/manual workflow uses the same native install path as # PR and E2E CI, which needs pnpm to bypass its bundled gyp_main.py. - name: Use external node-gyp to avoid pnpm's bundled copy diff --git a/.github/workflows/windows-signing-rehearsal.yml b/.github/workflows/windows-signing-rehearsal.yml index 54b751908dc..90ab8db137c 100644 --- a/.github/workflows/windows-signing-rehearsal.yml +++ b/.github/workflows/windows-signing-rehearsal.yml @@ -3,9 +3,11 @@ # Why: SignPath cannot deep-sign inside NSIS installers, so shipping signed # inner binaries (Orca.exe, node-pty *.node, DLLs — see issue #7785) requires # a two-request flow: sign the unpacked PE files first, then build the NSIS -# installer from the signed tree, then sign the installer. This workflow -# rehearses that entire flow from a branch, end to end, without publishing -# anything — so the release pipeline on main is never at risk while we verify. +# installer from the signed tree, then sign the installer. The NSIS uninstaller +# rides that same first request — it is captured through electron-builder's sign +# hook and swapped back in during the rebuild — so it adds no third approval. +# This workflow rehearses that entire flow from a branch, end to end, without +# publishing anything — so the release pipeline on main is never at risk. # # Runs only via manual dispatch. Use the test-signing policy for iteration # (auto-approved test certificate) and release-signing to rehearse the @@ -81,15 +83,27 @@ jobs: env: NODE_OPTIONS: --max-old-space-size=4096 - - name: Package unpacked Windows app + # Why a full --win build and not --dir: the NSIS uninstaller only exists + # inside the installer build, and it is the file the MDE update cluster + # flags. --dir would never produce it, so the rehearsal would not rehearse + # the uninstaller leg at all. This mirrors release-cut's first Windows pass. + - name: Package Windows app and export the NSIS uninstaller shell: pwsh + env: + # runner.temp, never the workspace: the all-negation `files` list in + # config/electron-builder.config.cjs packs whatever is left in the + # checkout root into app.asar. + ORCA_WIN_UNINSTALLER_EXPORT_PATH: ${{ runner.temp }}\uninstaller-signing\unsigned\orca-uninstaller.exe run: | node config/scripts/ensure-native-runtime.mjs --runtime=electron if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } - pnpm exec electron-builder --config config/electron-builder.config.cjs --win --dir --publish never + pnpm exec electron-builder --config config/electron-builder.config.cjs --win --publish never if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } if (-not (Test-Path 'dist/win-unpacked/Orca.exe')) { - throw 'electron-builder --dir did not produce dist/win-unpacked/Orca.exe' + throw 'electron-builder --win did not produce dist/win-unpacked/Orca.exe' + } + if (-not (Test-Path -LiteralPath $env:ORCA_WIN_UNINSTALLER_EXPORT_PATH)) { + throw "The sign hook did not export the NSIS uninstaller to $env:ORCA_WIN_UNINSTALLER_EXPORT_PATH" } # Why: only unsigned PE files go to SignPath. Files that already carry a @@ -132,6 +146,17 @@ jobs: Write-Host "Skipped $($skipped.Count) already-signed files:" $skipped | ForEach-Object { Write-Host " $_" } + # Why kept out of inner-signing-list.txt: that list drives the copy-back + # into dist/win-unpacked, and the uninstaller does not live there — it is + # re-injected through the electron-builder sign hook during the rebuild. + # No catch here, unlike the release job: the rehearsal exists to prove + # the flow, so a staging failure must fail it loudly. + $exportedUninstaller = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\unsigned\orca-uninstaller.exe' + $uninstallerStagePath = Join-Path $stage.FullName 'uninstaller\orca-uninstaller.exe' + New-Item -ItemType Directory -Force -Path (Split-Path $uninstallerStagePath) | Out-Null + Copy-Item -LiteralPath $exportedUninstaller -Destination $uninstallerStagePath -Force + Write-Host 'Staged the NSIS uninstaller for signing: uninstaller\orca-uninstaller.exe' + - name: Upload unsigned inner binaries for SignPath id: upload-unsigned-inner uses: actions/upload-artifact@v7 @@ -200,8 +225,27 @@ jobs: throw "Signed inner artifact did not round-trip cleanly ($($failures.Count) failures)." } + - name: Restore signed uninstaller for the installer rebuild + shell: pwsh + run: | + $signed = Get-ChildItem -Path signed-inner -Recurse -File -Filter 'orca-uninstaller.exe' | + Select-Object -First 1 + if ($null -eq $signed) { + throw 'SignPath did not return uninstaller/orca-uninstaller.exe; check the inner-binaries artifact configuration covers it.' + } + $signature = Get-AuthenticodeSignature -FilePath $signed.FullName + if ($null -eq $signature.SignerCertificate) { + throw 'The returned NSIS uninstaller carries no signature.' + } + $signedDir = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\signed' + New-Item -ItemType Directory -Force -Path $signedDir | Out-Null + Copy-Item -LiteralPath $signed.FullName -Destination (Join-Path $signedDir 'orca-uninstaller.exe') -Force + Write-Host ("{0,-14} uninstaller <{1}>" -f $signature.Status, $signature.SignerCertificate.Subject) + - name: Build NSIS installer from signed unpacked app shell: pwsh + env: + ORCA_WIN_UNINSTALLER_SIGNED_PATH: ${{ runner.temp }}\uninstaller-signing\signed\orca-uninstaller.exe run: | pnpm exec electron-builder --config config/electron-builder.config.cjs --win --publish never --prepackaged "$env:GITHUB_WORKSPACE\dist\win-unpacked" if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } @@ -215,6 +259,7 @@ jobs: with: name: orca-windows-installer-unsigned-${{ github.run_id }} path: dist/orca-windows-setup.exe + compression-level: 0 if-no-files-found: error - name: Submit Windows installer signing request @@ -288,20 +333,33 @@ jobs: run: | $report = New-Object System.Collections.Generic.List[string] $failures = New-Object System.Collections.Generic.List[string] + $advisories = New-Object System.Collections.Generic.List[string] $requireValid = $env:SIGNING_POLICY -eq 'release-signing' - function Test-Signature([string]$label, [string]$path) { + # -Advisory records a problem without failing the run. It exists for + # exactly one file (resources\elevate.exe, below) and must not be + # widened casually: the point of this workflow is to fail when signing + # is broken. + function Test-Signature([string]$label, [string]$path, [switch]$Advisory) { $signature = Get-AuthenticodeSignature -FilePath $path $subject = if ($null -eq $signature.SignerCertificate) { '' } else { $signature.SignerCertificate.Subject } $line = "{0,-14} {1} <{2}>" -f $signature.Status, $label, $subject $script:report.Add($line) Write-Host $line + $problem = $null if ($null -eq $signature.SignerCertificate -or $signature.Status -eq 'NotSigned') { - $script:failures.Add("unsigned: $label") + $problem = "unsigned: $label" } elseif ($script:requireValid -and $signature.Status -ne 'Valid') { - $script:failures.Add("not Valid under release-signing: $label ($($signature.Status))") + $problem = "not Valid under release-signing: $label ($($signature.Status))" } elseif ($script:requireValid -and $subject -notlike '*CN=SignPath Foundation*') { - $script:failures.Add("unexpected signer: $label ($subject)") + $problem = "unexpected signer: $label ($subject)" + } + if ($null -eq $problem) { return } + if ($Advisory) { + $script:advisories.Add($problem) + Write-Host "::warning::$problem - known pre-existing issue, not failing the rehearsal" + } else { + $script:failures.Add($problem) } } @@ -323,21 +381,155 @@ jobs: & $7za x 'dist/orca-windows-setup.exe' '-oextracted-app' -y | Out-Null $root = Resolve-Path 'extracted-app' + # The receipt only proves the import leg ran; it cannot prove what NSIS + # embedded, because the uninstaller lives in a compressed NSIS data + # section rather than the app 7z payload above and the bundled 7za has + # no NSIS handler. So the rehearsal - unlike the release job, which + # must not mutate the runner it publishes from - goes all the way: it + # installs the installer silently and inspects the uninstaller the + # installer actually wrote to disk. That is the file MDE flags. + $signedUninstaller = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\signed\orca-uninstaller.exe' + $receipt = "$signedUninstaller.embedded-sha256" + if (-not (Test-Path -LiteralPath $receipt)) { + $failures.Add('the sign hook did not embed the signed uninstaller into the rebuilt installer') + } else { + Test-Signature 'relayed: orca-uninstaller.exe' $signedUninstaller + } + + # Why a full 7-Zip attempt first: it is non-invasive. The runner image + # ships the complete 7z.exe, which - unlike the reduced 7za - has an + # NSIS handler. If it cannot read the section either, fall back to a + # real silent install. + $installedUninstaller = $null + $installedVia = $null + $expectedDigest = if (Test-Path -LiteralPath $receipt) { (Get-Content -LiteralPath $receipt -Raw).Trim() } else { $null } + $full7z = 'C:\Program Files\7-Zip\7z.exe' + if (Test-Path -LiteralPath $full7z) { + New-Item -ItemType Directory -Path nsis-extract -Force | Out-Null + & $full7z x -tnsis 'dist/orca-windows-setup.exe' '-onsis-extract' -y 2>&1 | Out-Null + $installedUninstaller = Get-ChildItem -Path nsis-extract -Recurse -File -Filter 'Uninstall*.exe' -ErrorAction SilentlyContinue | + Select-Object -First 1 + # Why the digest guard before trusting this route: 7-Zip's NSIS + # handler emits partial or garbled output on some NSIS builds, and a + # truncated extract would score NotSigned and fail the rehearsal as + # "the shipped uninstaller is unsigned" when nothing is wrong. Only + # trust it when it reproduces the bytes the relay embedded; otherwise + # fall through to the install route, which is ground truth. A name + # miss (the handler labelling the entry by its source name) falls + # through the same way. + if ($null -ne $installedUninstaller -and $null -ne $expectedDigest -and + (Get-FileHash -LiteralPath $installedUninstaller.FullName -Algorithm SHA256).Hash.ToLowerInvariant() -ne $expectedDigest) { + Write-Host "7-Zip's NSIS output did not match the relayed digest; falling back to a silent install." + $installedUninstaller = $null + } + if ($null -ne $installedUninstaller) { + $installedVia = "7-Zip's NSIS handler" + Write-Host "Read the embedded uninstaller with 7-Zip's NSIS handler: $($installedUninstaller.FullName)" + } else { + Write-Host "7-Zip's NSIS handler did not yield a usable uninstaller; falling back to a silent install." + } + } + + if ($null -eq $installedUninstaller) { + # Nothing here is published, so mutating this runner is free. + # Why -PassThru and a bounded wait rather than -Wait: a bare -Wait on + # an installer that ever prompts hangs to the job's 360-minute cap. + $installerProcess = Start-Process -FilePath (Resolve-Path 'dist/orca-windows-setup.exe') -ArgumentList '/S' -PassThru + if (-not $installerProcess.WaitForExit(300000)) { + $installerProcess | Stop-Process -Force -ErrorAction SilentlyContinue + $failures.Add('the silent install did not exit within 5 minutes; it is likely prompting') + } + # Why a poll rather than one Stop-Process: the oneClick installer + # launches the app as it finishes, so Orca.exe can appear *after* the + # installer process exits. A single silenced Stop-Process would miss + # it and leave Orca plus orca-terminal-daemon.exe holding handles + # under %LOCALAPPDATA%\Programs for the rest of the job. + for ($attempt = 0; $attempt -lt 20; $attempt++) { + $running = @(Get-Process -Name 'Orca' -ErrorAction SilentlyContinue) + if ($running.Count -gt 0) { + $running | Stop-Process -Force -ErrorAction SilentlyContinue + break + } + Start-Sleep -Milliseconds 500 + } + Get-Process -Name 'orca-terminal-daemon' -ErrorAction SilentlyContinue | + Stop-Process -Force -ErrorAction SilentlyContinue + $installedUninstaller = Get-ChildItem -Path "$env:LOCALAPPDATA\Programs" -Recurse -File -Filter 'Uninstall*.exe' -ErrorAction SilentlyContinue | + Where-Object { $_.FullName -like '*Orca*' } | + Select-Object -First 1 + if ($null -ne $installedUninstaller) { $installedVia = 'a silent install' } + } + + if ($null -eq $installedUninstaller) { + $failures.Add('could not obtain the uninstaller the installer ships; neither 7-Zip nor a silent install produced it') + } else { + # Why this digest comparison is the point of the whole rehearsal: + # unlike the release job's, it hashes a file NSIS itself wrote out + # rather than the file the hook copied, so it is the only check that + # proves the shipped installer embedded the SignPath-signed bytes. On + # the 7-Zip route the guard above already forced equality; on the + # install route this is the first time it is tested. + if ($null -ne $expectedDigest) { + $shippedDigest = (Get-FileHash -LiteralPath $installedUninstaller.FullName -Algorithm SHA256).Hash.ToLowerInvariant() + if ($shippedDigest -ne $expectedDigest) { + $failures.Add("the uninstaller the installer ships is not the relayed one (via $installedVia): $shippedDigest vs $expectedDigest") + } + } + Test-Signature "shipped: Uninstall Orca.exe (via $installedVia)" $installedUninstaller.FullName + } + foreach ($relative in Get-Content 'inner-signing-list.txt') { $path = Join-Path $root $relative if (-not (Test-Path $path)) { $failures.Add("missing from installer payload: $relative") continue } - Test-Signature "installed: $relative" $path + # Why elevate.exe alone is advisory: app-builder-lib re-copies the + # pristine cached elevate.exe over resources\elevate.exe on EVERY nsis + # pack - AppPackageHelper.packArch calls elevateHelper.copy() before + # buildAppPackage (nsisUtil.js), and CopyElevateHelper.copy does + # `copyFile(elevatePath, outFile, false)` then `signIf(outFile)`, which + # signs nothing because this build configures no certificate. So the + # signed copy restored into win-unpacked is clobbered by the rebuild. + # This predates the uninstaller relay and is not caused by it: with no + # `sign` hook, signIf already returned false at "no signing info + # identified" (windowsSignToolManager.js), so no signtool call was + # displaced. release-cut.yml mitigates it separately by pre-seeding the + # electron-builder cache ("Replace cached elevate.exe with the signed + # copy"); this workflow has no such step, which is why the clobber is + # visible here and not there. Mirroring that step here would not help: + # it only swaps when the copy is already Valid and SignPath-signed, so + # it no-ops under the test certificate. + # + # DO NOT relax that Valid + SignPath-signed guard to make this + # rehearsal go green. This workflow and release-cut.yml share the + # cache key `electron-builder-win-`, and that guard is + # the only thing stopping a test certificate from being seeded into + # the cache a real release restores from. Shipping users a binary + # signed by "Test certificate for 'Orca agent ide [OSS]'" is worse + # than shipping it unsigned. + # + # Fixing elevate.exe belongs in its own PR - it is a UAC elevation + # helper, and it deserves more scrutiny than a footnote in an + # uninstaller change. + if ($relative -eq 'resources\elevate.exe') { + Test-Signature "installed: $relative" $path -Advisory + } else { + Test-Signature "installed: $relative" $path + } } + if ($advisories.Count -gt 0) { + $report.Add('') + $report.Add('ADVISORY (known pre-existing, did not fail this run):') + $advisories | ForEach-Object { $report.Add(" $_") } + } Set-Content -Path 'signing-evidence.txt' -Value ($report -join "`n") if ($failures.Count -gt 0) { $failures | ForEach-Object { Write-Host "::error::$_" } throw "Signing rehearsal failed with $($failures.Count) problems." } - Write-Host "All $((Get-Content 'inner-signing-list.txt').Count) inner binaries plus the installer are signed." + Write-Host "All checked binaries are signed, including the uninstaller the installer writes to disk ($($advisories.Count) advisory)." - name: Upload rehearsal evidence and installer if: always() diff --git a/.gitignore b/.gitignore index 8be3fc5b6f4..6722fc5ae54 100644 --- a/.gitignore +++ b/.gitignore @@ -110,6 +110,8 @@ docs/** !docs/reference/macos-press-and-hold.md !docs/reference/orcad-operations.md !docs/reference/relay-grace-time-reconfiguration.md +!docs/reference/windows-cmd-shim-resolution.md +!docs/reference/windows-daemon-host-relocation.md !docs/reference/windows-edr-posture.md !docs/reference/windows-process-enumeration.md !docs/reference/wsl-runner-verification.md diff --git a/.oxlintrc.json b/.oxlintrc.json index 77a7e43e807..03cc659f494 100644 --- a/.oxlintrc.json +++ b/.oxlintrc.json @@ -2,6 +2,10 @@ "$schema": "./node_modules/oxlint/configuration_schema.json", "plugins": ["typescript", "react", "react-hooks", "react-perf", "unicorn"], "jsPlugins": [ + { + "name": "sort-comparator-performance", + "specifier": "./config/oxlint-plugins/sort-comparator-performance.mjs" + }, { "name": "mobile-pairing", "specifier": "./config/oxlint-plugins/mobile-pairing-qrcode-import.mjs" @@ -23,6 +27,7 @@ "correctness": "error" }, "rules": { + "sort-comparator-performance/no-repeated-collator": "warn", "app-store-performance/require-selector": "error", "app-store-performance/no-identity-selector": "error", "app-store-performance/no-fresh-selector-result": "error", diff --git a/AGENTS.md b/AGENTS.md index 8b0156ba6b1..b0947da0c2f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -4,6 +4,12 @@ All UI work — layout, color, typography, spacing, component selection, UX beha ## Electron UI Validation +Always run tests and agent-launched apps in the background with `ORCA_BACKGROUND_LAUNCH=1`. +Never steal monitor focus or reveal test windows: no `show()`, `showInactive()`, `bringToFront()`, +`app.focus()`, or OS activation. Use CDP screenshots of hidden renderers. Keep native-focus and +visible-window tests paused on the user's desktop; run them on an isolated display or CI. +Rebuild modified launch-policy code before running an app; stale build wrappers are not safe. + Use the `$electron` skill and Playwright CDP for rendered Orca UI checks. Do not use computer-use for Orca UI validation. # Style @@ -47,8 +53,9 @@ Orca targets macOS, Linux, and Windows. Keep all platform-dependent behavior beh - **Shortcut labels in UI**: Display `⌘` / `⇧` on Mac and `Ctrl+` / `Shift+` on other platforms. - **File paths**: Use `path.join` or Electron/Node path utilities — never assume `/` or `\`. - **Windows setup scripts**: the setup/issue-command runner is a `.cmd` batch file unless the script starts with a `#!` line — never derive that from the user's terminal-shell preference, and never launch a `.cmd` runner with a bare `cmd.exe /c` from a Git Bash pane (MSYS rewrites the `/c`). See [`docs/reference/windows-setup-shell.md`](./docs/reference/windows-setup-shell.md). -- **Windows child processes**: start them through `runProcess`/`spawnProcess` in `src/shared/child-process/` — never `child_process` directly. It pins `windowsHide`, refuses `shell: true`, and encodes `.cmd`/`.bat` arguments so neither `CommandLineToArgvW` nor `cmd.exe` mangles them. A ratchet test fails on any new direct import. +- **Windows child processes**: start them through `runProcess`/`spawnProcess` in `src/shared/child-process/` — never `child_process` directly. It pins `windowsHide`, refuses `shell: true`, and encodes `.cmd`/`.bat` arguments so neither `CommandLineToArgvW` nor `cmd.exe` mangles them. A ratchet test fails on any new direct import. Recognised npm/pnpm `.cmd` shims are resolved to their real target so the spawn skips `cmd.exe` entirely; see [`docs/reference/windows-cmd-shim-resolution.md`](./docs/reference/windows-cmd-shim-resolution.md) before adding a shim shape or debugging one. - **Windows process enumeration**: read the table through `src/main/windows/windows-process-table.ts`, never by forking `powershell.exe`. See [`docs/reference/windows-process-enumeration.md`](./docs/reference/windows-process-enumeration.md). +- **Windows daemon-host relocation**: the terminal daemon runs from a copy of the app runtime under `%LOCALAPPDATA%`, which is what survives an auto-update. Before touching that copy, its exe name, or the NSIS uninstall macro, read [`docs/reference/windows-daemon-host-relocation.md`](./docs/reference/windows-daemon-host-relocation.md). - **Windows EDR signal**: don't add `-ExecutionPolicy Bypass`, `-EncodedCommand`, `cmd.exe /c` with escaped free text, per-operation interpreter spawning, or runtime `Add-Type` compilation without reading [`docs/reference/windows-edr-posture.md`](./docs/reference/windows-edr-posture.md) first — behavioural EDR scores each of those, and being signed does not clear them. - **WSL commands**: build argv with `buildWslExecArgs` (always `--exec` — under `--`, `wsl.exe` expands `$name` in every argument and silently rewrites the script), and fence anything whose stdout you parse with `buildWslCapturedLoginShellCommand`, because the interactive login shell prints the distro banner to stdout. See [`docs/reference/wsl-command-execution.md`](./docs/reference/wsl-command-execution.md). - **Linux native modules**: keep the glibc floor at Ubuntu 20.04 / glibc 2.31. A module compiled from source on a newer runner can reference symbol versions absent on the floor and crash the app on startup. See [`docs/reference/linux-glibc-compatibility.md`](./docs/reference/linux-glibc-compatibility.md); packaging fails if a bundled native binary needs newer glibc. diff --git a/README.md b/README.md index 7a3cbe2360c..2ae59035da8 100644 --- a/README.md +++ b/README.md @@ -238,9 +238,9 @@ Pair with your desktop app to monitor and steer your agents from your phone. - **Discord:** Join the community on **[Discord](https://discord.gg/fzjDKHxv8Q)**. - **Twitter / X:** Follow **[@orca_build](https://x.com/orca_build)** for updates and announcements. -- **WeChat:** Scan to join the Orca community WeChat group 8. +- **WeChat:** Scan to join the Orca community WeChat group 8. Group 8 may be full; if so, scan the Group 9 QR code instead. - WeChat group 8 QR code for the Orca community + WeChat group 8 QR code for the Orca community  WeChat group 9 QR code for the Orca community - **Feedback & Ideas:** We ship fast. Missing something? [Request a new feature](https://github.com/stablyai/orca/issues). - **Privacy:** See the [privacy & telemetry docs](https://www.onorca.dev/docs/telemetry) for what anonymous usage data Orca collects and how to opt out. diff --git a/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts b/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts index 18ee4495078..b642d3cc3e1 100644 --- a/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts +++ b/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts @@ -6,7 +6,10 @@ import { livePreflightGcloud, runIncidentLivePreflight } from './incident-live-preflight-cli.js' -import type { IncidentSample } from './incident-monitor.js' +import { + INCIDENT_MONITOR_THRESHOLDS, + type IncidentSample +} from './incident-monitor.js' import type { AdmissionSelector } from './incident-selector.js' const directories: string[] = [] @@ -69,6 +72,7 @@ function sample(): IncidentSample { expectedSelector: selector, cells: [{ cellId: 'production-gce-c1', + region: 'us-central1', runtimeKnown: true, powered: true, expectedAdmissionState: 'existing-only' @@ -250,6 +254,30 @@ describe('relay incident live preflight', () => { )).rejects.toThrow('cloud-monitoring/threshold_max') }) + // Why: a frozen wave has to name what froze it without re-reading the sample. + it('names the signal and its numbers in the failure message', async () => { + const slowCell = sample() + slowCell.sources['active-probe']!.signals[ + 'cell.production-gce-c1.latency_ms' + ]!.value = 2_568 + await expect(runIncidentLivePreflight( + ['--state-file', stateFile()], + { now: () => now, collect: async () => slowCell } + )).rejects.toThrow( + 'relay live preflight failed: active-probe/threshold_max cell.production-gce-c1.latency_ms observed=2568 threshold=2000' + ) + + // A failure with no signal keeps the source/code token and drops the rest. + const stale = sample() + stale.sources['active-probe']!.observedAt = new Date(now - 60_001).toISOString() + await expect(runIncidentLivePreflight( + ['--state-file', stateFile()], + { now: () => now, collect: async () => stale } + )).rejects.toThrow( + 'relay live preflight failed: active-probe/source_stale observed=60001 threshold=60000' + ) + }) + it('enforces the signed migration policy', async () => { const inactiveTarget = sample() inactiveTarget.sources['director-admin']!.signals[ @@ -313,7 +341,7 @@ describe('relay incident live preflight', () => { it('retries freshness-only failures when explicitly requested', async () => { const stale = sample() stale.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = - new Date(now - 180_001).toISOString() + new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() const missing = sample() delete missing.sources['relay-logs'] const collect = vi.fn() @@ -331,11 +359,44 @@ describe('relay incident live preflight', () => { expect(wait).toHaveBeenNthCalledWith(2, 15_000) }) + it('retries a first-wave stale sample and passes on the fresh one', async () => { + const stale = sample() + stale.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = + new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() + const collect = vi.fn().mockResolvedValueOnce(stale).mockResolvedValueOnce(sample()) + const wait = vi.fn(async () => undefined) + await expect(runIncidentLivePreflight( + ['--state-file', stateFile(), '--wave-index', '0', '--retry-freshness'], + { now: () => now, collect, wait } + )).resolves.toBeUndefined() + expect(collect).toHaveBeenCalledTimes(2) + expect(wait).toHaveBeenCalledOnce() + }) + + it('stops retrying when the next wait would exceed the evidence-age bound', async () => { + const completedAt = now - 290_000 + const stale = sample() + stale.sources['cloud-monitoring']!.observedAt = new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() + const collect = vi.fn(async () => stale) + const wait = vi.fn(async () => undefined) + await expect(runIncidentLivePreflight( + ['--state-file', stateFile('strict', { + startedAt: new Date(completedAt - 17 * 60_000).toISOString(), + windowStartedAt: new Date(completedAt - 16 * 60_000).toISOString(), + lastSampleAt: new Date(completedAt - 30_000).toISOString(), + completedAt: new Date(completedAt).toISOString() + }), '--retry-freshness'], + { now: () => now, collect, wait } + )).rejects.toThrow('cloud-monitoring/source_stale') + expect(collect).toHaveBeenCalledOnce() + expect(wait).not.toHaveBeenCalled() + }) + it('does not retry a threshold failure', async () => { const unhealthy = sample() unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.value = 0.9 unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = - new Date(now - 180_001).toISOString() + new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() const collect = vi.fn(async () => unhealthy) const wait = vi.fn(async () => undefined) await expect(runIncidentLivePreflight( @@ -348,7 +409,7 @@ describe('relay incident live preflight', () => { it('fails closed after the bounded freshness retry window', async () => { const stale = sample() - stale.sources['cloud-monitoring']!.observedAt = new Date(now - 180_001).toISOString() + stale.sources['cloud-monitoring']!.observedAt = new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() const collect = vi.fn(async () => stale) const wait = vi.fn(async () => undefined) await expect(runIncidentLivePreflight( diff --git a/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts b/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts index e82627a3e80..c277325ed84 100644 --- a/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts +++ b/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts @@ -7,7 +7,9 @@ import { suppliedIdentityToken } from './incident-monitor-cli.js' import { AdmissionSelectorSchema, type AdmissionSelector } from './incident-selector.js' import { evaluateIncidentSample, + FRESHNESS_FAILURE_CODES, preDrainDryRunPassed, + type IncidentFailure, type IncidentSample } from './incident-monitor.js' import { createIncidentSampleCollector } from './incident-monitor-sources.js' @@ -18,12 +20,6 @@ const MONITOR_EVIDENCE_MAX_AGE_MS = 5 * 60_000 // Matches the same-cap cell job timeout-minutes; bounds each predecessor wave. const WAVE_PREDECESSOR_TIMEOUT_MS = 75 * 60_000 const WAVE_INDEX_PATTERN = /^[0-3]$/ -const FRESHNESS_FAILURE_CODES = new Set([ - 'signal_missing', - 'signal_stale', - 'source_missing', - 'source_stale' -]) export function livePreflightGcloud( gcloud: ReturnType, @@ -74,6 +70,17 @@ const PreflightStateSchema = z.object({ } }) +// Keep the source/code prefix other tooling matches on, then name the signal and +// its numbers so a frozen wave is attributable without re-reading the sample. +function describeFailure(failure: IncidentFailure): string { + const detail = [ + failure.signal, + failure.observed === undefined ? null : `observed=${failure.observed}`, + failure.threshold === undefined ? null : `threshold=${failure.threshold}` + ].filter((part): part is string => part !== null && part !== undefined) + return [`${failure.source}/${failure.code}`, ...detail].join(' ') +} + export async function runIncidentLivePreflight( argv: string[], dependencies: { @@ -173,10 +180,14 @@ export async function runIncidentLivePreflight( const freshnessOnly = evaluation.failures.every((failure) => FRESHNESS_FAILURE_CODES.has(failure.code) ) - if (!freshnessOnly || attempt === attempts) { + // Waiting must never carry the mutation past the same evidence-age bound + // the entry check enforces, so the wave budget also caps the retry window. + const budgetExhausted = + now() + FRESHNESS_RETRY_INTERVAL_MS - completedAt > maxEvidenceAgeMs + if (!freshnessOnly || attempt === attempts || budgetExhausted) { throw new Error( `relay live preflight failed: ${evaluation.failures - .map((failure) => `${failure.source}/${failure.code}`) + .map(describeFailure) .join(',')}` ) } diff --git a/cloud/apps/relay-ops/src/incident-monitor-cli.test.ts b/cloud/apps/relay-ops/src/incident-monitor-cli.test.ts index 3e1f20a3cbb..31e5a131d35 100644 --- a/cloud/apps/relay-ops/src/incident-monitor-cli.test.ts +++ b/cloud/apps/relay-ops/src/incident-monitor-cli.test.ts @@ -44,6 +44,7 @@ function sample(at: number): IncidentSample { expectedSelector: selector, cells: [{ cellId, + region: 'us-central1', runtimeKnown: true, powered: true, expectedAdmissionState: 'general' diff --git a/cloud/apps/relay-ops/src/incident-monitor-cli.ts b/cloud/apps/relay-ops/src/incident-monitor-cli.ts index e090be7ea58..adfe6cad480 100644 --- a/cloud/apps/relay-ops/src/incident-monitor-cli.ts +++ b/cloud/apps/relay-ops/src/incident-monitor-cli.ts @@ -50,6 +50,8 @@ const StateSchema = z.object({ continuityEvents: z.array(z.object({ recordedAt: z.string(), windowSequence: z.number().int().nonnegative(), + // Pre-2026-09-05 state files predate tolerated freshness gaps. + tolerated: z.boolean().default(false), failures: z.array(z.object({ code: z.string(), source: z.enum(['active-probe', 'cloud-monitoring', 'relay-logs', 'director-admin']), diff --git a/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts b/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts index 09b7b16fa45..93000131327 100644 --- a/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts +++ b/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts @@ -93,7 +93,7 @@ describe('incident monitor sources', () => { }) it('zero-fills an expired sparse lock-wait point', async () => { - let pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + let pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs const fetchImpl: typeof fetch = async () => Response.json({ timeSeries: [{ points: [{ @@ -141,7 +141,7 @@ describe('incident monitor sources', () => { it('freshens a sparse zero without masking a recent nonzero lock wait', async () => { let value = 0 - const pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + const pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs const readAt = now + 11_879 const fetchImpl: typeof fetch = async () => Response.json({ timeSeries: [{ @@ -317,6 +317,49 @@ describe('incident monitor sources', () => { }, endAt)).toBeNull() }) + // Why: the per-region cell latency bar is only correct if the tfvars region + // reaches the evaluator on every cell expectation. + it('carries the configured region onto every cell expectation', async () => { + const gcloud: GcloudClient = { + accessToken: async () => 'unused', + identityToken: async () => 'unused' + } + const selector = { + generation: 1, + membership: { + existingOnly: [], + migrationOnly: [], + general: productionCells + } + } + const fetchImpl: typeof fetch = async (_input, init) => { + const body = JSON.parse(String(init?.body)) as { cellId?: string; sourceCellId?: string } + if (!body.cellId && !body.sourceCellId) return Response.json({ selector }) + if (body.cellId) { + return Response.json({ + status: { + enabled: true, + connectionCapacity: { hardCap: 600 }, + runtime: { lastHeartbeatAt: now - 1_000, heartbeatFresh: true } + } + }) + } + return Response.json({ + blocked: 0, + blockedExpiredUnregistered: 0, + registeredTargetInactive: 0 + }) + } + const result = await directorSignals('production', selector, gcloud, now, fetchImpl) + const regionById = new Map(result.cells.map((cell) => [cell.cellId, cell.region])) + expect(regionById.get('production-gce-c1')).toBe('us-central1') + expect(regionById.get('production-gce-c27')).toBe('asia-east2') + expect(result.cells).toHaveLength(productionCells.length) + for (const cell of RELAY_OPS_ENVIRONMENTS.production.cells) { + expect(regionById.get(cell.cellId)).toBe(cell.region) + } + }) + it('aggregates admin state without returning tokens or response identities', async () => { const identityToken = 'secret.header.signature' const sensitiveIdentity = 'user@example.test' diff --git a/cloud/apps/relay-ops/src/incident-monitor-sources.ts b/cloud/apps/relay-ops/src/incident-monitor-sources.ts index 0b78c2c4f6b..a93c01b6099 100644 --- a/cloud/apps/relay-ops/src/incident-monitor-sources.ts +++ b/cloud/apps/relay-ops/src/incident-monitor-sources.ts @@ -95,7 +95,7 @@ export const GOOGLE_METRICS: GoogleMetricDefinition[] = [ 'resource.type="cloudsql_database" AND metric.label."wait_event_type"="Lock"', aggregation: 'latest-max', emptyIsZero: true, - zeroAfterMs: INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + zeroAfterMs: INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs }, { signal: 'cloud_sql.deadlocks', @@ -558,6 +558,7 @@ export async function directorSignals( selector, cells: statuses.map(({ cell }) => ({ cellId: cell.cellId, + region: cell.region, runtimeKnown: true, powered: true, expectedAdmissionState: selectorCellState(expectedSelector, cell.cellId) diff --git a/cloud/apps/relay-ops/src/incident-monitor.test.ts b/cloud/apps/relay-ops/src/incident-monitor.test.ts index 4e1da9fab26..c1a073cde4a 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.test.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from 'vitest' import { evaluateIncidentSample, INCIDENT_CHECKPOINT_MINUTES, + INCIDENT_FRESHNESS_TOLERANCE_SAMPLES, INCIDENT_MONITOR_THRESHOLDS, INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS, initialIncidentMonitorState, @@ -32,6 +33,7 @@ function healthySample(at = startedAt): IncidentSample { expectedSelector: selector, cells: [{ cellId: 'production-gce-c1', + region: 'us-central1', runtimeKnown: true, powered: true, expectedAdmissionState: 'general' @@ -150,6 +152,52 @@ describe('incident monitor evaluator', () => { }) }) + // Why: an asia-east2 cell's /ready reaches auth and Cloud SQL in us-central1, so + // from the US runner it measures p50 0.88 s / max 2.7 s and the flat 2 000 bar + // froze three healthy gates on 2026-09-05 (c27 at 2568/2668/2685 ms). + it('holds cell endpoint latency to a per-region bar', () => { + const asiaTail = healthySample() + asiaTail.cells[0]!.region = 'asia-east2' + asiaTail.sources['active-probe']!.signals['cell.production-gce-c1.latency_ms'] = + signal(2_685) + expect(evaluateIncidentSample(asiaTail, startedAt)).toMatchObject({ + status: 'green', + failures: [] + }) + + const asiaIncident = healthySample() + asiaIncident.cells[0]!.region = 'asia-east2' + asiaIncident.sources['active-probe']!.signals['cell.production-gce-c1.latency_ms'] = + signal(4_001) + expect(evaluateIncidentSample(asiaIncident, startedAt)).toMatchObject({ + status: 'freeze', + failures: [ + expect.objectContaining({ + code: 'threshold_max', + source: 'active-probe', + signal: 'cell.production-gce-c1.latency_ms', + observed: 4_001, + threshold: 4_000 + }) + ] + }) + + const usIncident = healthySample() + usIncident.sources['active-probe']!.signals['cell.production-gce-c1.latency_ms'] = + signal(2_001) + expect(evaluateIncidentSample(usIncident, startedAt)).toMatchObject({ + status: 'freeze', + failures: [ + expect.objectContaining({ + code: 'threshold_max', + signal: 'cell.production-gce-c1.latency_ms', + observed: 2_001, + threshold: 2_000 + }) + ] + }) + }) + it('allows missing auth readiness and legacy existing-only connections', () => { const sample = healthySample() const legacySelector = { @@ -182,12 +230,49 @@ describe('incident monitor evaluator', () => { code: 'source_missing', source: 'relay-logs' }) - const stale = healthySample(startedAt - 180_001) + const stale = healthySample( + startedAt - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) const failures = evaluateIncidentSample(stale, startedAt).failures expect(failures.some((failure) => failure.source === 'cloud-monitoring')).toBe(true) expect(failures.some((failure) => failure.source === 'active-probe')).toBe(true) }) + // Why: production run 33944873727 at 2026-09-05T04:46:09Z read + // cloud_sql.lock_waits 189 286 ms old and restarted a 15-minute window on + // Google's publish lag. Cloud SQL documents 60 s sampling plus up to 165 s of + // invisibility, so that age is Google's clock, not our fleet. + it('reads a 189-second cloud signal as fresh and holds the other sources at 180 s', () => { + const lagged = healthySample() + lagged.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, startedAt - 189_286) + expect(evaluateIncidentSample(lagged, startedAt)).toMatchObject({ + status: 'green', + failures: [] + }) + const laggedDirector = healthySample() + laggedDirector.sources['director-admin']!.observedAt = + new Date(startedAt - 189_286).toISOString() + expect(evaluateIncidentSample(laggedDirector, startedAt).failures).toContainEqual( + expect.objectContaining({ code: 'source_stale', source: 'director-admin' }) + ) + }) + + it('still fails a cloud signal past the documented publish lag', () => { + const dark = healthySample() + dark.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = signal( + 0, + startedAt - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) + expect(evaluateIncidentSample(dark, startedAt).failures).toContainEqual( + expect.objectContaining({ + code: 'signal_stale', + source: 'cloud-monitoring', + signal: 'cloud_sql.lock_waits' + }) + ) + }) + it('freezes on SQL, director, relay pool, heartbeat, and migration breaches', () => { const sample = healthySample() sample.sources['cloud-monitoring']!.signals['cloud_sql.cpu'] = signal(0.81) @@ -363,12 +448,14 @@ describe('incident monitor evaluator', () => { ] = signal(0) sample.cells.push({ cellId: 'production-gce-c2', + region: 'us-central1', runtimeKnown: true, powered: true, expectedAdmissionState: 'general' }) sample.cells.push({ cellId: 'production-gce-c3', + region: 'us-central1', runtimeKnown: true, powered: true, expectedAdmissionState: 'general' @@ -592,7 +679,7 @@ describe('incident monitor lifecycle', () => { 'restarts a %i-minute continuous window after stale telemetry', async (durationMinutes) => { let now = startedAt - let staleInjected = false + let staleSamples = INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1 const checkpoints: Array<[number, number]> = [] const state = initialIncidentMonitorState({ incidentId: 'incident-1', @@ -612,9 +699,11 @@ describe('incident monitor lifecycle', () => { now += ms }, collect: async () => { - if (!staleInjected && now === startedAt + 5 * 60_000) { - staleInjected = true - return healthySample(now - 180_001) + if (staleSamples > 0 && now >= startedAt + 5 * 60_000) { + staleSamples-- + return healthySample( + now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) } return healthySample(now) }, @@ -623,16 +712,20 @@ describe('incident monitor lifecycle', () => { checkpoints.push([summary.windowSequence, summary.checkpointMinute]) } }) + const restartMinute = 5 + INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1 expect(result.windowSequence).toBe(1) expect(result.windowStartedAt).toBe( - new Date(startedAt + 6 * 60_000).toISOString() + new Date(startedAt + restartMinute * 60_000).toISOString() ) expect(result.completedAt).toBe( - new Date(startedAt + (durationMinutes + 6) * 60_000).toISOString() + new Date(startedAt + (durationMinutes + restartMinute) * 60_000).toISOString() ) expect(result.sampleCount).toBe(durationMinutes + 1) - expect(result.continuityEvents).toHaveLength(1) - expect(result.continuityEvents[0]!.failures).toEqual( + expect(result.continuityEvents.map((event) => event.tolerated)).toEqual([ + ...Array(INCIDENT_FRESHNESS_TOLERANCE_SAMPLES).fill(true), + false + ]) + expect(result.continuityEvents.at(-1)!.failures).toEqual( expect.arrayContaining([ expect.objectContaining({ code: 'source_stale' }) ]) @@ -642,6 +735,188 @@ describe('incident monitor lifecycle', () => { } ) + // Why: run 33944873727 on 2026-09-05 restarted at 04:46:09Z on a single + // 189-second cloud reading and then blew the 25-minute lineage cap, so a + // green fleet produced no verdict at all. One unread sample now continues the + // window; the sample is still checked against every threshold it can read. + it('carries a 15-minute window through a single stale cloud sample', async () => { + let now = startedAt + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => { + const sample = healthySample(now) + if (now === startedAt + 10 * 60_000) { + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + } + return sample + }, + persist: async () => {}, + checkpoint: async () => {} + }) + expect(result.windowSequence).toBe(0) + expect(result.windowStartedAt).toBe(new Date(startedAt).toISOString()) + expect(result.completedAt).toBe(new Date(startedAt + 15 * 60_000).toISOString()) + expect(result.sampleCount).toBe(16) + expect(result.frozenAt).toBeNull() + expect(result.continuityEvents).toEqual([{ + recordedAt: new Date(startedAt + 10 * 60_000).toISOString(), + windowSequence: 0, + tolerated: true, + failures: [expect.objectContaining({ + code: 'signal_stale', + source: 'cloud-monitoring', + signal: 'cloud_sql.lock_waits' + })] + }]) + expect(preDrainDryRunPassed(result)).toBe(true) + }) + + it('gives a signal a fresh budget only after it reads fresh again', async () => { + let now = startedAt + const staleMinutes = new Set([3, 5, 6, 9, 10]) + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => { + const sample = healthySample(now) + if (staleMinutes.has((now - startedAt) / 60_000)) { + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + } + return sample + }, + persist: async () => {}, + checkpoint: async () => {} + }) + expect(result.windowSequence).toBe(0) + expect(result.continuityEvents).toHaveLength(staleMinutes.size) + expect(result.continuityEvents.every((event) => event.tolerated)).toBe(true) + expect(preDrainDryRunPassed(result)).toBe(true) + }) + + it('does not hand a resumed monitor a fresh tolerance budget', async () => { + let now = startedAt + 3 * 60_000 + const resumed = { + ...initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }), + windowStartedAt: new Date(startedAt).toISOString(), + lastSampleAt: new Date(startedAt + 2 * 60_000).toISOString(), + sampleCount: 3, + totalSampleCount: 3, + continuityEvents: Array.from( + { length: INCIDENT_FRESHNESS_TOLERANCE_SAMPLES }, + (_, index) => ({ + recordedAt: new Date(startedAt + (index + 1) * 60_000).toISOString(), + windowSequence: 0, + tolerated: true, + failures: [{ + code: 'signal_stale', + source: 'cloud-monitoring' as const, + signal: 'cloud_sql.lock_waits' + }] + }) + ) + } + const stop = new Error('stop after the resumed sample') + await expect(runIncidentMonitor(resumed, { + now: () => now, + wait: async () => { + throw stop + }, + collect: async () => { + const sample = healthySample(now) + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + return sample + }, + persist: async (state) => { + expect(state.windowSequence).toBe(1) + expect(state.windowStartedAt).toBeNull() + expect(state.continuityEvents.at(-1)!.tolerated).toBe(false) + }, + checkpoint: async () => {} + })).rejects.toThrow(stop) + }) + + it('freezes on a threshold breach that arrives with a tolerated stale signal', async () => { + let now = startedAt + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => { + const sample = healthySample(now) + if (now === startedAt + 2 * 60_000) { + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + sample.sources['cloud-monitoring']!.signals['cloud_sql.cpu'] = signal(0.81, now) + } + return sample + }, + persist: async () => {}, + checkpoint: async () => {} + }) + expect(result.frozenAt).toBe(new Date(startedAt + 2 * 60_000).toISOString()) + expect(result.failures).toContainEqual(expect.objectContaining({ + code: 'threshold_max', + signal: 'cloud_sql.cpu' + })) + expect(preDrainDryRunPassed(result)).toBe(false) + }) + it('resets at the next fresh sample after a runner gap', async () => { let now = startedAt + 10 * 60_000 const state = { @@ -690,13 +965,21 @@ describe('incident monitor lifecycle', () => { durationMinutes: 15, intervalMs: 60_000 }) + let staleSamples = INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1 const result = await runIncidentMonitor(state, { now: () => now, wait: async (ms) => { now += ms }, - collect: async () => - healthySample(now === startedAt + 10 * 60_000 ? now - 180_001 : now), + collect: async () => { + if (staleSamples > 0 && now >= startedAt + 10 * 60_000) { + staleSamples-- + return healthySample( + now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) + } + return healthySample(now) + }, persist: async () => {}, checkpoint: async () => {} }) @@ -706,7 +989,7 @@ describe('incident monitor lifecycle', () => { ) expect(result.frozenAt).not.toBeNull() expect(result.windowSequence).toBe(1) - expect(result.sampleCount).toBe(15) + expect(result.sampleCount).toBe(13) expect(result.failures).toContainEqual({ code: 'continuity_deadline_exceeded', source: 'active-probe', diff --git a/cloud/apps/relay-ops/src/incident-monitor.ts b/cloud/apps/relay-ops/src/incident-monitor.ts index a121568d918..0887bb2d1ee 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.ts @@ -1,3 +1,4 @@ +import type { RelayOpsRegion } from './environment-config.js' import { exactAdmissionSelector, type AdmissionSelector, @@ -6,10 +7,39 @@ import { export const INCIDENT_MONITOR_THRESHOLDS = { activeProbeMaxAgeMs: 60_000, - cloudDataMaxAgeMs: 180_000, + // Why: Cloud Monitoring publishes on Google's clock, not ours. Per the metric + // list read 2026-09-05, Cloud Run instance_count / cpu / memory / + // max_request_concurrencies / request_count are "Sampled every 60 seconds. + // After sampling, data is not visible for up to 120 seconds" (60+120=180 s), + // and Cloud SQL cpu / memory / num_backends / backends_in_wait / + // deadlock_count say "up to 165 seconds" (60+165=225 s). Window-sum signals + // age differently: observedAt is the newest point in the 5-minute query + // window, so a label series that stops emitting reads as 300 s old while its + // summed value is still complete. 330 s clears the worst of the three (the + // 300 s query window) plus ~30 s of collect-to-evaluate latency. The old + // 180 s bar restarted healthy 15-minute windows at 181 s, 189 s and 255 s on + // 2026-09-04/05, once burning the whole 25-minute lineage with no verdict. + cloudDataMaxAgeMs: 330_000, + // Why: the director admin API answers live on our own request, so hold its + // freshness bar where it sat while it shared cloudDataMaxAgeMs. + directorAdminMaxAgeMs: 180_000, + // Why: how long a nonzero backends-in-wait point is carried before it reads as + // zero. Held at the pre-2026-09-05 cloud bar: carrying it for the full + // cloudDataMaxAgeMs would hand the evaluator a point older than its own + // freshness bar as soon as collection latency is added. + cloudLockWaitCarryMs: 180_000, relayLogMaxAgeMs: 180_000, heartbeatMaxAgeMs: 45_000, endpointLatencyMs: 2_000, + // Why: a cell's /ready fetches the auth JWKS and runs SELECT 1 against Cloud SQL, + // both in us-central1, so from the US runner asia-east2 cells measure p50 0.88 s / + // max 2.7 s against 0.08-0.5 s for us-central1. The flat 2 000 bar froze three + // healthy 15-minute gates on 2026-09-05 (c27 at 2568/2668/2685 ms); hard faults + // are still caught by the .health/.ready equal-1 checks and the 8 s fetch timeout. + cellEndpointLatencyMs: { + 'us-central1': 2_000, + 'asia-east2': 4_000 + } as const satisfies Record, cloudSqlCpuUtilization: 0.8, cloudSqlMemoryUtilization: 0.9, // Why: healthy latest-sum backends idle near 100 but spike to 216 in 1-minute @@ -106,6 +136,7 @@ export type IncidentSource = { export type IncidentCellExpectation = { cellId: string + region: RelayOpsRegion runtimeKnown: boolean powered: boolean expectedAdmissionState: AdmissionState @@ -175,6 +206,7 @@ export type IncidentMonitorState = { continuityEvents: { recordedAt: string windowSequence: number + tolerated: boolean failures: IncidentFailure[] }[] frozenAt: string | null @@ -307,7 +339,7 @@ const SOURCE_MAX_AGE: Record = { 'active-probe': INCIDENT_MONITOR_THRESHOLDS.activeProbeMaxAgeMs, 'cloud-monitoring': INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs, 'relay-logs': INCIDENT_MONITOR_THRESHOLDS.relayLogMaxAgeMs, - 'director-admin': INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 'director-admin': INCIDENT_MONITOR_THRESHOLDS.directorAdminMaxAgeMs } function ageMs(timestamp: string, nowMs: number): number { @@ -384,7 +416,7 @@ function checkCell( 'active-probe', probe, `cell.${cell.cellId}.latency_ms`, - INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs, + INCIDENT_MONITOR_THRESHOLDS.cellEndpointLatencyMs[cell.region], 'max' ], [ @@ -608,14 +640,59 @@ function checkpointMinutes(durationMinutes: number): number[] { return INCIDENT_CHECKPOINT_MINUTES.filter((minute) => minute <= durationMinutes) } -const CONTINUITY_FAILURE_CODES = new Set([ - 'collector_failed', - 'monitor_gap', +// Freshness-only failures: we could not read a signal this sample. Distinct from +// collector_failed / monitor_gap, where the whole sample is absent. +export const FRESHNESS_FAILURE_CODES = new Set([ + 'signal_missing', 'signal_stale', 'source_missing', 'source_stale' ]) +const CONTINUITY_FAILURE_CODES = new Set([ + 'collector_failed', + 'monitor_gap', + ...FRESHNESS_FAILURE_CODES +]) + +// Why: Cloud Monitoring overshoots its own publish bar, and one unread sample is +// not evidence of an unhealthy fleet. Under the 25-minute lineage cap a restart +// past minute 10 costs the entire verdict, so a healthy fleet produced none on +// 2026-09-05. A signal may miss this many consecutive samples before the window +// restarts; the sample is still evaluated against every threshold it can read, +// and a threshold breach still freezes the run outright. +export const INCIDENT_FRESHNESS_TOLERANCE_SAMPLES = 2 + +function freshnessKey(failure: IncidentFailure): string { + return `${failure.source}/${failure.signal ?? '*'}` +} + +// Rebuild the per-signal tolerated streak from the trailing continuity events so a +// resumed monitor cannot hand a signal a fresh budget. +function resumeFreshnessStreaks( + state: IncidentMonitorState +): Map { + const events = state.continuityEvents + const streaks = new Map() + const last = events[events.length - 1] + if (!last?.tolerated) return streaks + for (const key of new Set(last.failures.map(freshnessKey))) { + let streak = 0 + let laterAt: number | null = null + for (let index = events.length - 1; index >= 0; index--) { + const event = events[index]! + const recordedAt = Date.parse(event.recordedAt) + if (!event.tolerated) break + if (laterAt !== null && laterAt - recordedAt > state.intervalMs * 1.5) break + if (!event.failures.some((failure) => freshnessKey(failure) === key)) break + streak++ + laterAt = recordedAt + } + streaks.set(key, streak) + } + return streaks +} + function resetContinuousWindow( state: IncidentMonitorState, recordedAt: string, @@ -631,6 +708,7 @@ function resetContinuousWindow( state.continuityEvents.push({ recordedAt, windowSequence: state.windowSequence, + tolerated: false, failures }) } @@ -681,6 +759,7 @@ export async function runIncidentMonitor( await dependencies.persist(state) return state } + const freshnessStreaks = resumeFreshnessStreaks(state) while (state.completedAt === null) { if (dependencies.now() > lineageDeadlineMs) { completeContinuityDeadline(state, dependencies.now(), lineageStartMs) @@ -715,9 +794,34 @@ export async function runIncidentMonitor( const thresholdFailures = evaluation.failures.filter((failure) => !CONTINUITY_FAILURE_CODES.has(failure.code) ) - if (continuityFailures.length > 0) { + const toleratedKeys = new Set( + state.windowStartedAt !== null && + continuityFailures.length > 0 && + continuityFailures.every((failure) => FRESHNESS_FAILURE_CODES.has(failure.code)) + ? continuityFailures.map(freshnessKey) + : [] + ) + for (const key of [...freshnessStreaks.keys()]) { + if (!toleratedKeys.has(key)) freshnessStreaks.delete(key) + } + let tolerated = toleratedKeys.size > 0 + for (const key of toleratedKeys) { + const streak = (freshnessStreaks.get(key) ?? 0) + 1 + freshnessStreaks.set(key, streak) + if (streak > INCIDENT_FRESHNESS_TOLERANCE_SAMPLES) tolerated = false + } + if (continuityFailures.length > 0 && !tolerated) { + freshnessStreaks.clear() resetContinuousWindow(state, evaluation.evaluatedAt, continuityFailures) } else { + if (tolerated) { + state.continuityEvents.push({ + recordedAt: evaluation.evaluatedAt, + windowSequence: state.windowSequence, + tolerated: true, + failures: continuityFailures + }) + } if (state.windowStartedAt === null) { state.windowStartedAt = evaluation.evaluatedAt } diff --git a/cloud/apps/relay-ops/src/resource-inventory.test.ts b/cloud/apps/relay-ops/src/resource-inventory.test.ts index 6d8b3070c98..e2cfa13dccb 100644 --- a/cloud/apps/relay-ops/src/resource-inventory.test.ts +++ b/cloud/apps/relay-ops/src/resource-inventory.test.ts @@ -13,6 +13,41 @@ const runService = { latestReadyRevision: 'projects/project/revisions/revision-one' } +const sleepingStagingGcloud: GcloudClient = { accessToken: async () => 'a'.repeat(40) } + +// Staging's Cloud SQL is stopped, so this inventory reads REST only and probes no endpoint. +type MigOutcome = 'ok' | 'throw' | 'missing' +const sleepingStagingFetch = (migOutcome: (migName: string) => MigOutcome): typeof fetch => + async (input) => { + const url = new URL(String(input)) + if (url.hostname === 'run.googleapis.com') return Response.json(runService) + if (url.hostname === 'sqladmin.googleapis.com') return Response.json({ + state: 'STOPPED', + databaseVersion: 'POSTGRES_17', + settings: { activationPolicy: 'NEVER', availabilityType: 'ZONAL', tier: 'db-custom-1-3840' } + }) + if (url.hostname === 'certificatemanager.googleapis.com') return Response.json({ + managed: { domains: ['*.relay-staging.onorca.dev'], state: 'ACTIVE' } + }) + if (url.pathname.includes('/instanceGroupManagers/')) { + const name = url.pathname.split('/').at(-1)! + const outcome = migOutcome(name) + if (outcome === 'throw') throw new TypeError('fetch failed') + if (outcome === 'missing') return new Response(null, { status: 404 }) + return Response.json({ + name, + targetSize: 0, + size: '0', + instanceGroup: `projects/project/zones/zone/instanceGroups/${name}`, + instanceTemplate: `projects/project/global/instanceTemplates/template-${name}`, + status: { isStable: true } + }) + } + if (url.pathname.includes('/instanceTemplates/')) return Response.json({ properties: {} }) + if (url.pathname.endsWith('/getHealth')) return Response.json([]) + throw new Error(`Unexpected request to ${url.hostname}${url.pathname}`) + } + describe('readResourceInventory', () => { it('does not delay a healthy endpoint sample', async () => { let calls = 0 @@ -23,8 +58,10 @@ describe('readResourceInventory', () => { calls += 1 return new Response(null, { status: 200 }) }, - async () => { - waits += 1 + { + wait: async () => { + waits += 1 + } } ) @@ -45,8 +82,10 @@ describe('readResourceInventory', () => { calls.set(path, call) return new Response(null, { status: path === '/ready' && call === 1 ? 503 : 200 }) }, - async (ms) => { - waits.push(ms) + { + wait: async (ms) => { + waits.push(ms) + } } ) @@ -65,17 +104,130 @@ describe('readResourceInventory', () => { calls += 1 return new Response(null, { status: 503 }) }, - async (ms) => { - waits.push(ms) + { + wait: async (ms) => { + waits.push(ms) + } } ) expect(result.health).toBe(false) expect(result.ready).toBe(false) expect(calls).toBe(4) + // A refusing endpoint is a reading, so only the independent retry runs. expect(waits).toEqual([11_000]) }) + it('treats a thrown fetch as no reading and re-asks that path once', async () => { + const calls: string[] = [] + const waits: number[] = [] + const result = await probeEndpointHealth( + 'https://c9.relay.onorca.dev', + async (input) => { + const path = new URL(String(input)).pathname + calls.push(path) + if (path === '/health' && calls.filter((call) => call === '/health').length === 1) { + throw new TypeError('fetch failed') + } + return new Response(null, { status: 200 }) + }, + { + wait: async (ms) => { + waits.push(ms) + } + } + ) + + expect(result.health).toBe(true) + expect(result.ready).toBe(true) + expect(calls.filter((call) => call === '/health')).toEqual(['/health', '/health']) + expect(waits).toEqual([1_000]) + }) + + it('fails closed when both attempts of a path throw', async () => { + const calls: string[] = [] + const waits: number[] = [] + const result = await probeEndpointHealth( + 'https://c9.relay.onorca.dev', + async (input) => { + const path = new URL(String(input)).pathname + calls.push(path) + if (path === '/health') throw new TypeError('fetch failed') + return new Response(null, { status: 200 }) + }, + { + wait: async (ms) => { + waits.push(ms) + } + } + ) + + expect(result.health).toBe(false) + expect(calls.filter((call) => call === '/health')).toHaveLength(4) + expect(waits).toEqual([1_000, 11_000, 1_000]) + }) + + it('accepts an auth-shaped endpoint that serves no readiness path', async () => { + const calls: string[] = [] + const waits: number[] = [] + const result = await probeEndpointHealth( + 'https://login.onorca.dev', + async (input) => { + const path = new URL(String(input)).pathname + calls.push(path) + return new Response(null, { status: path === '/ready' ? 404 : 200 }) + }, + { + requiresReady: false, + wait: async (ms) => { + waits.push(ms) + } + } + ) + + expect(result.health).toBe(true) + expect(result.ready).toBeNull() + expect(calls).toEqual(['/health']) + expect(waits).toEqual([]) + }) + + it('still requires readiness for the director and cells', async () => { + const waits: number[] = [] + const result = await probeEndpointHealth( + 'https://relay.onorca.dev', + async (input) => new Response(null, { + status: new URL(String(input)).pathname === '/ready' ? 503 : 200 + }), + { + wait: async (ms) => { + waits.push(ms) + } + } + ) + + expect(result.health).toBe(true) + expect(result.ready).toBe(false) + expect(waits).toEqual([11_000]) + }) + + it('measures latency as the answering round trip, not the retry delay', async () => { + let healthCalls = 0 + const result = await probeEndpointHealth( + 'https://c9.relay.onorca.dev', + async (input) => { + if (new URL(String(input)).pathname !== '/health') return new Response(null, { status: 200 }) + healthCalls += 1 + if (healthCalls === 1) throw new TypeError('fetch failed') + return new Response(null, { status: 200 }) + }, + { wait: async (ms) => await new Promise((resolve) => setTimeout(resolve, Math.min(ms, 60))) } + ) + + expect(result.health).toBe(true) + expect(result.latencyMs).not.toBeNull() + expect(result.latencyMs!).toBeLessThan(60) + }) + it('uses aggregate REST inventory without probing sleeping staging endpoints', async () => { const gcloud: GcloudClient = { accessToken: async () => 'a'.repeat(40) } let publicProbeCalls = 0 @@ -132,6 +284,74 @@ describe('readResourceInventory', () => { expect(JSON.stringify(result)).not.toContain('SECRET_TEXT') }) + it('re-asks a MIG read that failed once before calling a cell powered-unknown', async () => { + const parkedCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]! + const waits: number[] = [] + let parkedMigCalls = 0 + const result = await readResourceInventory( + RELAY_OPS_ENVIRONMENTS.staging, + sleepingStagingGcloud, + sleepingStagingFetch((migName) => { + if (!migName.endsWith(parkedCell.hostname)) return 'ok' + parkedMigCalls += 1 + return parkedMigCalls === 1 ? 'throw' : 'ok' + }), + { wait: async (ms) => { waits.push(ms) } } + ) + + const parked = result.cells.find((cell) => cell.cellId === parkedCell.cellId)! + // The MIG was fine and parked at zero; one transient read must not erase that reading. + expect(parked.targetSize).toBe(0) + expect(parkedMigCalls).toBe(2) + expect(waits).toEqual([1_000]) + expect(result.warnings).toEqual([]) + }) + + it('reports a MIG unavailable only when the retry fails too', async () => { + const parkedCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]! + const waits: number[] = [] + let parkedMigCalls = 0 + const result = await readResourceInventory( + RELAY_OPS_ENVIRONMENTS.staging, + sleepingStagingGcloud, + sleepingStagingFetch((migName) => { + if (!migName.endsWith(parkedCell.hostname)) return 'ok' + parkedMigCalls += 1 + return 'throw' + }), + { wait: async (ms) => { waits.push(ms) } } + ) + + const parked = result.cells.find((cell) => cell.cellId === parkedCell.cellId)! + expect(parked.targetSize).toBeNull() + expect(parked.backendHealth).toBe('unknown') + expect(parkedMigCalls).toBe(2) + expect(waits).toEqual([1_000]) + expect(result.warnings).toEqual([ + `${parkedCell.hostname.toUpperCase()} MIG inventory is unavailable.` + ]) + }) + + it('does not re-ask a MIG read the API answered with 404', async () => { + const missingCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]! + const waits: number[] = [] + let missingMigCalls = 0 + const result = await readResourceInventory( + RELAY_OPS_ENVIRONMENTS.staging, + sleepingStagingGcloud, + sleepingStagingFetch((migName) => { + if (!migName.endsWith(missingCell.hostname)) return 'ok' + missingMigCalls += 1 + return 'missing' + }), + { wait: async (ms) => { waits.push(ms) } } + ) + + expect(result.cells.find((cell) => cell.cellId === missingCell.cellId)!.targetSize).toBeNull() + expect(missingMigCalls).toBe(1) + expect(waits).toEqual([]) + }) + it('represents missing credentials as unknown inventory, never sleeping', async () => { const gcloud: GcloudClient = { accessToken: async () => { throw new Error('sensitive context') } diff --git a/cloud/apps/relay-ops/src/resource-inventory.ts b/cloud/apps/relay-ops/src/resource-inventory.ts index 62ed3fd862b..da490685198 100644 --- a/cloud/apps/relay-ops/src/resource-inventory.ts +++ b/cloud/apps/relay-ops/src/resource-inventory.ts @@ -102,6 +102,9 @@ export type ResourceInventory = { const unavailableEndpoint = (): EndpointHealth => ({ health: null, ready: null, latencyMs: null }) const independentEndpointRetryDelayMs = 11_000 +const transientProbeRetryDelayMs = 1_000 +const sleep = async (ms: number): Promise => + await new Promise((resolvePromise) => setTimeout(resolvePromise, ms)) function finalSegment(value: string): string { return value.split('/').at(-1) ?? value @@ -120,6 +123,12 @@ function parseService(value: unknown): ServiceInventory { } } +class GoogleApiError extends Error { + constructor(readonly status: number) { + super(`Google API returned ${status}`) + } +} + async function googleRequest( fetchImpl: typeof fetch, token: string, @@ -134,45 +143,97 @@ async function googleRequest( }, signal: AbortSignal.timeout(30_000) }) - if (!response.ok) throw new Error(`Google API returned ${response.status}`) + if (!response.ok) throw new GoogleApiError(response.status) return await response.json() } -async function endpointProbe(origin: string, fetchImpl: typeof fetch): Promise { - const startedAt = performance.now() - const check = async (path: '/health' | '/ready'): Promise => { +// A 404 is the API's answer about the resource; anything else is the absence of a reading, so re-ask. +async function readOnceMore( + read: () => Promise, + wait: (ms: number) => Promise +): Promise { + try { + return await read() + } catch (error) { + if (error instanceof GoogleApiError && error.status === 404) throw error + await wait(transientProbeRetryDelayMs) + return await read() + } +} + +// A reading the endpoint actually produced: ok is its answer, latencyMs is that answer's round trip. +type PathReading = { ok: boolean; latencyMs: number | null } + +async function probePath( + origin: string, + path: '/health' | '/ready', + fetchImpl: typeof fetch, + wait: (ms: number) => Promise +): Promise { + // null means the request never produced an answer (DNS/TCP/TLS failure or the 8s abort). + const attempt = async (): Promise => { + const startedAt = performance.now() try { const response = await fetchImpl(`${origin}${path}`, { redirect: 'error', signal: AbortSignal.timeout(8_000) }) - return response.ok + return { ok: response.ok, latencyMs: Math.round(performance.now() - startedAt) } } catch { - return false + return null } } - const [health, ready] = await Promise.all([check('/health'), check('/ready')]) - return { health, ready, latencyMs: Math.round(performance.now() - startedAt) } + const first = await attempt() + if (first) return first + // A thrown fetch is the absence of a reading, not an unhealthy answer, so re-ask before concluding. + await wait(transientProbeRetryDelayMs) + return (await attempt()) ?? { ok: false, latencyMs: null } +} + +async function endpointProbe( + origin: string, + fetchImpl: typeof fetch, + requiresReady: boolean, + wait: (ms: number) => Promise +): Promise { + const [health, ready] = await Promise.all([ + probePath(origin, '/health', fetchImpl, wait), + requiresReady ? probePath(origin, '/ready', fetchImpl, wait) : null + ]) + // Latency is the slowest answering round trip in this probe; retry delays are not serving latency. + const latencies = [health.latencyMs, ready?.latencyMs ?? null].filter( + (value): value is number => value !== null + ) + return { + health: health.ok, + ready: ready ? ready.ok : null, + latencyMs: latencies.length > 0 ? Math.max(...latencies) : null + } +} + +export type EndpointProbeOptions = { + // Auth serves no /ready by design, so it is judged on /health and latency alone. + requiresReady?: boolean + wait?: (ms: number) => Promise } export async function probeEndpointHealth( origin: string, fetchImpl: typeof fetch, - wait: (ms: number) => Promise = async (ms) => - await new Promise((resolvePromise) => setTimeout(resolvePromise, ms)) + options: EndpointProbeOptions = {} ): Promise { - const first = await endpointProbe(origin, fetchImpl) - if ( - first.health && - first.ready && - first.latencyMs !== null && - first.latencyMs <= INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs - ) { - return first - } + const requiresReady = options.requiresReady ?? true + const wait = options.wait ?? sleep + const accepted = (probe: EndpointHealth): boolean => + probe.health === true && + (!requiresReady || probe.ready === true) && + probe.latencyMs !== null && + probe.latencyMs <= INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs + const first = await endpointProbe(origin, fetchImpl, requiresReady, wait) + if (accepted(first)) return first // Outwait Relay's ten-second readiness cache before treating the retry as independent. await wait(independentEndpointRetryDelayMs) - return await endpointProbe(origin, fetchImpl) + return await endpointProbe(origin, fetchImpl, requiresReady, wait) } function imageDigest(template: z.infer): string | null { @@ -285,11 +346,17 @@ function unavailableInventory(environment: RelayOpsEnvironment, warning: string) } } +export type ResourceInventoryOptions = { + wait?: (ms: number) => Promise +} + export async function readResourceInventory( environment: RelayOpsEnvironment, gcloud: GcloudClient, - fetchImpl: typeof fetch = fetch + fetchImpl: typeof fetch = fetch, + options: ResourceInventoryOptions = {} ): Promise { + const wait = options.wait ?? sleep let token: string try { token = await gcloud.accessToken() @@ -316,7 +383,10 @@ export async function readResourceInventory( token, `https://certificatemanager.googleapis.com/v1/projects/${environment.project}/locations/global/certificates/${environment.certificateName}` ), - ...environment.cells.map((cell) => googleRequest(fetchImpl, token, migUrl(cell))) + // One transient Compute read must never become a verdict on a cell's power state. + ...environment.cells.map((cell) => + readOnceMore(async () => await googleRequest(fetchImpl, token, migUrl(cell)), wait) + ) ]) const warnings: string[] = [] const directorValue = parsed(settled[0]!, RunServiceSchema, 'Director service inventory is unavailable.', warnings) @@ -338,7 +408,8 @@ export async function readResourceInventory( ? [unavailableEndpoint(), unavailableEndpoint()] : await Promise.all([ probeEndpointHealth(environment.directorOrigin, fetchImpl), - probeEndpointHealth(environment.authOrigin, fetchImpl) + // The auth service exposes no /ready, so requiring it would fail every first probe. + probeEndpointHealth(environment.authOrigin, fetchImpl, { requiresReady: false }) ]) const cells = await Promise.all(environment.cells.map((cell, index) => readCell(environment, cell, migValues[index] ?? null, token, fetchImpl) diff --git a/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts b/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts index 6ac9521c3d6..80a74a47eeb 100644 --- a/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts +++ b/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts @@ -44,6 +44,12 @@ describePostgres('PostgreSQL assignment connection headroom', () => { `DELETE FROM relay_assignments WHERE user_id LIKE 'connection-headroom-postgres-%'` ) + // A snapshot left by an aborted run rejects the replayed watermark + // with stale_connection_snapshot. + await database.query( + `DELETE FROM relay_cell_connection_snapshots WHERE cell_id = ?`, + [cell.id] + ) await database.query( `DELETE FROM relay_cell_connection_runtime WHERE cell_id = ?`, [cell.id] diff --git a/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts b/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts index 10193b78cc6..cf8819686b5 100644 --- a/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts +++ b/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts @@ -38,6 +38,10 @@ describePostgres('PostgreSQL control supersession', () => { [identity.userId] ) await database.query(`DELETE FROM relay_assignments WHERE user_id = ?`, [identity.userId]) + // A snapshot left by an aborted run rejects the replayed watermark with stale_connection_snapshot. + await database.query(`DELETE FROM relay_cell_connection_snapshots WHERE cell_id = ?`, [ + cell.id + ]) await database.query(`DELETE FROM relay_cell_connection_runtime WHERE cell_id = ?`, [cell.id]) await database.query(`DELETE FROM relay_cell_connection_limits WHERE cell_id = ?`, [cell.id]) await database.query(`DELETE FROM relay_cell_runtime WHERE cell_id = ?`, [cell.id]) diff --git a/cloud/apps/relay/src/assignment-store.ts b/cloud/apps/relay/src/assignment-store.ts index d0517d46746..226df9b3984 100644 --- a/cloud/apps/relay/src/assignment-store.ts +++ b/cloud/apps/relay/src/assignment-store.ts @@ -645,9 +645,17 @@ export class RelayAssignmentStore { ): Promise { const now = this.now() return await this.database.transaction(async (transaction) => { - const lockedCells = inventoryFirst - ? await this.lockCellInventory(transaction, lockMode) + // Why: the retry exists to take a cell row before the assignment row, the + // order placement uses. It only ever needs the one cell this host is + // pinned to, so read the pin unlocked and lock that row alone; taking all + // 23 queued every sticky refresh in the fleet behind every other one. + const pinnedCellId = inventoryFirst + ? await this.pinnedCellId(transaction, identity) : undefined + const lockedCells = + pinnedCellId === undefined + ? undefined + : await this.lockCellRows(transaction, [pinnedCellId], lockMode) const existing = await this.assignmentRow(transaction, identity, inventoryFirst) if (!existing) return null const activityLeases = await this.lockAssignmentActivities(transaction, identity, true) @@ -661,6 +669,11 @@ export class RelayAssignmentStore { } const currentCellId = text(existing, 'cell_id') + // The pin moved between the unlocked read and the assignment lock, so the + // row held is the wrong one. Same recovery as losing the lock: retry. + if (pinnedCellId !== undefined && pinnedCellId !== currentCellId) { + throw new Error('database_lock_unavailable') + } const hadControl = holdsControlLease( activityLeases, currentCellId, @@ -701,14 +714,9 @@ export class RelayAssignmentStore { if (hadControl) { await this.touchAssignment(transaction, identity, leaseExpiresAt, now) } else { - const nextReservation = integer(currentRow, 'reserved_requests') + 1 - if (nextReservation > integer(currentRow, 'capacity_requests')) { - throw new Error('relay_capacity_exhausted') - } - await transaction.query( - `UPDATE relay_cells SET reserved_requests = ?, updated_at = ? WHERE cell_id = ?`, - [nextReservation, now, currentCellId] - ) + // Delta, not the value read from the snapshot: an absolute write here + // would clobber any concurrent movement of the same counter. + await this.adjustCellReservationAtomically(transaction, currentCellId, 1) await this.adjustActivityCount(transaction, identity, 'control', 1, leaseExpiresAt, now) await this.insertPendingControlLease( transaction, @@ -3202,8 +3210,7 @@ export class RelayAssignmentStore { ) const requestDelta = ACTIVITY_REQUEST_UNITS[kind] * (after - before) if (requestDelta !== 0) { - await this.lockCellInventory(transaction, 'request') - await this.adjustCellReservation(transaction, text(row, 'cell_id'), requestDelta) + await this.adjustCellReservationAtomically(transaction, text(row, 'cell_id'), requestDelta) } }) }) @@ -3263,9 +3270,12 @@ export class RelayAssignmentStore { } const units = ACTIVITY_REQUEST_UNITS[input.kind] if (existing) { - await this.lockCellInventory(transaction, 'request') + // Why: a client-chosen activity id can move between cells, so lock the + // one or two rows this path touches in cell_id order, the same order + // placement takes the inventory in, and no cycle can form. + await this.lockCellRows(transaction, [text(existing, 'cell_id'), input.cellId]) await this.removeActivityLease(transaction, identity, existing, now) - await this.adjustCellReservation(transaction, input.cellId, units) + await this.adjustCellReservationAtomically(transaction, input.cellId, units) } await this.adjustActivityCount(transaction, identity, input.kind, 1, expiresAt, now) await transaction.query( @@ -3580,8 +3590,7 @@ export class RelayAssignmentStore { ) await this.touchAssignment(transaction, identity, expiresAt, now) } else { - await this.lockCellInventory(transaction, 'request') - await this.adjustCellReservation(transaction, input.cellId, 1) + await this.adjustCellReservationAtomically(transaction, input.cellId, 1) await this.adjustActivityCount(transaction, identity, 'control', 1, expiresAt, now) await transaction.query( `INSERT INTO relay_assignment_activity_leases @@ -6863,13 +6872,24 @@ export class RelayAssignmentStore { targetCellId ] ) - const cells = await this.lockCellInventory(transaction, 'pool-default') + // Only the two cells this repairs need holding. The id set below is an + // existence check against a table that only reconcileCells writes, so it + // reads unlocked instead of dragging the other 21 rows into the section. + const cellIds = new Set( + (await transaction.query(`SELECT cell_id FROM relay_cells`)).map((row) => + text(row, 'cell_id') + ) + ) + const cells = await this.lockCellRows( + transaction, + [sourceCellId, targetCellId], + 'pool-default' + ) const assignmentKeys = new Set( assignments.map((row) => assignmentKey(text(row, 'user_id'), text(row, 'relay_host_id')) ) ) - const cellIds = new Set(cells.map((row) => text(row, 'cell_id'))) const assignmentCounts = new Map< string, { counts: Record; leaseExpiresAt: number } @@ -6923,9 +6943,7 @@ export class RelayAssignmentStore { ) } - for (const row of cells.filter((cell) => - [sourceCellId, targetCellId].includes(text(cell, 'cell_id')) - )) { + for (const row of cells) { const cellId = text(row, 'cell_id') const expected = cellUnits.get(cellId) ?? 0 if (expected > integer(row, 'capacity_requests')) { @@ -6954,6 +6972,43 @@ export class RelayAssignmentStore { return rows } + // Per-connection paths touch one or two cells. Locking exactly those rows, + // in the same ascending order the inventory lock uses (ORDER BY fixes the + // row-lock order), keeps them off the fleet-wide lock without a cycle. + // The wait policy follows the caller for the same reason the inventory lock's + // does: a sweep must not fail terminally on ordinary contention. Hold time is + // deliberately not sampled here — the metric tracks the fleet-wide lock these + // rows replace, and mixing in short single-row holds would flatter it. + private async lockCellRows( + database: RelayDatabase, + cellIds: string[], + mode: CellInventoryLockMode = 'request' + ): Promise { + const distinct = [...new Set(cellIds)] + const { measureHoldMs: _sampled, ...wait } = cellInventoryLockOptions(mode) + return await database.queryLocked( + `SELECT * FROM relay_cells WHERE cell_id IN (${distinct.map(() => '?').join(', ')}) + ORDER BY cell_id ASC`, + distinct, + wait + ) + } + + // Unlocked on purpose: this only names the row to lock next, and the caller + // re-checks the pin once the assignment row is held. + private async pinnedCellId( + database: RelayDatabase, + identity: AssignmentIdentity + ): Promise { + const row = ( + await database.query( + `SELECT cell_id FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + )[0] + return row ? text(row, 'cell_id') : undefined + } + private async lockGeneralCellInventory( database: RelayDatabase, mode: CellInventoryLockMode @@ -6972,10 +7027,11 @@ export class RelayAssignmentStore { private async leastLoadedCell( database: RelayDatabase, - lockedCells: SqlRow[] | undefined, + // Required: the one caller has already locked the inventory it selects from, + // and an optional parameter left a second fleet-wide lock reachable here. + rows: SqlRow[], preferredRegion: RelayRegion ): Promise { - const rows = lockedCells ?? (await this.lockCellInventory(database, 'pool-default')) const regions = new Map( (await database.query(`SELECT cell_id, region FROM relay_cell_regions`)).map((row) => [ text(row, 'cell_id'), @@ -7590,7 +7646,10 @@ export class RelayAssignmentStore { ) { throw new Error('activity_lease_shape_mismatch') } - const cells = await this.lockCellInventory(database, 'request') + // Why: this recomputes one cell's reservation from its leases, so only that + // row needs to be held; the 23-row inventory lock here serialised every + // desktop control rebind in the fleet behind every other one. + const cellRow = (await this.lockCellRows(database, [cellId]))[0] await database.query( `DELETE FROM relay_assignment_activity_leases WHERE user_id = ? AND relay_host_id = ? AND activity_kind = 'control' @@ -7611,7 +7670,6 @@ export class RelayAssignmentStore { [cellId] ) )[0]! - const cellRow = cells.find((cell) => text(cell, 'cell_id') === cellId) const cellUnits = integer(cellUnitsRow, 'request_units') if (!cellRow) throw new Error('assigned_cell_missing') if (cellUnits > integer(cellRow, 'capacity_requests')) { diff --git a/cloud/apps/relay/src/cell-inventory-lock-census.test.ts b/cloud/apps/relay/src/cell-inventory-lock-census.test.ts index 8ca7cee55f5..0ac4c8225e3 100644 --- a/cloud/apps/relay/src/cell-inventory-lock-census.test.ts +++ b/cloud/apps/relay/src/cell-inventory-lock-census.test.ts @@ -18,17 +18,21 @@ type CensusEntry = { method: string; mode: CensusMode; reach: Reachability } // assignment-store.ts, in source order. A new site fails this test until it is // classified here, which is the point. const CENSUS: CensusEntry[] = [ - { method: 'assignStickyOnce', mode: 'caller', reach: 'both' }, + // assignStickyOnce is gone from this list: its retry now locks only the row + // the host is pinned to (lockCellRows), which is what a sticky refresh + // touches. Placement below is the one genuinely fleet-wide decision left. { method: 'assignOnce', mode: 'caller', reach: 'both' }, { method: 'assignOnce', mode: 'caller', reach: 'both' }, { method: 'assignOnce', mode: 'nowait', reach: 'both' }, { method: 'assignOnce', mode: 'nowait', reach: 'both' }, { method: 'assignOnce', mode: 'nowait', reach: 'both' }, { method: 'refreshDrainMigrationLeasesOnce', mode: 'request', reach: 'request' }, - // Reachable from neither: changeActivity has no production callers, only tests. - { method: 'changeActivity', mode: 'request', reach: 'orphan' }, - { method: 'acquireActivity', mode: 'request', reach: 'request' }, - { method: 'activateControl', mode: 'request', reach: 'request' }, + // changeActivity, acquireActivity, activateControl and + // removeSupersededSameCellControls no longer take the inventory: they lock + // only the one or two cell rows they touch, in cell_id order (lockCellRows), + // so they cannot cycle with placement's ordered inventory lock, and the + // 23-row lock there had serialised every reconnect in the fleet behind every + // other one. { method: 'startEvacuation', mode: 'request', reach: 'request' }, { method: 'completeEvacuationFromDeadSourceOnce', mode: 'request', reach: 'request' }, { method: 'completeEvacuationFromDeadSourceOnce', mode: 'nowait', reach: 'request' }, @@ -47,9 +51,33 @@ const CENSUS: CensusEntry[] = [ { method: 'abortExpiredEvacuations', mode: 'nowait', reach: 'sweep' }, { method: 'releaseExpiredActivityLeases', mode: 'nowait', reach: 'sweep' }, { method: 'releaseExpiredActivity', mode: 'nowait', reach: 'sweep' }, - { method: 'reconcileReservationAccounting', mode: 'pool-default', reach: 'both' }, - { method: 'leastLoadedCell', mode: 'pool-default', reach: 'both' }, - { method: 'removeSupersededSameCellControls', mode: 'request', reach: 'request' } + // reconcileReservationAccounting and leastLoadedCell are gone too: the first + // repairs exactly two cells' counters and now holds only those rows, and the + // second selects from the inventory its single caller has already locked. +] + +// Every inline `FROM relay_cells ... FOR UPDATE` outside the named lock helpers, +// in source order: whole-table locks in reconciliation and sticky placement, +// and single-row locks for a cell the method is already scoped to (heartbeat, +// fence, drain generation, configuration, or a reservation adjust that runs +// under a lock its caller already holds). A new inline lock fails the census +// below until it is listed here; per-connection paths that touch more than one +// cell go through lockCellRows so the order is fixed. +const NAMED_LOCK_HELPERS = ['lockCellInventory', 'lockGeneralCellInventory', 'lockCellRows'] + +const INLINE_CELL_LOCK_SITES = [ + 'reconcileCellsWithOptions', + 'assignStickyOnce', + 'recordCellHeartbeat', + 'attestCellFence', + 'adoptLegacyCellFence', + 'commitLegacyCellFenceAdoption', + 'prepareCellFenceAttempt', + 'attestCellFenceAttempt', + 'attestCellFenceAttempt', + 'configureCell', + 'assertDrainCellGeneration', + 'adjustCellReservation' ] // The background sweeps, and nothing else. A method reachable from one of these @@ -151,6 +179,42 @@ describe('cell inventory lock call-site census', () => { ) }) + // Why: the census only sees lockCellInventory calls, so a hand-written + // `relay_cells ... FOR UPDATE` would escape classification entirely. + it('routes every relay_cells row lock through a named lock helper', () => { + const lines = storeSource() + const rawSites: string[] = [] + // Whole statements, not a fixed window: a wide column list or a raw + // FOR UPDATE inside query() must not slip past. + const source = lines.join('\n') + const bounds: { name: string; start: number }[] = [] + lines.forEach((line, index) => { + const declaration = DECLARATION.exec(line) + if (declaration) bounds.push({ name: declaration[1]!, start: index }) + }) + const methodAt = (offset: number): string => { + const lineIndex = source.slice(0, offset).split('\n').length - 1 + let name = '' + for (const bound of bounds) if (bound.start <= lineIndex) name = bound.name + return name + } + const tick = String.fromCharCode(96) + const statementCall = new RegExp( + '\\.(queryLocked|query)\\(\\s*' + tick + '([^' + tick + ']*)' + tick, + 'g' + ) + for (const call of source.matchAll(statementCall)) { + const statement = call[2]! + if (!/\bFROM\s+relay_cells\b/.test(statement)) continue + const locks = call[1] === 'queryLocked' || /\bFOR\s+UPDATE\b/.test(statement) + if (!locks) continue + const method = methodAt(call.index) + if (NAMED_LOCK_HELPERS.includes(method)) continue + rawSites.push(method) + } + expect(rawSites).toEqual(INLINE_CELL_LOCK_SITES) + }) + it('leaves no call site taking the inventory without naming a mode', () => { const source = readFileSync(new URL('./assignment-store.ts', import.meta.url), 'utf8') const unclassified = source diff --git a/cloud/apps/relay/src/cell-inventory-per-cell-locking-postgres.test.ts b/cloud/apps/relay/src/cell-inventory-per-cell-locking-postgres.test.ts new file mode 100644 index 00000000000..8c0ebd73f32 --- /dev/null +++ b/cloud/apps/relay/src/cell-inventory-per-cell-locking-postgres.test.ts @@ -0,0 +1,206 @@ +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip + +// Sorted ascending, and the host is pinned to the LAST id on purpose: the +// fleet-wide lock is one ordered scan, so it holds every earlier row while it +// waits on the pinned one. Pinning to the first id would make the two locking +// models indistinguishable. +const cells = ['a', 'b', 'c'].map((suffix) => ({ + id: `percell-postgres-${suffix}`, + url: `https://percell-postgres-${suffix}.example.com`, + capacityRequests: 1_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 +})) +const [cellA, cellB, cellC] = cells as [(typeof cells)[0], (typeof cells)[0], (typeof cells)[0]] +const identity = { userId: 'percell-postgres-user', relayHostId: 'percellhost00001' } + +function heartbeat(cell: (typeof cells)[number]) { + return { + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionInclusionWatermark: 1, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + } +} + +describePostgres('PostgreSQL per-cell inventory locking', () => { + const databases: RelayDatabase[] = [] + + beforeAll(async () => { + for (let index = 0; index < 3; index++) { + databases.push(await openRelayDatabase({ databaseUrl, dataDir: '' })) + } + }) + + async function removeTestRows(database: RelayDatabase): Promise { + await database.query( + `DELETE FROM relay_control_connection_reservations WHERE user_id LIKE 'percell-postgres-%'` + ) + for (const table of [ + 'relay_assignment_activity_leases', + 'relay_post_drain_migration_pins', + 'relay_assignment_migration_incarnations', + 'relay_assignment_migrations', + 'relay_assignment_region_preferences', + 'relay_assignments' + ]) { + await database.query(`DELETE FROM ${table} WHERE user_id LIKE 'percell-postgres-%'`) + } + for (const cell of cells) { + for (const table of [ + 'relay_cell_connection_snapshots', + 'relay_cell_connection_runtime', + 'relay_cell_connection_limits', + 'relay_cell_runtime', + 'relay_cells' + ]) { + await database.query(`DELETE FROM ${table} WHERE cell_id = ?`, [cell.id]) + } + } + } + + afterAll(async () => { + if (databases[0]) await removeTestRows(databases[0]) + for (const connection of databases) await connection.close() + }) + + async function pinHostToLastCell(store: RelayAssignmentStore): Promise { + await store.reconcileCells(cells) + for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell)) + await store.setCellEnabled(cellA.id, false) + await store.setCellEnabled(cellB.id, false) + const assignment = await store.assign(identity) + expect(assignment.cellId).toBe(cellC.id) + await store.setCellEnabled(cellA.id, true) + await store.setCellEnabled(cellB.id, true) + } + + async function lockWaiterAppeared(database: RelayDatabase): Promise { + const deadline = Date.now() + 4_000 + while (Date.now() < deadline) { + const rows = await database.query( + `SELECT count(*) AS waiting FROM pg_stat_activity + WHERE datname = current_database() AND wait_event_type = 'Lock'` + ) + if (Number(rows[0]!.waiting) > 0) return true + await new Promise((resolve) => setTimeout(resolve, 10)) + } + return false + } + + // Why: a sticky refresh whose first NOWAIT probe loses retries by taking a + // cell row before the assignment row. That retry used to take the whole + // inventory, so one busy cell stalled every other cell's reconnects. + it('waits only on the pinned cell row while refreshing a sticky assignment', async () => { + await removeTestRows(databases[0]!) + const store = new RelayAssignmentStore(databases[0]!, () => 100) + await pinHostToLastCell(store) + // A host whose control lease was already reaped still holds its pin; that + // is the shape that reaches the cell-row probe instead of touchAssignment. + await databases[0]!.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id = ?`, + [identity.userId] + ) + + let releaseRow!: () => void + const rowReleased = new Promise((resolve) => { + releaseRow = resolve + }) + let rowHeld!: () => void + const rowHeldPromise = new Promise((resolve) => { + rowHeld = resolve + }) + const holder = databases[1]!.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cellC.id]) + rowHeld() + await rowReleased + }) + await rowHeldPromise + + const refresh = store.assign(identity) + expect(await lockWaiterAppeared(databases[2]!)).toBe(true) + // The refresh is blocked on cell C. Every earlier row must still be free: + // the ordered fleet-wide scan would be holding both of them by now. + const heldWhileRefreshWaits: string[] = [] + await databases[2]!.transaction(async (transaction) => { + for (const cell of [cellA, cellB]) { + try { + await transaction.queryLocked( + `SELECT * FROM relay_cells WHERE cell_id = ?`, + [cell.id], + { failIfUnavailable: true } + ) + } catch { + heldWhileRefreshWaits.push(cell.id) + } + } + }) + releaseRow() + await holder + + expect(heldWhileRefreshWaits).toEqual([]) + expect((await refresh).cellId).toBe(cellC.id) + }, 15_000) + + // Why: the counter moves by a delta now instead of an absolute value read + // from a snapshot, so concurrent movement on the same cell must still sum. + it('keeps a cell reservation exact under concurrent same-cell activity', async () => { + await removeTestRows(databases[0]!) + const store = new RelayAssignmentStore(databases[0]!, () => 100) + await store.reconcileCells(cells) + for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell)) + await store.setCellEnabled(cellA.id, false) + await store.setCellEnabled(cellB.id, false) + + const hosts = Array.from({ length: 6 }, (_, index) => ({ + userId: `percell-postgres-user-${index}`, + relayHostId: `percellhost0000${index}` + })) + const stores = databases.map((database) => new RelayAssignmentStore(database, () => 100)) + await Promise.all(hosts.map((host, index) => stores[index % stores.length]!.assign(host))) + + // One splice each (2 units) on the same cell, from three connections at once. + await Promise.all( + hosts.map((host, index) => + stores[index % stores.length]!.acquireActivity(host, { + activityId: `splice:percell-${index}`, + kind: 'splice', + cellId: cellC.id + }) + ) + ) + const afterAcquire = await databases[0]!.query( + `SELECT reserved_requests FROM relay_cells WHERE cell_id = ?`, + [cellC.id] + ) + // 6 pending control grants + 6 splices at 2 units each. + expect(Number(afterAcquire[0]!.reserved_requests)).toBe(6 + 12) + + await Promise.all( + hosts.map((host, index) => + stores[index % stores.length]!.releaseActivity(host, `splice:percell-${index}`) + ) + ) + const afterRelease = await databases[0]!.query( + `SELECT reserved_requests FROM relay_cells WHERE cell_id = ?`, + [cellC.id] + ) + expect(Number(afterRelease[0]!.reserved_requests)).toBe(6) + await store.setCellEnabled(cellA.id, true) + await store.setCellEnabled(cellB.id, true) + }, 15_000) +}) diff --git a/cloud/apps/relay/src/control-rebind-inventory-lock-postgres.test.ts b/cloud/apps/relay/src/control-rebind-inventory-lock-postgres.test.ts new file mode 100644 index 00000000000..e990ac1ed1a --- /dev/null +++ b/cloud/apps/relay/src/control-rebind-inventory-lock-postgres.test.ts @@ -0,0 +1,260 @@ +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip + +// Three cells: the inventory lock covers more than the rows a move touches, and +// a high-to-low move exposes any lock taken out of cell_id order. +const cells = [ + { + id: 'rebind-inventory-postgres-a', + url: 'https://rebind-inventory-postgres-a.example.com', + capacityRequests: 1_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + }, + { + id: 'rebind-inventory-postgres-b', + url: 'https://rebind-inventory-postgres-b.example.com', + capacityRequests: 1_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + }, + { + id: 'rebind-inventory-postgres-c', + url: 'https://rebind-inventory-postgres-c.example.com', + capacityRequests: 1_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + } +] +const identity = { userId: 'rebind-inventory-postgres-user', relayHostId: 'rebindinvhost001' } + +function heartbeat(cell: (typeof cells)[number]) { + return { + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionInclusionWatermark: 1, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + } +} + +// Why: every desktop control rebind used to take the fleet-wide relay_cells +// FOR UPDATE lock, so a rebind on one cell queued behind whatever held any +// other cell's row, until COMMIT (55P03 at the request bound). A rebind only +// touches its own cell row, so it must proceed while another cell's row is +// held elsewhere. +describePostgres('PostgreSQL control rebind under a held cell row', () => { + const databases: RelayDatabase[] = [] + + beforeAll(async () => { + databases.push( + await openRelayDatabase({ databaseUrl, dataDir: '' }), + await openRelayDatabase({ databaseUrl, dataDir: '' }) + ) + }) + + async function removeTestRows(database: RelayDatabase): Promise { + await database.query( + `DELETE FROM relay_control_connection_reservations WHERE user_id = ?`, + [identity.userId] + ) + for (const table of [ + 'relay_assignment_activity_leases', + 'relay_post_drain_migration_pins', + 'relay_assignment_migration_incarnations', + 'relay_assignment_migrations', + 'relay_assignments' + ]) { + await database.query(`DELETE FROM ${table} WHERE user_id = ?`, [identity.userId]) + } + for (const cell of cells) { + for (const table of [ + 'relay_cell_connection_snapshots', + 'relay_cell_connection_runtime', + 'relay_cell_connection_limits', + 'relay_cell_runtime', + 'relay_cells' + ]) { + await database.query(`DELETE FROM ${table} WHERE cell_id = ?`, [cell.id]) + } + } + } + + afterAll(async () => { + if (databases[0]) await removeTestRows(databases[0]) + for (const connection of databases) await connection.close() + }) + + it("rebinds and supersedes a control while another cell's row is held", async () => { + // A prior aborted run leaves connection snapshots that reject a replayed watermark. + await removeTestRows(databases[0]!) + const store = new RelayAssignmentStore(databases[0]!, () => 100) + await store.reconcileCells(cells) + for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell)) + // Pin the host to cell A so placement is deterministic. + await store.setCellEnabled(cells[1]!.id, false) + await store.setCellEnabled(cells[2]!.id, false) + const assignment = await store.assign(identity) + expect(assignment.cellId).toBe(cells[0]!.id) + await store.setCellEnabled(cells[1]!.id, true) + await store.setCellEnabled(cells[2]!.id, true) + await store.activateControl(identity, { + cellId: cells[0]!.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1, + connectionInclusionWatermark: 10 + }) + + // Hold only cell B's row on a second connection, the way a rebind on B + // does, for longer than the request-path lock bound. + let releaseInventory!: () => void + const inventoryReleased = new Promise((resolve) => { + releaseInventory = resolve + }) + let inventoryHeld!: () => void + const inventoryHeldPromise = new Promise((resolve) => { + inventoryHeld = resolve + }) + const holder = databases[1]!.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cells[1]!.id]) + inventoryHeld() + await inventoryReleased + }) + await inventoryHeldPromise + + // A generation-2 rebind on cell A supersedes generation 1. It must not + // wait on cell B's row. + const startedAt = Date.now() + const blockedStatement = async (): Promise => { + const rows = await databases[1]!.query( + `SELECT left(query, 160) AS q FROM pg_stat_activity + WHERE datname = current_database() AND wait_event_type = 'Lock'` + ) + return rows.map((row) => String(row.q)).join(' | ') + } + const timeout = new Promise((_, reject) => + setTimeout( + () => + void blockedStatement().then((statement) => + reject(new Error(`rebind on cell A blocked behind cell B's row: ${statement}`)) + ), + 2_000 + ) + ) + const rebound = await Promise.race([ + store.activateControl(identity, { + cellId: cells[0]!.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 2, + connectionInclusionWatermark: 11 + }), + timeout + ]) + const elapsedMs = Date.now() - startedAt + releaseInventory() + await holder + + expect(rebound).toBe(`control:${cells[0]!.id}:2`) + expect(elapsedMs).toBeLessThan(2_000) + const controls = await databases[0]!.query( + `SELECT activity_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND activity_kind = 'control' ORDER BY activity_id`, + [identity.userId] + ) + expect(controls).toEqual([{ activity_id: `control:${cells[0]!.id}:2` }]) + const reserved = await databases[0]!.query( + `SELECT reserved_requests FROM relay_cells WHERE cell_id = ?`, + [cells[0]!.id] + ) + expect(Number(reserved[0]!.reserved_requests)).toBe(1) + }, 15_000) + + // Why: a phone's activity id is client-chosen and can follow the host across + // a migration, so acquireActivity may touch two cell rows. Moving from the + // higher cell to the lower one is where an unordered lock cycles with + // placement's ascending inventory lock (reproduced live before this fix). + it('moves an activity from a higher cell to a lower one in cell_id order', async () => { + await removeTestRows(databases[0]!) + const [cellA, cellB, cellC] = cells as [typeof cells[0], typeof cells[0], typeof cells[0]] + const store = new RelayAssignmentStore(databases[0]!, () => 100) + await store.reconcileCells(cells) + for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell)) + await store.setCellEnabled(cellA.id, false) + await store.setCellEnabled(cellB.id, false) + const assignment = await store.assign(identity) + expect(assignment.cellId).toBe(cellC.id) + await store.setCellEnabled(cellA.id, true) + await store.setCellEnabled(cellB.id, true) + const activityId = 'splice:rebind-inventory-postgres' + await store.acquireActivity(identity, { activityId, kind: 'splice', cellId: cellC.id }) + // The migration makes B authoritative; the lease still sits on C. + const migration = await store.startEvacuation(identity, cellB.id) + expect(migration.targetCellId).toBe(cellB.id) + + // Hold B elsewhere. An ordered move locks B first and queues here holding + // nothing else. Locking C first (the old lease's row, as an unordered move + // does) or the whole inventory (which takes A) shows up as a held row. + let releaseRow!: () => void + const rowReleased = new Promise((resolve) => { + releaseRow = resolve + }) + let rowHeld!: () => void + const rowHeldPromise = new Promise((resolve) => { + rowHeld = resolve + }) + const heldWhileMoverWaits: string[] = [] + const holder = databases[1]!.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cellB.id]) + rowHeld() + await rowReleased + for (const cell of [cellA, cellC]) { + try { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cell.id], { + failIfUnavailable: true + }) + } catch { + heldWhileMoverWaits.push(cell.id) + } + } + }) + await rowHeldPromise + const move = store.acquireActivity(identity, { activityId, kind: 'splice', cellId: cellB.id }) + let moved = false + void move.then(() => { + moved = true + }) + await new Promise((resolve) => setTimeout(resolve, 250)) + expect(moved).toBe(false) + releaseRow() + await holder + await move + expect(heldWhileMoverWaits).toEqual([]) + + const reservations = await databases[0]!.query( + `SELECT cell_id, reserved_requests FROM relay_cells + WHERE cell_id IN (?, ?, ?) ORDER BY cell_id ASC`, + [cellA.id, cellB.id, cellC.id] + ) + const reserved = reservations.map((row) => [String(row.cell_id), Number(row.reserved_requests)]) + expect(reserved).toEqual([ + [cellA.id, 0], + // Migration grant plus the moved splice, as in the SQLite origin-scoped + // reservation case: the lock change did not alter accounting. + [cellB.id, 6], + // The sticky grant stays on the source until the migration completes. + [cellC.id, 1] + ]) + }, 15_000) +}) diff --git a/cloud/apps/relay/src/database-postgres-timeout.test.ts b/cloud/apps/relay/src/database-postgres-timeout.test.ts index 7aba1234f7f..c9021a9ef18 100644 --- a/cloud/apps/relay/src/database-postgres-timeout.test.ts +++ b/cloud/apps/relay/src/database-postgres-timeout.test.ts @@ -2,7 +2,10 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const fakes = vi.hoisted(() => ({ configs: [] as Array>, - query: vi.fn(async () => ({ rows: [], rowCount: 0 })), + // Pool construction and pool shutdown interleaved, so "the schema pool is + // gone before the serving pool opens" is checkable rather than assumed. + lifecycle: [] as string[], + query: vi.fn(async (_sql: string) => ({ rows: [], rowCount: 0 })), release: vi.fn(), end: vi.fn(async () => undefined) })) @@ -13,20 +16,37 @@ vi.mock('pg', () => ({ totalCount = 1 idleCount = 1 waitingCount = 0 - end = fakes.end on = vi.fn() connect = vi.fn(async () => ({ query: fakes.query, release: fakes.release })) + private readonly label: string constructor(config: Record) { fakes.configs.push(config) + this.label = `max=${String(config.max)} statement_timeout=${String(config.statement_timeout)}` + fakes.lifecycle.push(`open ${this.label}`) + } + + async end(): Promise { + fakes.lifecycle.push(`end ${this.label}`) + await fakes.end() } } } })) -import { openRelayDatabase } from './database.js' +import { openRelayDatabase, relayPostgresStatementTimeoutMs } from './database.js' import { applyPostgresSchema } from './postgres-schema-startup.js' +const SCHEMA_POOL = { + max: 1, + application_name: 'orca-relay/director/director/schema', + connectionTimeoutMillis: 2_000, + // Why: DDL must not inherit the request deadline. + statement_timeout: 0, + lock_timeout: 1_000, + idle_in_transaction_session_timeout: 5_000 +} + afterEach(() => { vi.restoreAllMocks() }) @@ -34,9 +54,11 @@ afterEach(() => { describe('PostgreSQL relay deadlines', () => { beforeEach(() => { fakes.configs.length = 0 + fakes.lifecycle.length = 0 fakes.query.mockClear() fakes.release.mockClear() fakes.end.mockClear() + delete process.env.ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS }) it('bounds pool acquisition, statements, locks, and abandoned transactions', async () => { @@ -48,6 +70,7 @@ describe('PostgreSQL relay deadlines', () => { }) expect(fakes.configs).toEqual([ + expect.objectContaining(SCHEMA_POOL), expect.objectContaining({ max: 3, application_name: 'orca-relay/director/director', @@ -59,6 +82,110 @@ describe('PostgreSQL relay deadlines', () => { ]) await database.close() }) + + // Why: an untimed session left open would be a standing way for request work + // to escape the deadline this whole pool config exists to enforce. + it('closes the untimed schema pool before the serving pool opens', async () => { + const database = await openRelayDatabase({ + databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay', + dataDir: './unused', + poolMax: 3, + applicationName: 'orca-relay/director/director' + }) + + expect(fakes.lifecycle).toEqual([ + 'open max=1 statement_timeout=0', + 'end max=1 statement_timeout=0', + 'open max=3 statement_timeout=5000' + ]) + await database.close() + }) + + it('applies the schema on the untimed pool, never on the serving one', async () => { + fakes.query.mockClear() + const ddl: string[] = [] + fakes.query.mockImplementation(async (sql: string) => { + // Every statement issued before the serving pool exists is schema work. + if (fakes.lifecycle.length === 1) ddl.push(sql) + return { rows: [], rowCount: 0 } + }) + const database = await openRelayDatabase({ + databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay', + dataDir: './unused' + }) + + expect(ddl.length).toBeGreaterThan(0) + // Statements can open with a leading `--` rationale comment. + const body = (statement: string): string => + statement.replace(/^(?:\s*--[^\n]*\n)*\s*/, '') + expect(ddl.every((statement) => /^CREATE\b/i.test(body(statement)))).toBe(true) + // The backfill is DML, so it stays on the deadline-bearing serving pool. + expect(ddl.some((statement) => statement.includes('INSERT INTO'))).toBe(false) + await database.close() + }) + + it('takes the serving statement deadline from the environment', async () => { + process.env.ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS = '2500' + + const database = await openRelayDatabase({ + databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay', + dataDir: './unused' + }) + + expect(fakes.configs).toEqual([ + expect.objectContaining({ statement_timeout: 0 }), + expect.objectContaining({ statement_timeout: 2_500 }) + ]) + await database.close() + }) + + it.each(['0', '-1', '2.5', 'soon', ' '])( + 'refuses %s as a statement deadline instead of running unbounded', + (value) => { + expect(() => + relayPostgresStatementTimeoutMs({ ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS: value }) + ).toThrow('invalid_statement_timeout') + } + ) + + it.each([undefined, ''])('defaults to 5s when the environment says %s', (value) => { + expect( + relayPostgresStatementTimeoutMs( + value === undefined ? {} : { ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS: value } + ) + ).toBe(5_000) + }) + + // Why: a statement deadline that reaches the caller as a crash converts a + // transient stall into a failed assignment. It aborts the transaction exactly + // as a lock timeout does, so it belongs on the same bounded retry. + it('retries a statement timeout on a fresh client', async () => { + vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const database = await openRelayDatabase({ + databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay', + dataDir: './unused' + }) + let attempts = 0 + + const result = await database.transaction(async (transaction) => { + attempts += 1 + if (attempts === 1) { + await transaction.query('SELECT 1') + throw Object.assign(new Error('canceling statement due to statement timeout'), { + code: '57014' + }) + } + return 'committed' + }) + + expect(result).toBe('committed') + expect(attempts).toBe(2) + expect(console.warn).toHaveBeenCalledWith( + expect.stringContaining('"event":"orca_relay_postgres_transaction_retry"') + ) + expect(console.warn).toHaveBeenCalledWith(expect.stringContaining('"code":"57014"')) + await database.close() + }) }) describe('PostgreSQL schema startup', () => { diff --git a/cloud/apps/relay/src/database-statement-timeout-postgres.test.ts b/cloud/apps/relay/src/database-statement-timeout-postgres.test.ts new file mode 100644 index 00000000000..6b7ebb0334e --- /dev/null +++ b/cloud/apps/relay/src/database-statement-timeout-postgres.test.ts @@ -0,0 +1,98 @@ +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip +const applicationName = 'orca-relay/statement-timeout-postgres' + +describePostgres('PostgreSQL statement deadline', () => { + const databases: RelayDatabase[] = [] + + beforeAll(async () => { + databases.push(await openRelayDatabase({ databaseUrl, dataDir: '' })) + }) + + afterAll(async () => { + for (const database of databases) await database.close() + }) + + it('serves requests under the configured deadline', async () => { + const database = await openRelayDatabase({ databaseUrl, dataDir: '', statementTimeoutMs: 300 }) + databases.push(database) + + expect(await database.query(`SELECT current_setting('statement_timeout') AS statement_timeout`)).toEqual([ + { statement_timeout: '300ms' } + ]) + }) + + // Why: a real 57014 aborts the transaction exactly as a lock timeout does. If + // it escapes the bounded retry it becomes a failed assignment instead of a + // slow one. + it('retries a real statement timeout on a fresh client', async () => { + const database = await openRelayDatabase({ databaseUrl, dataDir: '', statementTimeoutMs: 300 }) + databases.push(database) + let attempts = 0 + + const result = await database.transaction(async (transaction) => { + attempts += 1 + if (attempts === 1) await transaction.query(`SELECT pg_sleep(2)`) + return attempts + }) + + expect(result).toBe(2) + }, 15_000) + + // Why: DDL runs on its own untimed connection. relay_invites carries a + // CREATE INDEX IF NOT EXISTS, which (unlike CREATE TABLE IF NOT EXISTS) + // really does queue behind an ACCESS EXCLUSIVE lock on the table. + it('applies the schema behind a held ACCESS EXCLUSIVE lock', async () => { + let releaseTable!: () => void + const tableReleased = new Promise((resolve) => { + releaseTable = resolve + }) + let tableHeld!: () => void + const tableHeldPromise = new Promise((resolve) => { + tableHeld = resolve + }) + const holder = databases[0]!.transaction(async (transaction) => { + await transaction.query(`LOCK TABLE relay_invites IN ACCESS EXCLUSIVE MODE`) + tableHeld() + await tableReleased + }) + await tableHeldPromise + + const opening = openRelayDatabase({ + databaseUrl, + dataDir: '', + applicationName, + // Far too short for a blocked DDL; the serving pool wears it, the schema + // connection must not. + statementTimeoutMs: 200 + }) + const blockedOnSchemaConnection = async (): Promise => { + const deadline = Date.now() + 4_000 + while (Date.now() < deadline) { + const rows = await databases[0]!.query( + `SELECT count(*) AS waiting FROM pg_stat_activity + WHERE datname = current_database() AND wait_event_type = 'Lock' + AND application_name = ?`, + [`${applicationName}/schema`] + ) + if (Number(rows[0]!.waiting) > 0) return true + await new Promise((resolve) => setTimeout(resolve, 10)) + } + return false + } + const blocked = await blockedOnSchemaConnection() + releaseTable() + await holder + + const database = await opening + databases.push(database) + expect(blocked).toBe(true) + // The serving pool still carries the short deadline it was opened with. + expect(await database.query(`SELECT current_setting('statement_timeout') AS statement_timeout`)).toEqual([ + { statement_timeout: '200ms' } + ]) + }, 15_000) +}) diff --git a/cloud/apps/relay/src/database.ts b/cloud/apps/relay/src/database.ts index 326ab010ccb..2558831ca64 100644 --- a/cloud/apps/relay/src/database.ts +++ b/cloud/apps/relay/src/database.ts @@ -796,12 +796,34 @@ class PostgresTransaction implements RelayDatabase { const POSTGRES_TRANSACTION_ATTEMPTS = 3 const POSTGRES_RETRY_MAX_DELAY_MS = 25 const POSTGRES_CONNECTION_TIMEOUT_MS = 2_000 -const POSTGRES_STATEMENT_TIMEOUT_MS = 5_000 +// Derivation: a control renewal must land inside its own 30s tick +// (RELAY_PROTOCOL_LIMITS.controlPingIntervalMs * 2), and a transaction gets +// POSTGRES_TRANSACTION_ATTEMPTS tries, so the worst case a renewal can spend in +// Postgres is attempts * timeout. 5s keeps that at 15s, half the tick, and still +// leaves room for the connect timeout above. +export const POSTGRES_STATEMENT_TIMEOUT_MS = 5_000 const POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS = 5_000 +export function relayPostgresStatementTimeoutMs( + env: NodeJS.ProcessEnv = process.env +): number { + const configured = env.ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS + if (configured === undefined || configured === '') return POSTGRES_STATEMENT_TIMEOUT_MS + const milliseconds = Number(configured) + // 0 is PostgreSQL's "no timeout"; refusing it keeps the deadline this exists + // to enforce from being disabled by a typo in an environment variable. + if (!Number.isInteger(milliseconds) || milliseconds < 1) { + throw new Error('invalid_statement_timeout') + } + return milliseconds +} + function retryablePostgresTransactionError(error: unknown): boolean { const code = String((error as { code?: unknown }).code) - return code === '40P01' || code === '40001' || code === '55P03' + // 57014 is the pool statement_timeout firing. It aborts the transaction the + // same way a lock timeout does, so it belongs on the bounded retry path + // rather than surfacing as a terminal failure to the caller. + return code === '40P01' || code === '40001' || code === '55P03' || code === '57014' } export function isRelayDatabaseTransientError(error: unknown): boolean { @@ -963,11 +985,36 @@ async function applySchema(database: RelayDatabase): Promise { } } -async function applySchemaWithPostgresRetries(database: RelayDatabase): Promise { - await applyPostgresSchema( - SCHEMA.split(';').filter((statement) => statement.trim()), - async (statement) => await database.query(statement) - ) +// Why: DDL is not a request. A CREATE INDEX on a grown table legitimately runs +// longer than the request statement_timeout, and inheriting that timeout would +// make every startup fail at the same statement instead of finishing once. One +// short-lived connection of its own, ended before the serving pool opens, keeps +// the untimed session off the request path entirely. +async function applySchemaOnUntimedPool( + databaseUrl: string, + applicationName: string | undefined +): Promise { + const pool = new pg.Pool({ + connectionString: databaseUrl, + max: 1, + application_name: applicationName ? `${applicationName}/schema` : undefined, + connectionTimeoutMillis: POSTGRES_CONNECTION_TIMEOUT_MS, + statement_timeout: 0, + // Kept: a DDL blocked behind another director's ACCESS EXCLUSIVE lock must + // yield to the bounded schema retry instead of holding the connection. + lock_timeout: POSTGRES_LOCK_TIMEOUT_MS, + idle_in_transaction_session_timeout: POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS + }) + absorbPostgresIdleClientErrors(pool) + const database = new PostgresDatabase(pool) + try { + await applyPostgresSchema( + SCHEMA.split(';').filter((statement) => statement.trim()), + async (statement) => await database.query(statement) + ) + } finally { + await database.close().catch(() => undefined) + } } async function backfillRelayCellRegions(database: RelayDatabase): Promise { @@ -983,15 +1030,17 @@ export async function openRelayDatabase(input: { dataDir: string poolMax?: number applicationName?: string + statementTimeoutMs?: number }): Promise { let database: RelayDatabase if (input.databaseUrl) { + await applySchemaOnUntimedPool(input.databaseUrl, input.applicationName) const pool = new pg.Pool({ connectionString: input.databaseUrl, max: input.poolMax ?? 10, application_name: input.applicationName, connectionTimeoutMillis: POSTGRES_CONNECTION_TIMEOUT_MS, - statement_timeout: POSTGRES_STATEMENT_TIMEOUT_MS, + statement_timeout: input.statementTimeoutMs ?? relayPostgresStatementTimeoutMs(), lock_timeout: POSTGRES_LOCK_TIMEOUT_MS, idle_in_transaction_session_timeout: POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS }) @@ -1004,8 +1053,7 @@ export async function openRelayDatabase(input: { database = new SqliteDatabase(sqlite) } try { - if (input.databaseUrl) await applySchemaWithPostgresRetries(database) - else await applySchema(database) + if (!input.databaseUrl) await applySchema(database) await backfillRelayCellRegions(database) return database } catch (error) { diff --git a/cloud/apps/relay/src/host-close-reason-memory.test.ts b/cloud/apps/relay/src/host-close-reason-memory.test.ts new file mode 100644 index 00000000000..2985e6f1a1d --- /dev/null +++ b/cloud/apps/relay/src/host-close-reason-memory.test.ts @@ -0,0 +1,82 @@ +import { ASSIGNMENT_LIMITS, RELAY_HOST_CLOSE_REASON } from '@orca-cloud/relay-contract' +import { describe, expect, it } from 'vitest' +import { HostCloseReasonMemory } from './host-close-reason-memory.js' + +function memoryAt(clock: { now: number }): HostCloseReasonMemory { + return new HostCloseReasonMemory(() => clock.now) +} + +describe('HostCloseReasonMemory', () => { + it('remembers only reasons it knows', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + memory.record('b', 'quitting') + memory.record('c', Buffer.alloc(0)) + memory.record('d', undefined) + + expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + expect(memory.read('b')).toBeNull() + expect(memory.read('c')).toBeNull() + expect(memory.read('d')).toBeNull() + }) + + it('accepts the reason as the Buffer a ws close delivers', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + + memory.record('a', Buffer.from(RELAY_HOST_CLOSE_REASON.SIGNED_OUT)) + + expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) + + it('expires an entry once its host may have been rebalanced away', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + clock.now += ASSIGNMENT_LIMITS.dormantTtlMs - 1 + expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + clock.now += 1 + expect(memory.read('a')).toBeNull() + expect(memory.size()).toBe(0) + }) + + it('forgets on demand', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + memory.forget('a') + + expect(memory.read('a')).toBeNull() + }) + + it('drops the oldest survivors rather than growing without bound', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + for (let index = 0; index < 50_050; index++) { + memory.record(`host-${index}`, RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + } + + expect(memory.size()).toBe(50_000) + expect(memory.read('host-0')).toBeNull() + expect(memory.read('host-50049')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) + + it('re-recording refreshes recency so a live host is not evicted first', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + memory.record('b', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + expect([...['a', 'b'].map((key) => memory.read(key))]).toEqual([ + RELAY_HOST_CLOSE_REASON.SIGNED_OUT, + RELAY_HOST_CLOSE_REASON.SIGNED_OUT + ]) + expect(memory.size()).toBe(2) + }) +}) diff --git a/cloud/apps/relay/src/host-close-reason-memory.ts b/cloud/apps/relay/src/host-close-reason-memory.ts new file mode 100644 index 00000000000..ed01aacd666 --- /dev/null +++ b/cloud/apps/relay/src/host-close-reason-memory.ts @@ -0,0 +1,72 @@ +import { + ASSIGNMENT_LIMITS, + relayHostCloseReasonFrom, + type RelayHostCloseReason +} from '@orca-cloud/relay-contract' + +// Retention matches the dormant assignment TTL: past it the host may have been +// rebalanced onto another cell, so this cell is no longer the one a phone asks. +const RETENTION_MS = ASSIGNMENT_LIMITS.dormantTtlMs +// A fleet-wide auth outage signs out every host at once; the cap bounds that +// burst well above any single cell's host count without becoming a leak. +const MAX_ENTRIES = 50_000 + +// Why in-memory and not Postgres: a phone reaches the cell its host's assignment +// row already names, which is the same cell that watched the control socket +// close. Losing this on a cell restart degrades to the pre-existing generic +// verdict, so the failure mode is the old behaviour rather than a wrong one. +export class HostCloseReasonMemory { + private readonly entries = new Map() + + constructor(private readonly now: () => number = Date.now) {} + + // Silently ignores anything that is not a known reason, which is every close + // from a host that predates the field and every abrupt 1006. + record(key: string, reason: unknown): void { + const parsed = relayHostCloseReasonFrom(reason) + if (!parsed) { + return + } + this.entries.delete(key) + this.entries.set(key, { reason: parsed, expiresAt: this.now() + RETENTION_MS }) + this.evict() + } + + forget(key: string): void { + this.entries.delete(key) + } + + read(key: string): RelayHostCloseReason | null { + const entry = this.entries.get(key) + if (!entry) { + return null + } + if (entry.expiresAt <= this.now()) { + this.entries.delete(key) + return null + } + return entry.reason + } + + size(): number { + return this.entries.size + } + + private evict(): void { + const now = this.now() + for (const [key, entry] of this.entries) { + if (entry.expiresAt > now) { + break + } + this.entries.delete(key) + } + // Insertion order is recency order (record deletes before setting), so the + // head is always the oldest survivor. + for (const key of this.entries.keys()) { + if (this.entries.size <= MAX_ENTRIES) { + break + } + this.entries.delete(key) + } + } +} diff --git a/cloud/apps/relay/src/host-session-client-accept.test.ts b/cloud/apps/relay/src/host-session-client-accept.test.ts new file mode 100644 index 00000000000..83b6c21f997 --- /dev/null +++ b/cloud/apps/relay/src/host-session-client-accept.test.ts @@ -0,0 +1,382 @@ +import { EventEmitter } from 'node:events' +import { RELAY_CLOSE_CODE } from '@orca-cloud/relay-contract' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type WebSocket from 'ws' +import type { RelayAssignmentStore } from './assignment-store.js' +import type { RelayConfig } from './config.js' +import type { CredentialReservation, RelayCredentialStore } from './credential-store.js' +import { + CONTROL_LEASE_JITTER_MS, + CONTROL_LEASE_MS, + HostSessionRegistry +} from './host-session-registry.js' +import type { RelayRuntimeObserver } from './relay-observability.js' +import type { RelayTokenClaims } from './relay-token-verifier.js' +import { ProcessQueuedByteBudget } from './splice-forwarder.js' + +// Incident 2026-09-04 ~01:05Z: the phone's dial bound ran out while the cell was +// still inside acceptClient's serialized Postgres phase (cell-inventory lock +// contention). The cell then finished the work for a socket nobody held, holding +// an activity lease for the 10s attach deadline before its timer unwound it, and +// logged `host_data_reservation_already_bound`. + +class FakeSocket extends EventEmitter { + readonly OPEN = 1 + readonly CLOSING = 2 + readonly CLOSED = 3 + readyState = this.OPEN + readonly send = vi.fn() + readonly close = vi.fn((code?: number, reason?: string) => { + this.readyState = this.CLOSED + this.emit('close', code, Buffer.from(reason ?? '')) + }) + readonly terminate = vi.fn(() => { + this.readyState = this.CLOSED + this.emit('close') + }) +} + +const config = { + port: 8080, + publicUrl: 'https://relay-c3.example.com', + cellUrl: 'https://relay-c3.example.com', + authIssuer: 'https://auth.example.com', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.com/jwks', + assignmentSigningKey: new Uint8Array(32), + role: 'cell', + cellId: 'production-gce-c3', + cells: [{ id: 'production-gce-c3', url: 'https://relay-c3.example.com', capacityRequests: 4_000 }], + adminAudience: 'https://relay-c3.example.com/v1/admin/drain', + deployServiceAccount: 'deploy@example.com', + runtimeServiceAccount: 'runtime@example.com', + adminJwksUrl: 'https://auth.example.com/admin-jwks', + databasePoolMax: 10, + publicAssignmentsEnabled: true, + publicAssignmentConcurrency: 2, + publicAssignmentQueueMax: 128, + publicAssignmentWaitMs: 4_000, + publicResolveConcurrency: 1, + publicResolveWaitMs: 5_000, + publicAssignmentRetryAfterSeconds: 5, + dataDir: './test-data' +} satisfies RelayConfig + +const identity = { + sub: 'user-1', + prof: 'profile-1', + relayHostId: 'abcdefghijklmnop', + purpose: 'host-control', + exp: 4_102_444_800 +} satisfies RelayTokenClaims + +function deferred(): { promise: Promise; resolve: (value: T) => void } { + let resolve!: (value: T) => void + const promise = new Promise((next) => (resolve = next)) + return { promise, resolve } +} + +const reservation: CredentialReservation = { + userId: identity.sub, + relayHostId: identity.relayHostId, + credentialKind: 'resume', + relayDeviceId: 'device-1', + tokenHash: 'hash', + reservationId: 'reservation-1', + leaseExpiresAt: Date.now() + 60_000, + acceptedCredentialVersion: 2, + acceptedAs: 'current' +} + +function harness(options: { random?: () => number; now?: () => number } = {}) { + const acquireActivity = vi.fn().mockResolvedValue(undefined) + const releaseActivity = vi.fn().mockResolvedValue(true) + const assignments = { + activateControl: vi.fn().mockResolvedValue('control:production-gce-c3:1'), + markMigrationTargetRegistered: vi.fn().mockResolvedValue(undefined), + resolve: vi.fn().mockResolvedValue({ cellId: config.cellId }), + acquireActivity, + renewControlActivity: vi.fn().mockResolvedValue(undefined), + releaseActivity + } as unknown as RelayAssignmentStore + const store = { + resolveResume: vi.fn().mockResolvedValue({ userId: identity.sub }), + reserveCredential: vi.fn().mockResolvedValue(reservation), + failReservation: vi.fn().mockResolvedValue(undefined) + } + const observer = { + recordAuth: vi.fn(), + recordForwardedBytes: vi.fn(), + recordHttp: vi.fn(), + recordReconnect: vi.fn(), + recordSql: vi.fn(), + recordClientAcceptAbandoned: vi.fn() + } satisfies RelayRuntimeObserver + const registry = new HostSessionRegistry( + config, + vi.fn(), + store as unknown as RelayCredentialStore, + assignments, + new ProcessQueuedByteBudget(), + observer, + options.now, + options.random + ) + const activate = ( + registry as unknown as { + activate: ( + socket: WebSocket, + identity: RelayTokenClaims, + existing: null, + generation: number, + rebind: boolean, + assignmentEpoch: number, + appVersion: string + ) => Promise + } + ).activate.bind(registry) + return { registry, store, assignments, acquireActivity, releaseActivity, observer, activate } +} + +async function activeHost(h: ReturnType): Promise { + const control = new FakeSocket() + await h.activate(control as unknown as WebSocket, identity, null, 1, false, 1, '1.4.197') + return control +} + +describe('client accept abandoned mid-DB-phase', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => { + vi.clearAllTimers() + vi.useRealTimers() + }) + + it('stops after a slow activity acquire when the phone already hung up', async () => { + const h = harness() + const control = await activeHost(h) + const slowAcquire = deferred() + h.acquireActivity.mockReturnValueOnce(slowAcquire.promise) + const capacity = { bind: vi.fn(), release: vi.fn() } + const client = new FakeSocket() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + try { + const accepting = h.registry.acceptClient( + client as unknown as WebSocket, + identity.relayHostId, + 'credential', + capacity + ) + await vi.advanceTimersByTimeAsync(0) + expect(h.acquireActivity).toHaveBeenCalledOnce() + // The phone's 12s bound fires while the cell still waits on Postgres. + client.close(1000, 'client bound') + capacity.release() + slowAcquire.resolve() + await accepting + + // No conn-open reached the desktop; nothing pending; the lease it just took is + // released instead of leaking to expiry cleanup; bind never throws. + expect(control.send).not.toHaveBeenCalledWith(expect.stringContaining('conn-open')) + expect(capacity.bind).not.toHaveBeenCalled() + const session = h.registry.get({ userId: identity.sub, relayHostId: identity.relayHostId }) + expect(session?.pendingConns.size).toBe(0) + expect(h.store.failReservation).toHaveBeenCalledWith(reservation) + expect(h.releaseActivity).toHaveBeenCalledWith( + { userId: identity.sub, relayHostId: identity.relayHostId }, + expect.stringMatching(/^confirmation:/) + ) + expect(h.observer.recordClientAcceptAbandoned).toHaveBeenCalledWith( + 'activity', + expect.any(Number) + ) + const line = warn.mock.calls.map((call) => String(call[0])).find((entry) => + entry.includes('orca_relay_client_accept_abandoned') + ) + expect(line).toBeDefined() + expect(JSON.parse(line!)).toMatchObject({ stage: 'activity' }) + expect(line).not.toContain(identity.relayHostId) + } finally { + warn.mockRestore() + h.registry.drain(0) + vi.advanceTimersByTime(0) + } + }) + + it('stops after a slow credential reservation without acquiring an activity lease', async () => { + const h = harness() + await activeHost(h) + const slowReserve = deferred() + h.store.reserveCredential.mockReturnValueOnce(slowReserve.promise) + const client = new FakeSocket() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + try { + const accepting = h.registry.acceptClient( + client as unknown as WebSocket, + identity.relayHostId, + 'credential' + ) + await vi.advanceTimersByTimeAsync(0) + client.close(1000, 'client bound') + slowReserve.resolve(reservation) + await accepting + + expect(h.acquireActivity).not.toHaveBeenCalled() + expect(h.store.failReservation).toHaveBeenCalledWith(reservation) + expect(h.observer.recordClientAcceptAbandoned).toHaveBeenCalledWith( + 'credential', + expect.any(Number) + ) + } finally { + warn.mockRestore() + h.registry.drain(0) + vi.advanceTimersByTime(0) + } + }) + + it('stops after a slow resume lookup before starting the invite and assignment lookups', async () => { + const h = harness() + await activeHost(h) + const store = h.store as typeof h.store & { resolveInviteForMove: ReturnType } + store.resolveInviteForMove = vi.fn().mockResolvedValue(null) + const slowResume = deferred() + h.store.resolveResume.mockReturnValueOnce(slowResume.promise) + const resolveAssignment = (h.assignments as unknown as { resolve: ReturnType }) + .resolve + resolveAssignment.mockClear() + const client = new FakeSocket() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + try { + const accepting = h.registry.acceptClient( + client as unknown as WebSocket, + identity.relayHostId, + 'credential' + ) + await vi.advanceTimersByTimeAsync(0) + client.close(1000, 'client bound') + slowResume.resolve(null) + await accepting + + expect(store.resolveInviteForMove).not.toHaveBeenCalled() + expect(resolveAssignment).not.toHaveBeenCalled() + expect(h.store.reserveCredential).not.toHaveBeenCalled() + expect(h.observer.recordClientAcceptAbandoned).toHaveBeenCalledWith( + 'assignment', + expect.any(Number) + ) + } finally { + warn.mockRestore() + h.registry.drain(0) + vi.advanceTimersByTime(0) + } + }) + + it('stops after a slow same-cell assignment resolve, before reserving a credential', async () => { + const h = harness() + await activeHost(h) + const resolveAssignment = (h.assignments as unknown as { resolve: ReturnType }) + .resolve + const slowResolve = deferred<{ cellId: string }>() + resolveAssignment.mockReturnValueOnce(slowResolve.promise) + const client = new FakeSocket() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + try { + const accepting = h.registry.acceptClient( + client as unknown as WebSocket, + identity.relayHostId, + 'credential' + ) + await vi.advanceTimersByTimeAsync(0) + client.close(1000, 'client bound') + // A correct, same-cell assignment: only the closed socket stops the accept. + slowResolve.resolve({ cellId: config.cellId }) + await accepting + + // Proves the accept reached the third guard, not the first. + expect(resolveAssignment).toHaveBeenCalled() + expect(h.store.reserveCredential).not.toHaveBeenCalled() + expect(h.observer.recordClientAcceptAbandoned).toHaveBeenCalledWith( + 'assignment', + expect.any(Number) + ) + } finally { + warn.mockRestore() + h.registry.drain(0) + vi.advanceTimersByTime(0) + } + }) + + it('still opens the connection when the phone is holding on', async () => { + const h = harness() + const control = await activeHost(h) + const capacity = { bind: vi.fn(), release: vi.fn() } + const client = new FakeSocket() + await h.registry.acceptClient( + client as unknown as WebSocket, + identity.relayHostId, + 'credential', + capacity + ) + expect(control.send).toHaveBeenCalledWith(expect.stringContaining('"type":"conn-open"')) + expect(capacity.bind).toHaveBeenCalledOnce() + expect(h.observer.recordClientAcceptAbandoned).not.toHaveBeenCalled() + expect(client.close).not.toHaveBeenCalled() + h.registry.drain(0) + vi.advanceTimersByTime(0) + }) +}) + +describe('control lease jitter', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => { + vi.clearAllTimers() + vi.useRealTimers() + }) + + it('grants a lease uniformly around its mean so cohorts drift apart at the same mean rate', async () => { + const now = 1_700_000_000_000 + const helloAck = (socket: FakeSocket) => + JSON.parse( + String(socket.send.mock.calls.find((call) => String(call[0]).includes('host-hello-ack'))![0]) + ) as { leaseExpiresAt: number } + + const shortest = harness({ now: () => now, random: () => 0 }) + const shortestAck = helloAck(await activeHost(shortest)) + const centered = harness({ now: () => now, random: () => 0.5 }) + const centeredAck = helloAck(await activeHost(centered)) + const longestRoll = 0.999999 + const longest = harness({ now: () => now, random: () => longestRoll }) + const longestAck = helloAck(await activeHost(longest)) + + // Pinned, not bounded: a jitter clamped to one side still satisfies an upper + // bound, so only the exact top of the band proves it is symmetric. + const longestOffset = Math.floor((longestRoll * 2 - 1) * CONTROL_LEASE_JITTER_MS) + expect(shortestAck.leaseExpiresAt).toBe(now + CONTROL_LEASE_MS - CONTROL_LEASE_JITTER_MS) + expect(centeredAck.leaseExpiresAt).toBe(now + CONTROL_LEASE_MS) + expect(longestAck.leaseExpiresAt).toBe(now + CONTROL_LEASE_MS + longestOffset) + shortest.registry.drain(0) + centered.registry.drain(0) + longest.registry.drain(0) + vi.advanceTimersByTime(0) + }) + + it('rebinds re-roll the jitter instead of pinning the cohort phase', async () => { + const now = 1_700_000_000_000 + let roll = 0 + const h = harness({ now: () => now, random: () => roll }) + const first = await activeHost(h) + const session = h.registry.get({ userId: identity.sub, relayHostId: identity.relayHostId })! + const firstLease = session.leaseExpiresAt + roll = 0.75 + const rebind = new FakeSocket() + await ( + h.registry as unknown as { + activate: (...args: unknown[]) => Promise + } + ).activate(rebind as unknown as WebSocket, identity, session, 1, true, 1, '1.4.197') + expect(session.leaseExpiresAt).toBe(now + CONTROL_LEASE_MS + CONTROL_LEASE_JITTER_MS / 2) + expect(session.leaseExpiresAt).not.toBe(firstLease) + expect(first.close).toHaveBeenCalledWith(RELAY_CLOSE_CODE.PEER_DROPPED, 'control rebound') + h.registry.drain(0) + vi.advanceTimersByTime(0) + }) +}) diff --git a/cloud/apps/relay/src/host-session-registry.ts b/cloud/apps/relay/src/host-session-registry.ts index 11b7d1de030..1b7ed3df4af 100644 --- a/cloud/apps/relay/src/host-session-registry.ts +++ b/cloud/apps/relay/src/host-session-registry.ts @@ -14,7 +14,8 @@ import { HostHelloSchema, InviteCreateSchema, RELAY_PROTOCOL_LIMITS, - RELAY_CLOSE_CODE + RELAY_CLOSE_CODE, + type RelayHostCloseReason } from '@orca-cloud/relay-contract' import nacl from 'tweetnacl' import type WebSocket from 'ws' @@ -25,9 +26,10 @@ import { RelayCredentialStore, type CredentialReservation } from './credential-store.js' +import { HostCloseReasonMemory } from './host-close-reason-memory.js' import { relayHostLogDigest } from './relay-host-log-digest.js' import type { RelayTokenClaims } from './relay-token-verifier.js' -import type { RelayRuntimeObserver } from './relay-observability.js' +import type { RelayClientAcceptStage, RelayRuntimeObserver } from './relay-observability.js' import type { PendingHostDataReservation } from './relay-connection-ledger.js' import { closeRelayWebSocket } from './relay-websocket-close.js' import { ProcessQueuedByteBudget, wireSplice } from './splice-forwarder.js' @@ -127,9 +129,23 @@ function send(socket: WebSocket, type: string, message: object): void { // stalled predecessor only accumulates doomed sockets. const ACTIVATION_QUEUE_WAIT_MS = 30_000 +// Why: this lease bounds how long a host lingers on a cell after a missed drain, +// and rebinding it is the only passive rebalancing we have, so it has to stay +// finite. 6h keeps both properties while cutting control-activation traffic on +// the contended cell-inventory lock ~6x; the relay JWT (5 min, refreshed by the +// desktop) and the 75s silence watchdog are enforced separately, so a longer +// grant authorizes nothing extra. Symmetric jitter walks same-minute reconnect +// cohorts apart across cycles without changing the mean rebind rate. +export const CONTROL_LEASE_MS = 6 * 60 * 60 * 1000 +export const CONTROL_LEASE_JITTER_MS = 30 * 60 * 1000 + export class HostSessionRegistry { private readonly sessions = new Map() private readonly activationQueues = new Map>() + // Why it outlives `sessions`: the orphan grace deletes the session within 30s, + // but a signed-out desktop never comes back, so the phone that asks minutes + // later would otherwise find nothing to explain its rejection with. + private readonly hostCloseReasons = new HostCloseReasonMemory(() => this.now()) private draining = false constructor( @@ -139,9 +155,16 @@ export class HostSessionRegistry { private readonly assignments: RelayAssignmentStore, private readonly queuedByteBudget: ProcessQueuedByteBudget, private readonly observer: RelayRuntimeObserver, - private readonly now: () => number = Date.now + private readonly now: () => number = Date.now, + private readonly random: () => number = Math.random ) {} + // Uniform over [CONTROL_LEASE_MS - jitter, CONTROL_LEASE_MS + jitter). + private controlLeaseExpiresAt(): number { + const offset = Math.floor((this.random() * 2 - 1) * CONTROL_LEASE_JITTER_MS) + return this.now() + CONTROL_LEASE_MS + offset + } + async acceptClient( socket: WebSocket, hostId: string, @@ -153,10 +176,31 @@ export class HostSessionRegistry { this.rejectClient(socket, RELAY_CLOSE_CODE.DRAINING) return } + // Why: the accept runs several serialized Postgres calls behind the contended + // cell-inventory lock, and phones bound their dial. Finishing the work for a + // phone that already hung up took an activity lease held for the 10s attach + // deadline, then failed at bind with host_data_reservation_already_bound. + const acceptStartedAt = this.now() + const abandonedByClient = (stage: RelayClientAcceptStage, cleanup?: () => void): boolean => { + if (socket.readyState === socket.OPEN) return false + capacityReservation?.release() + cleanup?.() + const elapsedMs = this.now() - acceptStartedAt + this.observer.recordClientAcceptAbandoned?.(stage, elapsedMs) + console.warn( + JSON.stringify({ event: 'orca_relay_client_accept_abandoned', stage, elapsedMs }) + ) + return true + } if (this.config.role === 'cell') { - const outerIdentity = - (await this.store.resolveResume(hostId, credential)) ?? - (await this.store.resolveInviteForMove(hostId, credential)) + // Each lookup is its own pooled round trip; stop between them once the phone + // has left instead of running the rest of the chain for nobody. + let outerIdentity = await this.store.resolveResume(hostId, credential) + if (abandonedByClient('assignment')) return + if (!outerIdentity) { + outerIdentity = await this.store.resolveInviteForMove(hostId, credential) + if (abandonedByClient('assignment')) return + } const assignment = outerIdentity ? await this.assignments.resolve({ userId: outerIdentity.userId, relayHostId: hostId }) : null @@ -166,6 +210,7 @@ export class HostSessionRegistry { this.rejectClient(socket, RELAY_CLOSE_CODE.WRONG_CELL) return } + if (abandonedByClient('assignment')) return } const reservation = await this.store.reserveCredential(hostId, credential) if (!reservation) { @@ -175,7 +220,9 @@ export class HostSessionRegistry { return } this.observer.recordAuth(true) - const session = this.sessions.get(this.key(reservation.userId, hostId)) + if (abandonedByClient('credential', () => this.failReservationBestEffort(reservation))) return + const sessionKey = this.key(reservation.userId, hostId) + const session = this.sessions.get(sessionKey) if ( !session || session.state !== 'active' || @@ -184,7 +231,13 @@ export class HostSessionRegistry { ) { capacityReservation?.release() await this.store.failReservation(reservation) - this.rejectClient(socket, RELAY_CLOSE_CODE.HOST_OFFLINE) + // The only rejection that can name a cause: the host is genuinely absent. + // The attach-deadline 4404 below fires while control is still connected. + this.rejectClient( + socket, + RELAY_CLOSE_CODE.HOST_OFFLINE, + this.hostCloseReasons.read(sessionKey) + ) return } if (session.activeConnIds.size + session.pendingConns.size >= 8) { @@ -214,6 +267,14 @@ export class HostSessionRegistry { return } } + if ( + abandonedByClient('activity', () => { + this.failReservationBestEffort(reservation) + if (credentialActivityId) this.releaseActivityBestEffort(identity, credentialActivityId) + }) + ) { + return + } const attachTimer = setTimeout(() => { session.pendingConns.delete(connId) capacityReservation?.release() @@ -727,7 +788,7 @@ export class HostSessionRegistry { existing.socket = socket existing.state = existing.regionalDrainAttemptId ? 'drain-only' : 'active' existing.appVersion = appVersion - existing.leaseExpiresAt = this.now() + 55 * 60 * 1000 + existing.leaseExpiresAt = this.controlLeaseExpiresAt() existing.lastPongAt = this.now() existing.activityRenewalDueAt = this.now() + RELAY_PROTOCOL_LIMITS.controlPingIntervalMs @@ -778,7 +839,7 @@ export class HostSessionRegistry { appVersion, state: 'active', socket, - leaseExpiresAt: this.now() + 55 * 60 * 1000, + leaseExpiresAt: this.controlLeaseExpiresAt(), orphanTimer: null, heartbeatTimer: null, lastPongAt: this.now(), @@ -793,7 +854,10 @@ export class HostSessionRegistry { regionalDrainTimer: null, regionalDrainExpiresAt: null } - this.sessions.set(this.key(identity.sub, identity.relayHostId), session) + const sessionKey = this.key(identity.sub, identity.relayHostId) + // A host that proved itself again is not signed out, whatever it said last. + this.hostCloseReasons.forget(sessionKey) + this.sessions.set(sessionKey, session) this.wireActiveControl(session) this.sendHelloAck(session) } @@ -813,6 +877,11 @@ export class HostSessionRegistry { }) socket.once('close', (code, reason) => { this.observer.recordControlClose?.(code) + // Guarded on identity: a predecessor retired by a rebind must not stamp a + // cause onto the live session that replaced it. + if (session.socket === socket) { + this.hostCloseReasons.record(this.key(session.identity.sub, session.relayHostId), reason) + } // One line per control close makes reconnect churners attributable by // host digest without exposing the raw relay host id. console.warn( @@ -1187,9 +1256,16 @@ export class HostSessionRegistry { if (session.socket) send(session.socket, 'control-error', { ...(reqId ? { reqId } : {}), code }) } - private rejectClient(socket: WebSocket, code: number): void { + // hostCloseReason rides the WebSocket close reason, never relay-hello: every + // shipped phone parses relay-hello with a strict schema that rejects an + // unknown key, and none of them read the close reason at all. + private rejectClient( + socket: WebSocket, + code: number, + hostCloseReason?: RelayHostCloseReason | null + ): void { send(socket, 'relay-hello', { ok: false, code }) - closeRelayWebSocket(socket, code, 'relay connection rejected') + closeRelayWebSocket(socket, code, hostCloseReason ?? 'relay connection rejected') } private releaseControlActivity(session: HostSession): void { diff --git a/cloud/apps/relay/src/host-signed-out-rejection.test.ts b/cloud/apps/relay/src/host-signed-out-rejection.test.ts new file mode 100644 index 00000000000..0f8542c960f --- /dev/null +++ b/cloud/apps/relay/src/host-signed-out-rejection.test.ts @@ -0,0 +1,206 @@ +import { EventEmitter } from 'node:events' +import { + CONTROL_CONTINUITY_LIMITS, + RELAY_CLOSE_CODE, + RELAY_HOST_CLOSE_REASON +} from '@orca-cloud/relay-contract' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type WebSocket from 'ws' +import type { RelayAssignmentStore } from './assignment-store.js' +import type { RelayConfig } from './config.js' +import type { RelayCredentialStore } from './credential-store.js' +import { HostSessionRegistry } from './host-session-registry.js' +import type { RelayRuntimeObserver } from './relay-observability.js' +import type { RelayTokenClaims } from './relay-token-verifier.js' +import { ProcessQueuedByteBudget } from './splice-forwarder.js' + +class FakeSocket extends EventEmitter { + readonly OPEN = 1 + readonly CLOSED = 3 + readyState = this.OPEN + readonly send = vi.fn() + readonly close = vi.fn((code?: number, reason?: string) => { + this.readyState = this.CLOSED + this.emit('close', code, Buffer.from(reason ?? '')) + }) + readonly terminate = vi.fn(() => { + this.readyState = this.CLOSED + this.emit('close', 1006, Buffer.alloc(0)) + }) +} + +const config = { + port: 8080, + publicUrl: 'https://relay-c3.example.com', + cellUrl: 'https://relay-c3.example.com', + authIssuer: 'https://auth.example.com', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.com/jwks', + assignmentSigningKey: new Uint8Array(32), + role: 'cell', + cellId: 'production-gce-c3', + cells: [] +} as unknown as RelayConfig + +const identity = { + sub: 'user-1', + prof: 'profile-1', + org: 'org-1', + relayHostId: 'AbCdEf0123_-xyZ9' +} as unknown as RelayTokenClaims + +const reservation = { + userId: identity.sub, + relayHostId: identity.relayHostId, + credentialKind: 'resume', + relayDeviceId: 'device-1', + leaseExpiresAt: Date.now() + 60_000 +} + +function createRegistry() { + const store = { + resolveResume: vi.fn().mockResolvedValue({ userId: identity.sub }), + reserveCredential: vi.fn().mockResolvedValue(reservation), + failReservation: vi.fn().mockResolvedValue(undefined) + } + const assignments = { + activateControl: vi.fn().mockResolvedValue('control:production-gce-c3:1'), + markMigrationTargetRegistered: vi.fn().mockResolvedValue(undefined), + resolve: vi.fn().mockResolvedValue({ cellId: config.cellId }), + acquireActivity: vi.fn().mockResolvedValue(undefined), + renewControlActivity: vi.fn().mockResolvedValue(undefined), + releaseActivity: vi.fn().mockResolvedValue(true) + } as unknown as RelayAssignmentStore + const observer = { + recordAuth: vi.fn(), + recordForwardedBytes: vi.fn(), + recordHttp: vi.fn(), + recordReconnect: vi.fn(), + recordSql: vi.fn(), + recordControlClose: vi.fn(), + recordSpliceClose: vi.fn() + } satisfies RelayRuntimeObserver + const registry = new HostSessionRegistry( + config, + vi.fn(), + store as unknown as RelayCredentialStore, + assignments, + new ProcessQueuedByteBudget(), + observer + ) + const activate = (socket: WebSocket, generation: number): Promise => + ( + registry as unknown as { + activate: ( + socket: WebSocket, + identity: RelayTokenClaims, + existing: null, + generation: number, + rebind: boolean, + assignmentEpoch: number, + appVersion: string + ) => Promise + } + ).activate(socket, identity, null, generation, false, 1, '1.4.173') + return { registry, activate } +} + +async function dialPhone(registry: HostSessionRegistry): Promise { + const phone = new FakeSocket() + await registry.acceptClient(phone as unknown as WebSocket, identity.relayHostId, 'credential') + return phone +} + +// The 4404 hello body is unchanged: every shipped phone parses it with a strict +// schema, so the cause has to ride the close frame instead. +const HOST_OFFLINE_HELLO = JSON.stringify({ + type: 'relay-hello', + ok: false, + code: RELAY_CLOSE_CODE.HOST_OFFLINE +}) + +describe('host sign-out reason on phone rejection', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => { + vi.clearAllTimers() + vi.useRealTimers() + }) + + it('names the sign-out to a phone that arrives after the host is gone', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + control.close(1000, RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.send).toHaveBeenCalledWith(HOST_OFFLINE_HELLO) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + RELAY_HOST_CLOSE_REASON.SIGNED_OUT + ) + }) + + it('says nothing when the host died without naming a cause', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + control.terminate() + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + 'relay connection rejected' + ) + }) + + it('ignores a close reason the host invented', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + control.close(1000, 'signed-out-ish') + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + 'relay connection rejected' + ) + }) + + it('forgets the sign-out once the host proves itself again', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + control.close(1000, RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const reconnected = new FakeSocket() + await activate(reconnected as unknown as WebSocket, 2) + // Drop it abruptly, as a network death would, so only the stale memory + // could still name a cause. + reconnected.terminate() + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + 'relay connection rejected' + ) + }) + + // A live host is present: the 4404 there is an attach deadline, not absence. + it('never names a cause while the host control is connected', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + const phone = await dialPhone(registry) + expect(phone.close).not.toHaveBeenCalled() + expect(control.send).toHaveBeenCalledWith(expect.stringContaining('"type":"conn-open"')) + }) +}) diff --git a/cloud/apps/relay/src/postgres-transaction-recovery.test.ts b/cloud/apps/relay/src/postgres-transaction-recovery.test.ts index a49c07d2e7d..ae9d52a7c86 100644 --- a/cloud/apps/relay/src/postgres-transaction-recovery.test.ts +++ b/cloud/apps/relay/src/postgres-transaction-recovery.test.ts @@ -503,12 +503,19 @@ describePostgres('PostgreSQL transaction recovery', () => { const directorLockOrder: string[] = [] const assignmentDatabase = new TransactionProbeDatabase(database, async (phase, sql) => { if (phase === 'before') { - if (sql.includes('FROM relay_assignments WHERE user_id = ?')) { + // Only locked statements reach this hook, so classifying the pin read + // is what proves it stays unlocked: if it ever grows a FOR UPDATE it + // shows up in the order below instead of silently joining the queue. + if (sql.includes('SELECT cell_id FROM relay_assignments')) { + directorLockOrder.push('pin-read') + } else if (sql.includes('FROM relay_assignments WHERE user_id = ?')) { directorLockOrder.push('assignment') } else if (sql.includes('FROM relay_assignment_activity_leases')) { directorLockOrder.push('activity') } else if (sql.includes('FROM relay_cells ORDER BY')) { directorLockOrder.push('cell-inventory') + } else if (sql.includes('FROM relay_cells WHERE cell_id IN')) { + directorLockOrder.push('cell-rows') } else if (sql.includes('FROM relay_cells WHERE cell_id = ?')) { directorLockOrder.push('cell') } @@ -530,11 +537,14 @@ describePostgres('PostgreSQL transaction recovery', () => { }) await expect(legacyTransaction).resolves.toBeUndefined() expect(assignmentDatabase.attempts).toBe(2) + // The retry still takes a cell row before the assignment row — the order + // that avoids the legacy cycle — but only the pinned row, never the + // inventory. expect(directorLockOrder).toEqual([ 'assignment', 'activity', 'cell', - 'cell-inventory', + 'cell-rows', 'assignment', 'activity' ]) diff --git a/cloud/apps/relay/src/relay-observability.test.ts b/cloud/apps/relay/src/relay-observability.test.ts index 2b9ceb0b72a..fc8a4fcb4af 100644 --- a/cloud/apps/relay/src/relay-observability.test.ts +++ b/cloud/apps/relay/src/relay-observability.test.ts @@ -195,16 +195,23 @@ describe('relay observability', () => { observability.recordControlClose(4402) observability.recordSpliceClose('host-oversize-frame') observability.recordSpliceClose('queue-limit') + observability.recordClientAcceptAbandoned('activity', 14_250.4) + observability.recordClientAcceptAbandoned('activity', 2_000) + observability.recordClientAcceptAbandoned('credential', 3_000) observability.flush(counts) observability.flush(counts) expect(entries[0]).toMatchObject({ controlClosesByCodeDelta: { 1006: 2, 4402: 1 }, - spliceClosesByTriggerDelta: { 'host-oversize-frame': 1, 'queue-limit': 1 } + spliceClosesByTriggerDelta: { 'host-oversize-frame': 1, 'queue-limit': 1 }, + clientAcceptsAbandonedByStageDelta: { activity: 2, credential: 1 }, + clientAcceptAbandonedMsMax: 14_250.4 }) expect(entries[1]).toMatchObject({ controlClosesByCodeDelta: {}, - spliceClosesByTriggerDelta: {} + spliceClosesByTriggerDelta: {}, + clientAcceptsAbandonedByStageDelta: {}, + clientAcceptAbandonedMsMax: 0 }) }) diff --git a/cloud/apps/relay/src/relay-observability.ts b/cloud/apps/relay/src/relay-observability.ts index 2266217d607..59ff437e40b 100644 --- a/cloud/apps/relay/src/relay-observability.ts +++ b/cloud/apps/relay/src/relay-observability.ts @@ -64,8 +64,12 @@ export interface RelayRuntimeObserver { }): void recordControlClose?(code: number): void recordSpliceClose?(trigger: string): void + recordClientAcceptAbandoned?(stage: RelayClientAcceptStage, elapsedMs: number): void } +// Which serialized accept step the phone had already hung up behind. +export type RelayClientAcceptStage = 'assignment' | 'credential' | 'activity' + type RelayMetricDeltas = { forwardedBytes: number authSuccesses: number @@ -87,6 +91,8 @@ type RelayMetricDeltas = { unavailableRegions: Record controlClosesByCode: Record spliceClosesByTrigger: Record + clientAcceptsAbandonedByStage: Record + clientAcceptAbandonedMsMax: number controlRenewalLatenciesMs: number[] controlRenewalsByOutcome: Record controlActivityRecoveries: number @@ -116,6 +122,8 @@ const emptyDeltas = (): RelayMetricDeltas => ({ unavailableRegions: {}, controlClosesByCode: {}, spliceClosesByTrigger: {}, + clientAcceptsAbandonedByStage: {}, + clientAcceptAbandonedMsMax: 0, controlRenewalLatenciesMs: [], controlRenewalsByOutcome: {}, controlActivityRecoveries: 0, @@ -228,6 +236,14 @@ export class RelayObservability implements RelayRuntimeObserver { (this.deltas.spliceClosesByTrigger[trigger] ?? 0) + 1 } + recordClientAcceptAbandoned(stage: RelayClientAcceptStage, elapsedMs: number): void { + increment(this.deltas.clientAcceptsAbandonedByStage, stage) + this.deltas.clientAcceptAbandonedMsMax = Math.max( + this.deltas.clientAcceptAbandonedMsMax, + elapsedMs + ) + } + start(readCounts: () => RelayProcessCounts, intervalMs = 30_000): void { if (this.timer) return this.eventLoop.enable() @@ -289,6 +305,8 @@ export class RelayObservability implements RelayRuntimeObserver { unavailableRegionsDelta: deltas.unavailableRegions, controlClosesByCodeDelta: deltas.controlClosesByCode, spliceClosesByTriggerDelta: deltas.spliceClosesByTrigger, + clientAcceptsAbandonedByStageDelta: deltas.clientAcceptsAbandonedByStage, + clientAcceptAbandonedMsMax: Number(deltas.clientAcceptAbandonedMsMax.toFixed(3)), sqlQueriesDelta: deltas.sqlQueries, sqlFailuresDelta: deltas.sqlFailures, sqlLatencyMsMax: Number(deltas.sqlLatencyMsMax.toFixed(3)), diff --git a/cloud/apps/relay/src/relay-server.ts b/cloud/apps/relay/src/relay-server.ts index 32a83962d81..6331b584b1b 100644 --- a/cloud/apps/relay/src/relay-server.ts +++ b/cloud/apps/relay/src/relay-server.ts @@ -88,6 +88,7 @@ export function createRelayServer( database: RelayDatabase, options: { now?: () => number + random?: () => number connectionLedgerLimits?: { hardCap: number; controlReserve: number } cellIncarnation?: string } = {} @@ -123,7 +124,8 @@ export function createRelayServer( assignments, queuedBytes, observability, - options.now + options.now, + options.random ) const app = createRelayApp(config, { store, diff --git a/cloud/dev/fixtures/terraform-root-partition/families.json b/cloud/dev/fixtures/terraform-root-partition/families.json index 9664c1eb299..dfe100fd2dd 100644 --- a/cloud/dev/fixtures/terraform-root-partition/families.json +++ b/cloud/dev/fixtures/terraform-root-partition/families.json @@ -133,10 +133,15 @@ "google_logging_metric.relay_snapshot", "google_monitoring_alert_policy.relay_assignment_5xx", "google_monitoring_alert_policy.relay_assignment_edge_429", + "google_monitoring_alert_policy.relay_cell_process_exit", + "google_monitoring_alert_policy.relay_cloud_nat_port_drops", "google_monitoring_alert_policy.relay_cloud_sql_backends", + "google_monitoring_alert_policy.relay_cloud_sql_checkpoint_loop", + "google_monitoring_alert_policy.relay_cloud_sql_disk", "google_monitoring_alert_policy.relay_custom", "google_monitoring_alert_policy.relay_gce_connection_headroom", "google_monitoring_alert_policy.relay_postgres_retry_exhausted", + "google_monitoring_dashboard.relay_incident", "google_project_iam_custom_role.github_production_relay_capacity_mutation", "google_project_iam_custom_role.github_relay_asia_topology_mutation", "google_project_iam_custom_role.github_relay_asia_topology_read", diff --git a/cloud/dev/scripts/operate-relay-regional-rehome.mjs b/cloud/dev/scripts/operate-relay-regional-rehome.mjs index 2ccce39926d..3887408520f 100644 --- a/cloud/dev/scripts/operate-relay-regional-rehome.mjs +++ b/cloud/dev/scripts/operate-relay-regional-rehome.mjs @@ -1,4 +1,5 @@ import { pathToFileURL } from 'node:url' +import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs' import { inspectAdmissionSelector } from './relay-admission-selector.mjs' const DIRECTOR_ORIGIN = 'https://relay.onorca.dev' @@ -229,15 +230,20 @@ export async function recoverRegionalRehomeEnable(config, post) { export async function operateRegionalRehome(config, dependencies = {}) { const fetchImpl = dependencies.fetch ?? fetch const post = dependencies.post ?? (async (path, body) => await responseJson( - await fetchImpl(`${config.directorOrigin}${path}`, { - method: 'POST', - headers: { - authorization: `Bearer ${config.token}`, - 'content-type': 'application/json' + // Generation-guarded writes make a retry a no-op or an explicit mismatch, never a double apply. + await fetchAdminOnceMore( + fetchImpl, + `${config.directorOrigin}${path}`, + { + method: 'POST', + headers: { + authorization: `Bearer ${config.token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify(body) }, - body: JSON.stringify(body), - signal: AbortSignal.timeout(30_000) - }), + { wait: dependencies.wait } + ), path )) if (config.mode === 'recover-enable') { diff --git a/cloud/dev/scripts/operate-relay-regional-rehome.test.mjs b/cloud/dev/scripts/operate-relay-regional-rehome.test.mjs index 8ffe38dfe09..bfea6769ec4 100644 --- a/cloud/dev/scripts/operate-relay-regional-rehome.test.mjs +++ b/cloud/dev/scripts/operate-relay-regional-rehome.test.mjs @@ -263,3 +263,55 @@ test('main executes recovery mode and emits verified disabled control', async () control: control(6, false) }) }) + +test('retries a transient 503 on the director control endpoint', async () => { + const config = parseRegionalRehomeArguments( + argumentsFor('inspect'), + { ORCA_RELAY_ADMIN_ID_TOKEN: 'token' } + ) + const paths = [] + let selectorCalls = 0 + const result = await operateRegionalRehome(config, { + wait: async () => {}, + fetch: async (url) => { + const path = new URL(url).pathname + paths.push(path) + if (path === '/v1/admin/admission-selector/status') { + selectorCalls += 1 + // The first read of each admin path 503s the way a warming instance does. + if (selectorCalls === 1) return new Response('warming up', { status: 503 }) + return Response.json({ selector: { generation: 11, membership } }) + } + if (paths.filter((value) => value === path).length === 1) { + return new Response('warming up', { status: 503 }) + } + return Response.json({ v: 1, control: control(4, false) }) + } + }) + assert.equal(result.control.generation, 4) + assert.deepEqual(paths, [ + '/v1/admin/admission-selector/status', + '/v1/admin/admission-selector/status', + '/v1/admin/regional-rehome-control', + '/v1/admin/regional-rehome-control' + ]) +}) + +test('fails when both attempts at the director control endpoint return 503', async () => { + const config = parseRegionalRehomeArguments( + argumentsFor('inspect'), + { ORCA_RELAY_ADMIN_ID_TOKEN: 'token' } + ) + let calls = 0 + await assert.rejects( + operateRegionalRehome(config, { + wait: async () => {}, + fetch: async () => { + calls += 1 + return new Response('warming up', { status: 503 }) + } + }), + /returned 503/ + ) + assert.equal(calls, 2) +}) diff --git a/cloud/dev/scripts/prepare-relay-production-capacity-canary.mjs b/cloud/dev/scripts/prepare-relay-production-capacity-canary.mjs index 5791c9f20e6..7967d164e4b 100644 --- a/cloud/dev/scripts/prepare-relay-production-capacity-canary.mjs +++ b/cloud/dev/scripts/prepare-relay-production-capacity-canary.mjs @@ -1,10 +1,12 @@ import { pathToFileURL } from 'node:url' +import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs' import { applyExactAdmissionSelector, inspectAdmissionSelector, membershipWithStates, selectorCellState } from './relay-admission-selector.mjs' +import { SAME_CAP_CELLS } from './relay-production-same-cap-wave.mjs' const DIRECTOR_ORIGIN = 'https://relay.onorca.dev' export const PRODUCTION_CAPACITY_CELL_IDS = [ @@ -30,6 +32,9 @@ function cellOrigin(cellId) { return `https://${cellId.slice('production-gce-'.length)}.relay.onorca.dev` } +// The same-cap roll covers the Asia cells the US-only capacity rollout never touches. +const APPROVED_CELL_LISTS = { 'same-cap': SAME_CAP_CELLS } + export function parseProductionCapacityCellArguments(argv) { const values = {} for (let index = 0; index < argv.length; index += 2) { @@ -41,8 +46,15 @@ export function parseProductionCapacityCellArguments(argv) { if (!['isolate', 'drain', 'activate'].includes(values.mode)) { throw new Error('--mode must be isolate, drain, or activate') } + const approvedList = values['approved-cells'] + if (approvedList !== undefined && !APPROVED_CELL_LISTS[approvedList]) { + throw new Error('--approved-cells is not a known allowlist') + } + const approvedCellIds = approvedList === undefined + ? PRODUCTION_CAPACITY_CELL_IDS + : APPROVED_CELL_LISTS[approvedList] const cellId = values['cell-id'] - if (!PRODUCTION_CAPACITY_CELL_IDS.includes(cellId)) { + if (!approvedCellIds.includes(cellId)) { throw new Error('production capacity target is not approved') } const expectedCellOrigin = cellOrigin(cellId) @@ -72,12 +84,16 @@ export async function prepareProductionCapacityCell(config, overrides = {}) { if (!token || token.length > 8_192) throw new Error('admin identity token is unavailable') const postAt = async (origin, path, body) => await responseJson( - await fetchImpl(`${origin}${path}`, { - method: 'POST', - headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, - body: JSON.stringify(body), - signal: AbortSignal.timeout(30_000) - }), + await fetchAdminOnceMore( + fetchImpl, + `${origin}${path}`, + { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify(body) + }, + { wait: overrides.wait } + ), path ) const post = async (path, body) => await postAt(config.directorOrigin, path, body) diff --git a/cloud/dev/scripts/prepare-relay-production-capacity-canary.test.mjs b/cloud/dev/scripts/prepare-relay-production-capacity-canary.test.mjs index c5d0a9db3bc..274a60d2198 100644 --- a/cloud/dev/scripts/prepare-relay-production-capacity-canary.test.mjs +++ b/cloud/dev/scripts/prepare-relay-production-capacity-canary.test.mjs @@ -104,6 +104,47 @@ describe('production Relay capacity cell admission', () => { '--cell-id', 'production-gce-c7', '--mode', 'isolate' ]), /origin is not exact/) + assert.throws(() => parseProductionCapacityCellArguments([ + '--director-origin', 'https://relay.onorca.dev', + '--cell-origin', 'https://c27.relay.onorca.dev', + '--cell-id', 'production-gce-c27', + '--mode', 'isolate' + ]), /not approved/) + }) + + it('admits the same-cap Asia cells only under the same-cap allowlist', () => { + for (const cellId of ['production-gce-c27', 'production-gce-c28', 'production-gce-c29']) { + const hostname = cellId.slice('production-gce-'.length) + assert.deepEqual(parseProductionCapacityCellArguments([ + '--director-origin', 'https://relay.onorca.dev', + '--cell-origin', `https://${hostname}.relay.onorca.dev`, + '--cell-id', cellId, + '--approved-cells', 'same-cap', + '--mode', 'isolate' + ]), { + directorOrigin: 'https://relay.onorca.dev', + cellOrigin: `https://${hostname}.relay.onorca.dev`, + cellId, + mode: 'isolate' + }) + } + for (const cellId of ['production-gce-c17', 'production-gce-c18', 'production-gce-c30']) { + const hostname = cellId.slice('production-gce-'.length) + assert.throws(() => parseProductionCapacityCellArguments([ + '--director-origin', 'https://relay.onorca.dev', + '--cell-origin', `https://${hostname}.relay.onorca.dev`, + '--cell-id', cellId, + '--approved-cells', 'same-cap', + '--mode', 'isolate' + ]), /not approved/) + } + assert.throws(() => parseProductionCapacityCellArguments([ + '--director-origin', 'https://relay.onorca.dev', + '--cell-origin', 'https://c27.relay.onorca.dev', + '--cell-id', 'production-gce-c27', + '--approved-cells', 'every-cell', + '--mode', 'isolate' + ]), /not a known allowlist/) }) it('isolates only the selected cell without depending on its runtime', async () => { @@ -170,4 +211,42 @@ describe('production Relay capacity cell admission', () => { /irreversible/ ) }) + + it('retries a transient 503 on the cell drain endpoint', async () => { + let calls = 0 + const result = await prepareProductionCapacityCell( + { ...config, mode: 'drain' }, + { + token: 'token', + wait: async () => {}, + fetch: async (url) => { + assert.equal(new URL(url).pathname, '/v1/admin/drain') + calls += 1 + if (calls === 1) return response({ error: 'warming up' }, 503) + return response({ v: 1, draining: true }) + } + } + ) + assert.equal(calls, 2) + assert.deepEqual(result, { changed: false, drained: true }) + }) + + it('fails when both drain attempts return a transient 503', async () => { + let calls = 0 + await assert.rejects( + prepareProductionCapacityCell( + { ...config, mode: 'drain' }, + { + token: 'token', + wait: async () => {}, + fetch: async () => { + calls += 1 + return response({ error: 'warming up' }, 503) + } + } + ), + /returned 503/ + ) + assert.equal(calls, 2) + }) }) diff --git a/cloud/dev/scripts/probe-relay-rehome-trust.mjs b/cloud/dev/scripts/probe-relay-rehome-trust.mjs index 7500d8bd14c..9a7d505d4bb 100644 --- a/cloud/dev/scripts/probe-relay-rehome-trust.mjs +++ b/cloud/dev/scripts/probe-relay-rehome-trust.mjs @@ -1,4 +1,5 @@ import { pathToFileURL } from 'node:url' +import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs' const PRODUCTION_CELL = /^production-gce-c(?:7|8|9|10|13|14|15|16|19|20|21|22|23|24|25|26)$/ const DIRECTOR_ORIGIN = 'https://relay.onorca.dev' @@ -35,7 +36,8 @@ export function parseRehomeTrustProbeArguments(argv, environment = process.env) export async function probeRehomeTrust(config, dependencies = {}) { const fetchImpl = dependencies.fetch ?? fetch - const response = await fetchImpl( + const response = await fetchAdminOnceMore( + fetchImpl, `${config.directorOrigin}/v1/admin/regional-rehome-trust-probe`, { method: 'POST', @@ -47,9 +49,9 @@ export async function probeRehomeTrust(config, dependencies = {}) { v: 1, sourceCellId: config.cellId, sourceCellIncarnation: config.cellIncarnation - }), - signal: AbortSignal.timeout(30_000) - } + }) + }, + { wait: dependencies.wait } ) const body = await response.json().catch(() => ({})) if (!response.ok) { diff --git a/cloud/dev/scripts/probe-relay-rehome-trust.test.mjs b/cloud/dev/scripts/probe-relay-rehome-trust.test.mjs index 509e9d53c7d..7d7b2cd95ac 100644 --- a/cloud/dev/scripts/probe-relay-rehome-trust.test.mjs +++ b/cloud/dev/scripts/probe-relay-rehome-trust.test.mjs @@ -68,3 +68,46 @@ test('rejects partial or mismatched proof', async () => { /incomplete/ ) }) + +const provenProbe = { + v: 1, + dedicatedIdentity: { + firstOutcome: 'host-not-connected', + secondOutcome: 'host-not-connected', + accepted: true, + idempotent: true + }, + sharedRuntimeIdentityRejected: true, + proven: true +} + +test('retries a transient 503 on the trust probe and proves on the second answer', async () => { + const config = parseRehomeTrustProbeArguments(argv, environment) + let calls = 0 + const result = await probeRehomeTrust(config, { + wait: async () => {}, + fetch: async () => { + calls += 1 + if (calls === 1) return new Response('warming up', { status: 503 }) + return Response.json(provenProbe) + } + }) + assert.equal(calls, 2) + assert.equal(result.proven, true) +}) + +test('fails when both trust-probe attempts return a transient 503', async () => { + const config = parseRehomeTrustProbeArguments(argv, environment) + let calls = 0 + await assert.rejects( + probeRehomeTrust(config, { + wait: async () => {}, + fetch: async () => { + calls += 1 + return new Response('warming up', { status: 503 }) + } + }), + /returned 503/ + ) + assert.equal(calls, 2) +}) diff --git a/cloud/dev/scripts/relay-admin-endpoint-retry-workflow.test.mjs b/cloud/dev/scripts/relay-admin-endpoint-retry-workflow.test.mjs new file mode 100644 index 00000000000..196fff9edf3 --- /dev/null +++ b/cloud/dev/scripts/relay-admin-endpoint-retry-workflow.test.mjs @@ -0,0 +1,43 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import { test } from 'node:test' +import { fileURLToPath } from 'node:url' +import { relayWorkflowUrl } from './relay-repository.mjs' + +const WORKFLOWS = [ + 'deploy-relay-production-same-cap-job.yml', + 'operate-relay-production-rehome-job.yml' +] + +function workflow(name) { + return readFileSync(fileURLToPath(relayWorkflowUrl(name)), 'utf8') +} + +// A single transient 5xx from a warming instance behind the global load balancer +// must not fail a canary, so no admin endpoint may be read by a bare curl. +test('no admin endpoint is reached by a curl without a bounded retry', () => { + for (const name of WORKFLOWS) { + for (const invocation of workflow(name).split(/\bcurl\b/).slice(1)) { + const flags = invocation.split('\n }')[0] + assert.match(flags, /--retry 3 --retry-delay 2 --retry-connrefused/, name) + assert.match(flags, /--max-time 30/, name) + // --retry-all-errors would also retry 401, 403, and 409, which are final. + assert.doesNotMatch(flags, /--retry-all-errors/, name) + } + } +}) + +test('every retried admin request captures only the final attempt body', () => { + const job = workflow('deploy-relay-production-same-cap-job.yml') + // --fail-with-body writes every failed attempt to stdout, so a retried + // request must land in a file curl truncates per attempt. + assert.match(job, /--output "\$\{out\}"/) + assert.equal(job.split('admin_post() {').length - 1, 2) + for (const call of [ + /CURRENT_RUNTIME="\$\(admin_post current-runtime/, + /CURRENT_DIRECTOR_STATUS="\$\(admin_post current-cell-status/, + /TARGET_RUNTIME="\$\(admin_post target-runtime/, + /TARGET_DIRECTOR_STATUS="\$\(admin_post target-cell-status/ + ]) assert.match(job, call) + assert.doesNotMatch(job, /\$\(curl /) +}) diff --git a/cloud/dev/scripts/relay-admin-transient-retry.mjs b/cloud/dev/scripts/relay-admin-transient-retry.mjs new file mode 100644 index 00000000000..9995be96a8f --- /dev/null +++ b/cloud/dev/scripts/relay-admin-transient-retry.mjs @@ -0,0 +1,29 @@ +// A single transient 5xx (load-balancer warm-up behind a fresh instance) must not fail a +// deploy step. 4xx is never retried: auth and generation-mismatch answers are final. +const TRANSIENT_STATUSES = [500, 502, 503, 504] +const RETRY_DELAY_MS = 2_000 +const REQUEST_TIMEOUT_MS = 30_000 + +export function isTransientAdminStatus(status) { + return TRANSIENT_STATUSES.includes(status) +} + +// Each attempt gets its own timeout budget, so a reused signal cannot abort the retry. +export async function fetchAdminOnceMore(fetchImpl, url, init, overrides = {}) { + const wait = overrides.wait ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms))) + const timeoutMs = overrides.timeoutMs ?? REQUEST_TIMEOUT_MS + const retryDelayMs = overrides.retryDelayMs ?? RETRY_DELAY_MS + const attempt = async () => + await fetchImpl(url, { ...init, signal: AbortSignal.timeout(timeoutMs) }) + let response + try { + response = await attempt() + } catch { + await wait(retryDelayMs) + return await attempt() + } + if (!isTransientAdminStatus(response.status)) return response + await response.arrayBuffer?.().catch(() => undefined) + await wait(retryDelayMs) + return await attempt() +} diff --git a/cloud/dev/scripts/relay-admin-transient-retry.test.mjs b/cloud/dev/scripts/relay-admin-transient-retry.test.mjs new file mode 100644 index 00000000000..ec041084344 --- /dev/null +++ b/cloud/dev/scripts/relay-admin-transient-retry.test.mjs @@ -0,0 +1,130 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs' + +const url = 'https://relay.onorca.dev/v1/admin/cell-status' +const init = { method: 'POST', body: '{"v":1}' } + +function recordingWait(waits) { + return async (ms) => { waits.push(ms) } +} + +test('a single transient 5xx is retried and the second answer is returned', async () => { + const waits = [] + const statuses = [503, 200] + let calls = 0 + const response = await fetchAdminOnceMore( + async () => { + calls += 1 + const status = statuses.shift() + return new Response(JSON.stringify({ ok: status === 200 }), { status }) + }, + url, + init, + { wait: recordingWait(waits) } + ) + assert.equal(calls, 2) + assert.equal(response.status, 200) + assert.deepEqual(waits, [2_000]) + assert.deepEqual(await response.json(), { ok: true }) +}) + +test('a connection failure is retried and the second answer is returned', async () => { + const waits = [] + let calls = 0 + const response = await fetchAdminOnceMore( + async () => { + calls += 1 + if (calls === 1) throw new TypeError('fetch failed') + return Response.json({ ok: true }) + }, + url, + init, + { wait: recordingWait(waits) } + ) + assert.equal(calls, 2) + assert.equal(response.status, 200) + assert.deepEqual(waits, [2_000]) +}) + +test('two transient failures surface the second answer without a third attempt', async () => { + let calls = 0 + const response = await fetchAdminOnceMore( + async () => { + calls += 1 + return new Response('down', { status: 503 }) + }, + url, + init, + { wait: async () => {} } + ) + assert.equal(calls, 2) + assert.equal(response.status, 503) +}) + +test('two connection failures rethrow the second error', async () => { + let calls = 0 + await assert.rejects( + fetchAdminOnceMore( + async () => { + calls += 1 + throw new TypeError(`fetch failed ${calls}`) + }, + url, + init, + { wait: async () => {} } + ), + /fetch failed 2/ + ) + assert.equal(calls, 2) +}) + +test('4xx is final: auth and generation-mismatch answers are never retried', async () => { + for (const status of [400, 401, 403, 404, 409, 429]) { + let calls = 0 + const response = await fetchAdminOnceMore( + async () => { + calls += 1 + return new Response('no', { status }) + }, + url, + init, + { wait: async () => { throw new Error('must not wait') } } + ) + assert.equal(calls, 1, `status ${status} must not be retried`) + assert.equal(response.status, status) + } +}) + +test('each attempt carries its own unexpired timeout signal', async () => { + const signals = [] + await fetchAdminOnceMore( + async (_url, attemptInit) => { + signals.push(attemptInit.signal) + return new Response('down', { status: 502 }) + }, + url, + init, + { wait: async () => {}, timeoutMs: 30_000 } + ) + assert.equal(signals.length, 2) + assert.notEqual(signals[0], signals[1]) + assert.equal(signals[1].aborted, false) +}) + +test('the caller init is forwarded unchanged apart from the signal', async () => { + let seen + await fetchAdminOnceMore( + async (seenUrl, attemptInit) => { + seen = { seenUrl, attemptInit } + return Response.json({}) + }, + url, + { method: 'POST', headers: { authorization: 'Bearer t' }, body: '{"v":1}' }, + { wait: async () => {} } + ) + assert.equal(seen.seenUrl, url) + assert.equal(seen.attemptInit.method, 'POST') + assert.deepEqual(seen.attemptInit.headers, { authorization: 'Bearer t' }) + assert.equal(seen.attemptInit.body, '{"v":1}') +}) diff --git a/cloud/dev/scripts/relay-evidence-code-provenance.mjs b/cloud/dev/scripts/relay-evidence-code-provenance.mjs new file mode 100644 index 00000000000..233a8139b85 --- /dev/null +++ b/cloud/dev/scripts/relay-evidence-code-provenance.mjs @@ -0,0 +1,94 @@ +import { spawnSync } from 'node:child_process' +import { fileURLToPath } from 'node:url' +import { + RELAY_REPOSITORY_ROOT, + relayTreePath, + relayWorkflowPath +} from './relay-repository.mjs' + +const SHA = /^[a-f0-9]{40}$/ + +// Every file that decides how relay evidence is produced, sealed, verified, and then spent against +// production; identical content across two commits is what makes the older commit's verdict binding. +export const TRUSTED_EVIDENCE_CODE_PATHS = [ + // Produces and seals the 15-minute dry-run evidence. + relayWorkflowPath('monitor-relay-production.yml'), + relayWorkflowPath('monitor-relay-production-job.yml'), + // Download it, verify its authority, and mutate production on it. + relayWorkflowPath('deploy-relay-production-same-cap.yml'), + relayWorkflowPath('deploy-relay-production-same-cap-job.yml'), + relayWorkflowPath('operate-relay-production-rehome.yml'), + relayWorkflowPath('operate-relay-production-rehome-job.yml'), + // Sealing, verification, the wave/canary authority, and the path constants below. + relayTreePath('dev/scripts/relay-evidence-code-provenance.mjs'), + relayTreePath('dev/scripts/relay-monitor-evidence.mjs'), + relayTreePath('dev/scripts/relay-production-same-cap-wave.mjs'), + relayTreePath('dev/scripts/relay-repository.mjs'), + // Every other script those jobs run against live production. + relayTreePath('dev/scripts/infra.mjs'), + relayTreePath('dev/scripts/operate-relay-regional-rehome.mjs'), + relayTreePath('dev/scripts/prepare-relay-production-capacity-canary.mjs'), + relayTreePath('dev/scripts/probe-relay-rehome-trust.mjs'), + relayTreePath('dev/scripts/validate-relay-capacity-plan.mjs'), + relayTreePath('dev/scripts/verify-relay-capacity-transition.mjs'), + // The monitor itself and the live preflight recheck, plus anything that changes their behaviour. + relayTreePath('apps/relay-ops'), + relayTreePath('package.json'), + relayTreePath('pnpm-lock.yaml'), + relayTreePath('pnpm-workspace.yaml'), + // The Cloud SQL rollout lease every mutation job takes and releases. + '.github/actions/cloud-sql-rollout-lease' +] + +function git(root, args) { + const result = spawnSync('git', ['-C', root, ...args], { encoding: 'utf8' }) + if (result.error) throw new Error('relay evidence provenance cannot run git') + return result +} + +/** + * Accepts evidence sealed at a different commit only when the current commit descends from it and + * every trusted path is byte-identical, so the verdict provably came from this exact code. Anything + * git cannot answer (no checkout, unknown commit, shallow clone) fails closed. + */ +export function requireSameEvidenceCode({ + sealedSha, + currentSha, + label, + repositoryRoot = fileURLToPath(RELAY_REPOSITORY_ROOT) +}) { + if (!SHA.test(sealedSha ?? '') || !SHA.test(currentSha ?? '')) { + throw new Error(`${label} commit is invalid`) + } + if (sealedSha === currentSha) return + if (git(repositoryRoot, ['rev-parse', '--git-dir']).status !== 0) { + throw new Error(`${label} commit cannot be compared without a git checkout`) + } + for (const sha of [sealedSha, currentSha]) { + if (git(repositoryRoot, ['rev-parse', '--verify', '--quiet', `${sha}^{commit}`]).status !== 0) { + throw new Error( + `${label} commit ${sha} is unknown to this checkout; check out with fetch-depth: 0` + ) + } + } + const ancestry = git(repositoryRoot, ['merge-base', '--is-ancestor', sealedSha, currentSha]) + if (ancestry.status === 1) { + throw new Error(`${label} commit ${sealedSha} is not an ancestor of ${currentSha}`) + } + if (ancestry.status !== 0) { + throw new Error(`${label} commit ancestry could not be determined`) + } + const diff = git(repositoryRoot, [ + 'diff', + '--name-only', + sealedSha, + currentSha, + '--', + ...TRUSTED_EVIDENCE_CODE_PATHS + ]) + if (diff.status !== 0) throw new Error(`${label} commit comparison failed`) + const changed = diff.stdout.split('\n').filter(Boolean) + if (changed.length > 0) { + throw new Error(`${label} code changed after it was sealed: ${changed.join(',')}`) + } +} diff --git a/cloud/dev/scripts/relay-monitor-evidence.mjs b/cloud/dev/scripts/relay-monitor-evidence.mjs index 7f387663f60..26eb37d0d4d 100644 --- a/cloud/dev/scripts/relay-monitor-evidence.mjs +++ b/cloud/dev/scripts/relay-monitor-evidence.mjs @@ -2,6 +2,7 @@ import { createHash } from 'node:crypto' import { chmod, readFile, readdir, stat, writeFile } from 'node:fs/promises' import { basename, join, resolve } from 'node:path' import { pathToFileURL } from 'node:url' +import { requireSameEvidenceCode } from './relay-evidence-code-provenance.mjs' const SAFE_ID = /^[A-Za-z0-9][A-Za-z0-9._-]{1,127}$/ const SHA = /^[a-f0-9]{40}$/ @@ -102,7 +103,7 @@ export async function createEvidenceManifest(argv) { return manifest } -async function readAndVerifyManifest(directory, expected) { +async function readAndVerifyManifest(directory, expected, sameCodeCommit) { const manifest = JSON.parse( await readFile(join(directory, 'evidence-manifest.json'), 'utf8') ) @@ -111,11 +112,23 @@ async function readAndVerifyManifest(directory, expected) { manifest.incidentId !== expected.incidentId || manifest.runId !== expected.runId || manifest.runAttempt !== expected.runAttempt || - manifest.commitSha !== expected.commitSha || - manifest.mode !== expected.mode + !SHA.test(manifest.commitSha ?? '') || + manifest.mode !== expected.mode || + (!sameCodeCommit && manifest.commitSha !== expected.commitSha) ) { throw new Error('relay monitor evidence provenance does not match') } + // Unrelated merges land on main every few minutes, so the deployer resolves a newer commit than + // the monitor it must trust; identical monitor and mutation code is the property the SHA stood in + // for. Restore and mutation keep the exact-SHA bind: both run at the commit that sealed them. + if (sameCodeCommit) { + requireSameEvidenceCode({ + sealedSha: manifest.commitSha, + currentSha: expected.commitSha, + label: 'relay monitor evidence', + ...sameCodeCommit + }) + } const names = Object.keys(manifest.files ?? {}) if (!names.includes(`${expected.incidentId}.state.json`)) { throw new Error('relay monitor evidence has no durable state') @@ -209,12 +222,12 @@ function validCompletedDryRunState(state, expected, nowMs, maxAgeMs) { ) } -export async function verifyDryRunAuthority(argv, now = Date.now) { +export async function verifyDryRunAuthority(argv, now = Date.now, repositoryRoot) { const values = argumentsByName(argv) const directory = resolve(values.directory ?? '') const expected = provenance(values) if (expected.mode !== 'dry-run') throw new Error('relay mutation requires dry-run evidence') - const manifest = await readAndVerifyManifest(directory, expected) + const manifest = await readAndVerifyManifest(directory, expected, { repositoryRoot }) const state = JSON.parse( await readFile(join(directory, `${expected.incidentId}.state.json`), 'utf8') ) diff --git a/cloud/dev/scripts/relay-monitor-evidence.test.mjs b/cloud/dev/scripts/relay-monitor-evidence.test.mjs index 45116761119..43d2ac02763 100644 --- a/cloud/dev/scripts/relay-monitor-evidence.test.mjs +++ b/cloud/dev/scripts/relay-monitor-evidence.test.mjs @@ -1,9 +1,15 @@ import assert from 'node:assert/strict' -import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises' +import { execFileSync } from 'node:child_process' +import { mkdir, mkdtemp, readFile, rm, stat, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' -import { join } from 'node:path' +import { dirname, join } from 'node:path' import test from 'node:test' -import { relayWorkflowPath, relayWorkflowUrl } from './relay-repository.mjs' +import { TRUSTED_EVIDENCE_CODE_PATHS } from './relay-evidence-code-provenance.mjs' +import { + RELAY_REPOSITORY_ROOT, + relayWorkflowPath, + relayWorkflowUrl +} from './relay-repository.mjs' import { createEvidenceManifest, verifyDryRunAuthority, @@ -12,7 +18,7 @@ import { } from './relay-monitor-evidence.mjs' const now = Date.parse('2026-07-28T12:00:00.000Z') -const provenance = [ +const provenanceFor = (commitSha) => [ '--incident-id', 'relay-123', '--run-id', @@ -20,10 +26,11 @@ const provenance = [ '--run-attempt', '1', '--commit-sha', - 'a'.repeat(40), + commitSha, '--mode', 'dry-run' ] +const provenance = provenanceFor('a'.repeat(40)) const selector = { generation: 2, membership: { @@ -513,3 +520,157 @@ test('monitor uses a reusable job so exact job_workflow_ref is present', async ( assert.match(job, /workflow_call:/) assert.match(job, /environment: production/) }) + +function gitIn(root, ...args) { + return execFileSync('git', ['-C', root, ...args], { encoding: 'utf8' }).trim() +} + +// A real repository shaped like main under unrelated merge traffic: one sealed commit, a +// descendant that only touched untrusted files, a descendant that touched the monitor, and a +// sibling that never descended from the seal. +async function trustedCodeRepository() { + const root = await mkdtemp(join(tmpdir(), 'relay-evidence-repository-')) + gitIn(root, 'init', '--quiet') + gitIn(root, 'config', 'user.email', 'relay@example.test') + gitIn(root, 'config', 'user.name', 'Relay Evidence Test') + gitIn(root, 'config', 'commit.gpgsign', 'false') + const commit = async (path, body, message) => { + await mkdir(dirname(join(root, path)), { recursive: true }) + await writeFile(join(root, path), body) + gitIn(root, 'add', '--all') + gitIn(root, 'commit', '--quiet', '--no-verify', '--message', message) + return gitIn(root, 'rev-parse', 'HEAD') + } + const base = await commit( + 'cloud/apps/relay-ops/src/incident-monitor.ts', + 'export const v = 1\n', + 'monitor' + ) + const sealed = await commit('README.md', 'base\n', 'base') + const sameCode = await commit('README.md', 'an unrelated merge\n', 'unrelated') + const changedCode = await commit( + 'cloud/apps/relay-ops/src/incident-monitor.ts', + 'export const v = 2\n', + 'monitor change' + ) + // Branches before the seal, so the seal is not in its history even though its code matches. + gitIn(root, 'checkout', '--quiet', '--detach', base) + const sibling = await commit('README.md', 'a divergent line\n', 'divergent') + return { root, sealed, sameCode, changedCode, sibling } +} + +const authorityAt = (directory, commitSha, repositoryRoot) => verifyDryRunAuthority( + [ + '--directory', + directory, + ...provenanceFor(commitSha), + '--required-migration-policy', + 'strict' + ], + () => now, + repositoryRoot +) + +test('accepts dry-run evidence sealed by identical code at an ancestor commit', async () => { + const repository = await trustedCodeRepository() + const directory = await evidenceDirectory() + try { + await createEvidenceManifest([ + '--directory', + directory, + ...provenanceFor(repository.sealed) + ]) + // An exact match never consults git: a root with no checkout at all still verifies. + await assert.doesNotReject(authorityAt(directory, repository.sealed, directory)) + await assert.doesNotReject(authorityAt(directory, repository.sameCode, repository.root)) + } finally { + await rm(repository.root, { recursive: true, force: true }) + await rm(directory, { recursive: true, force: true }) + } +}) + +test('rejects dry-run evidence whose monitor code or lineage differs', async () => { + const repository = await trustedCodeRepository() + const directory = await evidenceDirectory() + try { + await createEvidenceManifest([ + '--directory', + directory, + ...provenanceFor(repository.sealed) + ]) + await assert.rejects( + authorityAt(directory, repository.changedCode, repository.root), + /code changed after it was sealed: cloud\/apps\/relay-ops\/src\/incident-monitor\.ts/ + ) + await assert.rejects( + authorityAt(directory, repository.sibling, repository.root), + /is not an ancestor of/ + ) + // Fails closed: a shallow clone that never fetched the sealed commit proves nothing. + await assert.rejects( + authorityAt(directory, 'f'.repeat(40), repository.root), + /unknown to this checkout/ + ) + // Fails closed: no checkout to compare against. + await assert.rejects( + authorityAt(directory, repository.sameCode, directory), + /cannot be compared without a git checkout/ + ) + } finally { + await rm(repository.root, { recursive: true, force: true }) + await rm(directory, { recursive: true, force: true }) + } +}) + +test('keeps restore and mutation bound to the exact sealing commit', async () => { + const repository = await trustedCodeRepository() + const directory = await evidenceDirectory() + try { + await createEvidenceManifest([ + '--directory', + directory, + ...provenanceFor(repository.sealed) + ]) + await assert.rejects( + verifyRestoredEvidence([ + '--directory', + directory, + ...provenanceFor(repository.sameCode) + ]), + /provenance does not match/ + ) + await assert.rejects( + verifyMutationEvidence( + [ + '--directory', + directory, + ...provenanceFor(repository.sameCode), + '--mutation-mode', + 'execute', + '--source-cell-id', + 'c1', + '--director-origin', + 'https://relay.example' + ], + { ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }, + async () => Response.json({ selector }), + () => now + ), + /provenance does not match/ + ) + } finally { + await rm(repository.root, { recursive: true, force: true }) + await rm(directory, { recursive: true, force: true }) + } +}) + +// A trusted path that no longer exists silently stops being compared, so the same-code rule would +// pass over code it was written to pin. +test('every trusted provenance path exists in this checkout', async () => { + for (const path of TRUSTED_EVIDENCE_CODE_PATHS) { + await assert.doesNotReject( + stat(new URL(path, RELAY_REPOSITORY_ROOT)), + `${path} is missing` + ) + } +}) diff --git a/cloud/dev/scripts/relay-production-same-cap-wave.mjs b/cloud/dev/scripts/relay-production-same-cap-wave.mjs index e391c4c4381..6e84c1c9104 100644 --- a/cloud/dev/scripts/relay-production-same-cap-wave.mjs +++ b/cloud/dev/scripts/relay-production-same-cap-wave.mjs @@ -1,5 +1,6 @@ import { readFileSync } from 'node:fs' import { pathToFileURL } from 'node:url' +import { requireSameEvidenceCode } from './relay-evidence-code-provenance.mjs' export const SAME_CAP_CELLS = [ 'production-gce-c7', 'production-gce-c8', 'production-gce-c9', 'production-gce-c10', @@ -85,10 +86,10 @@ export function canaryAuthority(input) { } } -export function verifyCanaryAuthority(authority, expected) { +export function verifyCanaryAuthority(authority, expected, repositoryRoot) { if ( authority?.v !== 1 || - authority.commitSha !== expected.commitSha || + !/^[0-9a-f]{40}$/.test(authority.commitSha ?? '') || authority.runId !== expected.runId || authority.targetDigest !== expected.targetDigest || authority.rollbackDigest !== expected.rollbackDigest || @@ -96,6 +97,14 @@ export function verifyCanaryAuthority(authority, expected) { authority.rehomeGeneration !== Number(expected.rehomeGeneration) || !SAME_CAP_CELLS.includes(authority.cellId) ) throw new Error('canary authority does not match this batch') + // The batch dispatch resolves main after the canary sealed, so bind to the same code, not the + // same SHA; every field above still pins this batch to that exact canary. + requireSameEvidenceCode({ + sealedSha: authority.commitSha, + currentSha: expected.commitSha, + label: 'relay same-cap canary authority', + repositoryRoot + }) return authority } diff --git a/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs b/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs index 0b45ae85a99..d636c324b33 100644 --- a/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs +++ b/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs @@ -1,4 +1,8 @@ import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { dirname, join } from 'node:path' import { test } from 'node:test' import { canaryAuthority, @@ -104,3 +108,66 @@ test('seals and verifies canary authority for later batches', () => { rehomeGeneration: '4' }), /does not match/) }) + +function gitIn(root, ...args) { + return execFileSync('git', ['-C', root, ...args], { encoding: 'utf8' }).trim() +} + +async function canaryRepository() { + const root = await mkdtemp(join(tmpdir(), 'relay-same-cap-canary-')) + gitIn(root, 'init', '--quiet') + gitIn(root, 'config', 'user.email', 'relay@example.test') + gitIn(root, 'config', 'user.name', 'Relay Wave Test') + gitIn(root, 'config', 'commit.gpgsign', 'false') + const commit = async (path, body, message) => { + await mkdir(dirname(join(root, path)), { recursive: true }) + await writeFile(join(root, path), body) + gitIn(root, 'add', '--all') + gitIn(root, 'commit', '--quiet', '--no-verify', '--message', message) + return gitIn(root, 'rev-parse', 'HEAD') + } + const sealed = await commit( + 'cloud/dev/scripts/relay-production-same-cap-wave.mjs', + 'export const v = 1\n', + 'wave' + ) + const sameCode = await commit('README.md', 'an unrelated merge\n', 'unrelated') + const changedCode = await commit( + 'cloud/dev/scripts/relay-production-same-cap-wave.mjs', + 'export const v = 2\n', + 'wave change' + ) + return { root, sealed, sameCode, changedCode } +} + +test('a batch trusts a canary sealed by identical code at an ancestor commit', async () => { + const repository = await canaryRepository() + try { + const authority = canaryAuthority({ + cellIds: 'production-gce-c7', + targetDigest, + rollbackDigest, + confirmation: `ROLL_RELAY_SAME_CAP ${targetDigest} production-gce-c7`, + commitSha: repository.sealed, + runId: '42', + selectorGeneration: '11', + rehomeGeneration: '4' + }) + const verifyAt = (commitSha, repositoryRoot) => verifyCanaryAuthority(authority, { + commitSha, + runId: '42', + targetDigest, + rollbackDigest, + selectorGeneration: '13', + rehomeGeneration: '4' + }, repositoryRoot) + assert.equal(verifyAt(repository.sameCode, repository.root).cellId, 'production-gce-c7') + assert.throws( + () => verifyAt(repository.changedCode, repository.root), + /code changed after it was sealed/ + ) + assert.throws(() => verifyAt('f'.repeat(40), repository.root), /unknown to this checkout/) + } finally { + await rm(repository.root, { recursive: true, force: true }) + } +}) diff --git a/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs b/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs index e31403f6dd6..a77ab93cf15 100644 --- a/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs +++ b/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs @@ -73,7 +73,10 @@ test('same-cap wrapper is reusable, canary-bound, and sequential', () => { job, /--rollback-image "\$\{DESIRED_IMAGE\}" \\\n {16}--rehome-director-service-account "\$\{DIRECTOR_RUNTIME_SERVICE_ACCOUNT\}"/ ) - assert.match(job, /host-drain \\\n {14}\| jq -e '\.changes == 2' >\/dev\/null/) + assert.match( + job, + /host-drain \\\n {16}--regional-rehome-protocol "\$\{DESIRED_REHOME_PROTOCOL\}" \\\n {14}\| jq -e '\.changes == 2' >\/dev\/null/ + ) assert.match(job, /resume requires the isolated migration-only cell/) assert.match(job, /test "\$\{TARGET_INCARNATION\}" = "\$\{SOURCE_INCARNATION\}"/) assert.match(job, /\(.regionalRehomeProtocol \/\/ 0\) == \$protocol/) @@ -92,7 +95,11 @@ test('same-cap wrapper is reusable, canary-bound, and sequential', () => { // age checks must scale by wave or cell_2+ can never pass; the bound's // per-wave step is the cell job timeout, so the two must move together. assert.match(job, /--required-migration-policy strict \\\n --wave-index "\$\{WAVE_INDEX\}"/) - assert.match(job, /dry-run\.state\.json" \\\n --wave-index "\$\{WAVE_INDEX\}" "\$\{RETRY_ARGS\[@\]\}"/) + // Wave 0 must retry freshness-only failures too: one Cloud Monitoring publish + // lag at the sample instant is not health evidence, and single-shot wave 0 + // failed a whole batch on a series that was fresh again a minute later. + assert.match(job, /dry-run\.state\.json" \\\n --wave-index "\$\{WAVE_INDEX\}" --retry-freshness/) + assert.doesNotMatch(job, /RETRY_ARGS/) assert.match(job, /timeout-minutes: 75/) // Both age gates step by the cell job timeout above; the constant is // duplicated across the two languages, so pin each copy to it. diff --git a/cloud/dev/scripts/relay-repository.mjs b/cloud/dev/scripts/relay-repository.mjs index 7e8b01e4799..040acef5fe5 100644 --- a/cloud/dev/scripts/relay-repository.mjs +++ b/cloud/dev/scripts/relay-repository.mjs @@ -1,4 +1,6 @@ import { readFileSync } from 'node:fs' +import { relative } from 'node:path' +import { fileURLToPath } from 'node:url' // Single place naming the repository the Relay workflows live in and where their files sit. The // public-repo copy moves this tree under cloud/, prefixes every workflow filename, and changes the @@ -11,6 +13,19 @@ export const RELAY_WORKFLOW_FILE_PREFIX = 'cloud-' // this tree moves under cloud/, so the depth changes at the copy even though the layout does not. export const RELAY_WORKFLOW_DIRECTORY = new URL('../../../.github/workflows/', import.meta.url) +// Repository root, derived from the one directory above that already tracks the copy's depth. +export const RELAY_REPOSITORY_ROOT = new URL('../../', RELAY_WORKFLOW_DIRECTORY) + +// Repository-relative path for a file in this tree. The prefix is 'cloud/' here and empty where +// the tree is the repository root, so callers naming git paths never restate the layout. +export function relayTreePath(suffix) { + const prefix = relative( + fileURLToPath(RELAY_REPOSITORY_ROOT), + fileURLToPath(new URL('../../', import.meta.url)) + ).split(/[\\/]/).filter(Boolean) + return [...prefix, suffix].join('/') +} + export function relayWorkflowFile(name) { return `${RELAY_WORKFLOW_FILE_PREFIX}${name}` } diff --git a/cloud/dev/scripts/relay-same-cap-script-census.test.mjs b/cloud/dev/scripts/relay-same-cap-script-census.test.mjs new file mode 100644 index 00000000000..743aef7fc2d --- /dev/null +++ b/cloud/dev/scripts/relay-same-cap-script-census.test.mjs @@ -0,0 +1,217 @@ +import assert from 'node:assert/strict' +import { spawnSync } from 'node:child_process' +import { readFileSync } from 'node:fs' +import { describe, it } from 'node:test' +import { parseProductionCapacityCellArguments } from './prepare-relay-production-capacity-canary.mjs' +import { SAME_CAP_CELLS } from './relay-production-same-cap-wave.mjs' +import { readRelayWorkflow } from './relay-repository.mjs' +import { validateCapacityPlan } from './validate-relay-capacity-plan.mjs' + +const workflow = readRelayWorkflow('deploy-relay-production-same-cap-job.yml') +const capacityWorkflow = readRelayWorkflow('deploy-relay-production-capacity-job.yml') +const production = readFileSync( + new URL('../../infra/terraform/environments/production.tfvars', import.meta.url), + 'utf8' +) +const REHOME_SOURCE_CELLS = rehomeSourceCells() +const DIRECTOR_IDENTITY = 'relay-director@onorca-cloud.iam.gserviceaccount.com' +const AUDIENCE = 'https://relay.onorca.dev/v1/admin/host-drain' +const ROLLBACK_IMAGE = `us-central1-docker.pkg.dev/p/orca-cloud/relay@sha256:${'d'.repeat(64)}` +const TARGET_IMAGE = `us-central1-docker.pkg.dev/p/orca-cloud/relay@sha256:${'e'.repeat(64)}` + +// The startup template emits rehome trust only for cells in this list, so it is what decides +// whether a cell's plan may carry those lines at all. +function rehomeSourceCells() { + const start = production.indexOf('relay_region_rehome_source_cell_ids = [') + assert.notEqual(start, -1, 'production.tfvars has no rehome source cell list') + const end = production.indexOf(']', start) + assert.notEqual(end, -1, 'the rehome source cell list is unterminated') + return new Set( + [...production.slice(start, end).matchAll(/"([^"]+)"/g)].map(([, cell]) => cell) + ) +} + +function startupScript({ cap, image, trusted }) { + return [ + ` printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '${cap}'`, + ` printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '60'`, + ...(trusted ? [ + ` printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '${DIRECTOR_IDENTITY}'`, + ` printf 'ORCA_RELAY_REHOME_AUDIENCE=%s\\n' '${AUDIENCE}'` + ] : []), + `printf 'ORCA_RELAY_IMAGE_DIGEST=%s\\n' '${image.split('@')[1]}'`, + `docker pull '${image}'`, + 'docker run --detach \\', + ' --name orca-relay \\', + ` '${image}'` + ].join('\n') +} + +// The exact shape the apply step's plan has: template replaced, MIG rebound to it. +function rollPlan({ cellId, cap, protocol }) { + return { + configuration: { + root_module: { + resources: [{ + address: 'google_compute_instance_group_manager.relay_gce_cell', + expressions: { + version: [{ + instance_template: { + references: [ + 'google_compute_instance_template.relay_gce_cell', + 'each.key' + ] + }, + name: { constant_value: 'primary' } + }] + } + }] + } + }, + resource_changes: [ + { + address: `google_compute_instance_template.relay_gce_cell[${JSON.stringify(cellId)}]`, + change: { + actions: ['create', 'delete'], + before: { + metadata_startup_script: startupScript({ + cap, + image: ROLLBACK_IMAGE, + trusted: protocol === 1 + }) + }, + after: { + metadata_startup_script: startupScript({ + cap, + image: TARGET_IMAGE, + trusted: protocol === 1 + }), + self_link: null + }, + after_unknown: { self_link: true } + } + }, + { + address: `google_compute_instance_group_manager.relay_gce_cell[${JSON.stringify(cellId)}]`, + change: { + actions: ['update'], + before: { target_size: 1, version: [{ instance_template: 'old' }] }, + after: { target_size: 1, version: [{ instance_template: null }] }, + after_unknown: { version: [{ instance_template: true }] } + } + } + ] + } +} + +function hostname(cellId) { + return cellId.slice('production-gce-'.length) +} + +// The job resolves cap and region from the cell id before any admin call; run that block alone. +function resolveCellShape(cellId) { + const start = workflow.indexOf(' TARGET_HOSTNAME="${TARGET_CELL_ID#production-gce-}"') + assert.notEqual(start, -1, 'the same-cap cell shape block is missing') + const end = workflow.indexOf('\n esac\n', start) + assert.notEqual(end, -1, 'the same-cap cell shape block has no esac') + const script = workflow.slice(start, end + '\n esac'.length).replace(/^ {10}/gm, '') + return spawnSync('bash', [ + '-euo', + 'pipefail', + '-c', + `${script}\necho "\${EXPECTED_REGION} \${EXPECTED_HARD_CAP}"` + ], { env: { ...process.env, TARGET_CELL_ID: cellId }, encoding: 'utf8' }) +} + +describe('same-cap roll scripts accept every same-cap cell', () => { + it('parses every wave cell through the same-cap canary allowlist', () => { + for (const cellId of SAME_CAP_CELLS) { + for (const mode of ['isolate', 'drain', 'activate']) { + assert.deepEqual(parseProductionCapacityCellArguments([ + '--director-origin', 'https://relay.onorca.dev', + '--cell-origin', `https://${hostname(cellId)}.relay.onorca.dev`, + '--cell-id', cellId, + '--approved-cells', 'same-cap', + '--mode', mode + ]), { + directorOrigin: 'https://relay.onorca.dev', + cellOrigin: `https://${hostname(cellId)}.relay.onorca.dev`, + cellId, + mode + }) + } + } + }) + + it('resolves a cap and region for every wave cell and refuses anything else', () => { + for (const cellId of SAME_CAP_CELLS) { + const resolved = resolveCellShape(cellId) + assert.equal(resolved.status, 0, `${cellId}: ${resolved.stderr}`) + assert.match(resolved.stdout.trim(), /^(us-central1 1000|asia-east2 3000)$/) + } + assert.equal(resolveCellShape('production-gce-c17').status, 1) + assert.equal(resolveCellShape('production-gce-c30').status, 1) + }) + + it('passes the same-cap allowlist on every canary invocation the job runs', () => { + const invocations = workflow.split('prepare-relay-production-capacity-canary.mjs').slice(1) + assert.equal(invocations.length, 4) + for (const invocation of invocations) { + const lines = invocation.split('\n') + const end = lines.findIndex((line) => !line.endsWith('\\')) + const call = lines.slice(0, end + 1).join(' ') + assert.match(call, /--approved-cells same-cap/) + assert.match(call, /--mode (isolate|drain|activate)/) + } + }) + + it('passes this cell\'s rehome protocol on every plan validation the job runs', () => { + const invocations = workflow.split('validate-relay-capacity-plan.mjs').slice(1) + assert.equal(invocations.length, 2) + for (const invocation of invocations) { + const lines = invocation.split('\n') + const end = lines.findIndex((line) => !line.trimEnd().endsWith('\\')) + const call = lines.slice(0, end + 1).join(' ') + assert.match(call, /--mode same-cap-cell/) + assert.match(call, /--regional-rehome-protocol "\$\{DESIRED_REHOME_PROTOCOL\}"/) + } + }) + + it('validates a correct plan for every wave cell at that cell\'s rehome protocol', () => { + for (const cellId of SAME_CAP_CELLS) { + const [region, cap] = resolveCellShape(cellId).stdout.trim().split(' ') + const protocol = REHOME_SOURCE_CELLS.has(cellId) ? 1 : 0 + assert.equal(protocol, region === 'us-central1' ? 1 : 0, cellId) + const config = { + mode: 'same-cap-cell', + cellId, + hardCap: Number(cap), + unobservedBound: 60, + image: TARGET_IMAGE, + rollbackImage: ROLLBACK_IMAGE, + rehomeDirectorServiceAccount: DIRECTOR_IDENTITY, + rehomeAudience: AUDIENCE, + regionalRehomeProtocol: String(protocol) + } + const plan = rollPlan({ cellId, cap, protocol }) + assert.deepEqual( + validateCapacityPlan(plan, config), + { mode: 'same-cap-cell', changes: 2 }, + cellId + ) + // The other protocol must reject the same plan, or the flag decides nothing. + assert.throws( + () => validateCapacityPlan(plan, { + ...config, + regionalRehomeProtocol: String(1 - protocol) + }), + /reviewed image and capacity/, + cellId + ) + } + }) + + it('leaves the US-only capacity job on the default allowlist', () => { + assert.doesNotMatch(capacityWorkflow, /--approved-cells/) + }) +}) diff --git a/cloud/dev/scripts/validate-relay-capacity-plan.mjs b/cloud/dev/scripts/validate-relay-capacity-plan.mjs index 7307295d206..294e85ae31d 100644 --- a/cloud/dev/scripts/validate-relay-capacity-plan.mjs +++ b/cloud/dev/scripts/validate-relay-capacity-plan.mjs @@ -4,7 +4,18 @@ import { pathToFileURL } from 'node:url' const SERVICE_ACCOUNT_EMAIL = /^[a-z][a-z0-9-]{4,28}[a-z0-9]@[a-z0-9-]+\.iam\.gserviceaccount\.com$/ -function parseArguments(argv) { +const REHOME_CONFIG = + /^ printf 'ORCA_RELAY_REHOME_(?:DIRECTOR_SERVICE_ACCOUNT|AUDIENCE)=%s\\n' '[^'\n]+'$/ + +// Only cells listed as regional rehome sources get rehome trust lines in their startup script. +function rehomeProtocol({ regionalRehomeProtocol }) { + if (![0, 1, '0', '1'].includes(regionalRehomeProtocol)) { + throw new Error('same-cap Terraform plan has an invalid regional rehome protocol') + } + return Number(regionalRehomeProtocol) +} + +export function parseCapacityPlanArguments(argv) { const values = {} for (let index = 0; index < argv.length; index += 2) { const key = argv[index] @@ -31,8 +42,12 @@ function parseArguments(argv) { values.mode === 'same-cap-cell' && (!values['rollback-image'] || !values['rehome-director-service-account'] || - !values['rehome-audience']) + !values['rehome-audience'] || + !['0', '1'].includes(values['regional-rehome-protocol'])) ) throw new Error('same-cap validation requires rollback image and rehome trust config') + if (values.mode !== 'same-cap-cell' && values['regional-rehome-protocol'] !== undefined) { + throw new Error('--regional-rehome-protocol applies only to same-cap-cell validation') + } if (values.mode === 'same-cap-image' && !values['rollback-image']) { throw new Error('same-cap image validation requires a rollback image') } @@ -51,7 +66,8 @@ function parseArguments(argv) { capacityServiceAccount: values['capacity-service-account'], rollbackImage: values['rollback-image'], rehomeDirectorServiceAccount: values['rehome-director-service-account'], - rehomeAudience: values['rehome-audience'] + rehomeAudience: values['rehome-audience'], + regionalRehomeProtocol: values['regional-rehome-protocol'] } } @@ -175,15 +191,13 @@ function normalizedStartupScript( /^ printf 'ORCA_RELAY_CELL_CONNECTION_(?:HARD_CAP|UNOBSERVED_BOUND)=%s\\n' '[0-9]+'$/ const capacityIdentity = /^ printf 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT=%s\\n' '[a-z][a-z0-9-]{4,28}[a-z0-9]@[a-z0-9-]+\.iam\.gserviceaccount\.com'$/ - const rehomeConfig = - /^ printf 'ORCA_RELAY_REHOME_(?:DIRECTOR_SERVICE_ACCOUNT|AUDIENCE)=%s\\n' '[^'\n]+'$/ return script .split('\n') .filter( (line) => (preserveCapacity || !capacityAssignment.test(line)) && (!stripCapacityIdentity || !capacityIdentity.test(line)) && - (!stripRehomeConfig || !rehomeConfig.test(line)) + (!stripRehomeConfig || !REHOME_CONFIG.test(line)) ) .join('\n') .replaceAll(image, '') @@ -213,7 +227,8 @@ function requireDesiredStartupScript(script, config) { ` printf 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT=%s\\n' '${config.capacityServiceAccount}'` ]) } - if (config.mode === 'same-cap-cell') { + const rehomeTrusted = config.mode === 'same-cap-cell' && rehomeProtocol(config) === 1 + if (rehomeTrusted) { expected.push( [ /^ printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '[^'\n]+'$/, @@ -225,9 +240,15 @@ function requireDesiredStartupScript(script, config) { ] ) } + // A protocol-0 cell is not a rehome source, so gaining any rehome trust line is real drift. + const unexpectedRehome = + config.mode === 'same-cap-cell' && + !rehomeTrusted && + lines.some((line) => REHOME_CONFIG.test(line)) if ( typeof script !== 'string' || relayImage(script) !== config.image || + unexpectedRehome || expected.some(([pattern, line]) => !hasExactSingleAssignment(lines, pattern, line)) ) { throw new Error('cell plan does not contain the reviewed image and capacity') @@ -450,6 +471,9 @@ export function validateCapacityPlan(plan, config) { ) { throw new Error('capacity Terraform plan has an invalid service account') } + if (config.mode === 'same-cap-cell') { + rehomeProtocol(config) + } if ( config.mode === 'same-cap-cell' && (!SERVICE_ACCOUNT_EMAIL.test(config.rehomeDirectorServiceAccount ?? '') || @@ -504,7 +528,7 @@ export function validateCapacityPlan(plan, config) { } export function main(argv = process.argv.slice(2)) { - const config = parseArguments(argv) + const config = parseCapacityPlanArguments(argv) const plan = JSON.parse(readFileSync(0, 'utf8')) process.stdout.write(`${JSON.stringify({ event: 'relay_capacity_plan_verified', ...validateCapacityPlan(plan, config) })}\n`) } diff --git a/cloud/dev/scripts/validate-relay-capacity-plan.test.mjs b/cloud/dev/scripts/validate-relay-capacity-plan.test.mjs index 207285dc570..fb6ccb57e1c 100644 --- a/cloud/dev/scripts/validate-relay-capacity-plan.test.mjs +++ b/cloud/dev/scripts/validate-relay-capacity-plan.test.mjs @@ -1,6 +1,9 @@ import assert from 'node:assert/strict' import { test } from 'node:test' -import { validateCapacityPlan as validateCapacityPlanRaw } from './validate-relay-capacity-plan.mjs' +import { + parseCapacityPlanArguments, + validateCapacityPlan as validateCapacityPlanRaw +} from './validate-relay-capacity-plan.mjs' const config = { cellId: 'staging-gce-c3', @@ -466,7 +469,8 @@ test('same-cap mode preserves 1000/60 while adding only the reviewed trust confi image, rollbackImage, rehomeDirectorServiceAccount: directorIdentity, - rehomeAudience: audience + rehomeAudience: audience, + regionalRehomeProtocol: '1' } assert.deepEqual( validateCapacityPlan({ resource_changes: [template, manager] }, sameCapConfig), @@ -644,3 +648,134 @@ test('same-cap mode preserves 1000/60 while adding only the reviewed trust confi { mode: 'same-cap-image', changes: 1, changeKind: 'manager-convergence' } ) }) + +test('protocol-0 same-cap cells roll without rehome trust lines', () => { + const rollbackImage = `us-docker.pkg.dev/project/relay/image@sha256:${'d'.repeat(64)}` + const image = `us-docker.pkg.dev/project/relay/image@sha256:${'e'.repeat(64)}` + const directorIdentity = 'relay-director@project.iam.gserviceaccount.com' + const audience = 'https://relay.example.com/v1/admin/host-drain' + const startup = ({ selectedImage, trust = false }) => [ + ` printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '3000'`, + ` printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '60'`, + ` printf 'ORCA_RELAY_CELL_REGION=%s\\n' 'asia-east2'`, + ...(trust ? [ + ` printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '${directorIdentity}'`, + ` printf 'ORCA_RELAY_REHOME_AUDIENCE=%s\\n' '${audience}'` + ] : []), + `printf 'ORCA_RELAY_IMAGE_DIGEST=%s\\n' '${selectedImage.split('@')[1]}'`, + `docker pull '${selectedImage}'`, + 'docker run --detach \\', + ' --name orca-relay \\', + ` '${selectedImage}'` + ].join('\n') + const template = { + address: 'google_compute_instance_template.relay_gce_cell["production-gce-c27"]', + change: { + actions: ['create', 'delete'], + before: { metadata_startup_script: startup({ selectedImage: rollbackImage }) }, + after: { metadata_startup_script: startup({ selectedImage: image }), self_link: null }, + after_unknown: { self_link: true } + } + } + const manager = { + address: 'google_compute_instance_group_manager.relay_gce_cell["production-gce-c27"]', + change: { + actions: ['update'], + before: { target_size: 1, version: [{ instance_template: 'old' }] }, + after: { target_size: 1, version: [{ instance_template: null }] }, + after_unknown: { version: [{ instance_template: true }] } + } + } + const asiaConfig = { + cellId: 'production-gce-c27', + hardCap: 3_000, + unobservedBound: 60, + mode: 'same-cap-cell', + image, + rollbackImage, + rehomeDirectorServiceAccount: directorIdentity, + rehomeAudience: audience, + regionalRehomeProtocol: '0' + } + assert.deepEqual( + validateCapacityPlan({ resource_changes: [template, manager] }, asiaConfig), + { mode: 'same-cap-cell', changes: 2 } + ) + const gainsTrust = structuredClone(template) + gainsTrust.change.after.metadata_startup_script = startup({ + selectedImage: image, + trust: true + }) + assert.throws( + () => validateCapacityPlan({ resource_changes: [gainsTrust, manager] }, asiaConfig), + /reviewed image and capacity/ + ) + // Under protocol 1 that same script is the reviewed roll: trust is added, not drift. + assert.deepEqual( + validateCapacityPlan( + { resource_changes: [gainsTrust, manager] }, + { ...asiaConfig, regionalRehomeProtocol: '1' } + ), + { mode: 'same-cap-cell', changes: 2 } + ) + // A protocol-1 cell whose script has no rehome lines is the pre-existing failure, unchanged. + assert.throws( + () => validateCapacityPlan( + { resource_changes: [template, manager] }, + { ...asiaConfig, regionalRehomeProtocol: '1' } + ), + /reviewed image and capacity/ + ) + for (const protocol of [undefined, '', '2', 'yes']) { + assert.throws( + () => validateCapacityPlan( + { resource_changes: [template, manager] }, + { ...asiaConfig, regionalRehomeProtocol: protocol } + ), + /invalid regional rehome protocol/ + ) + } +}) + +test('the rehome protocol argument is required by same-cap-cell mode alone', () => { + const image = `us-docker.pkg.dev/project/relay/image@sha256:${'e'.repeat(64)}` + const rollbackImage = `us-docker.pkg.dev/project/relay/image@sha256:${'d'.repeat(64)}` + const sameCapArguments = (...extra) => [ + '--mode', 'same-cap-cell', + '--cell-id', 'production-gce-c27', + '--hard-cap', '3000', + '--unobserved-bound', '60', + '--image', image, + '--rollback-image', rollbackImage, + '--rehome-director-service-account', 'relay-director@project.iam.gserviceaccount.com', + '--rehome-audience', 'https://relay.onorca.dev/v1/admin/host-drain', + ...extra + ] + assert.equal( + parseCapacityPlanArguments(sameCapArguments('--regional-rehome-protocol', '0')) + .regionalRehomeProtocol, + '0' + ) + assert.throws( + () => parseCapacityPlanArguments(sameCapArguments()), + /requires rollback image and rehome trust config/ + ) + for (const protocol of ['', '2', 'true']) { + assert.throws( + () => parseCapacityPlanArguments(sameCapArguments('--regional-rehome-protocol', protocol)), + /requires rollback image and rehome trust config/ + ) + } + assert.throws( + () => parseCapacityPlanArguments([ + '--mode', 'bootstrap-cell', + '--cell-id', 'staging-gce-c3', + '--hard-cap', '1000', + '--unobserved-bound', '60', + '--image', image, + '--capacity-service-account', 'orca-cap@onorca-cloud.iam.gserviceaccount.com', + '--regional-rehome-protocol', '0' + ]), + /applies only to same-cap-cell validation/ + ) +}) diff --git a/cloud/dev/scripts/verify-relay-capacity-transition.mjs b/cloud/dev/scripts/verify-relay-capacity-transition.mjs index e5ebe77d45f..b81ea15afb3 100644 --- a/cloud/dev/scripts/verify-relay-capacity-transition.mjs +++ b/cloud/dev/scripts/verify-relay-capacity-transition.mjs @@ -1,4 +1,5 @@ import { pathToFileURL } from 'node:url' +import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs' const CAPACITY_PROTOCOL = 2 @@ -378,9 +379,12 @@ export async function verifyCapacityTransition(config, overrides = {}) { const token = overrides.token ?? process.env.ORCA_RELAY_ADMIN_ID_TOKEN if (!token || token.length > 8_192) throw new Error('admin identity token is unavailable') const health = await responseJson( - await fetchImpl(`${config.directorOrigin}/health`, { - signal: AbortSignal.timeout(15_000) - }), + await fetchAdminOnceMore( + fetchImpl, + `${config.directorOrigin}/health`, + {}, + { wait, timeoutMs: 15_000 } + ), 'director health' ) if (health.ok !== true || health.connectionCapacityProtocol !== CAPACITY_PROTOCOL) { @@ -394,12 +398,16 @@ export async function verifyCapacityTransition(config, overrides = {}) { lastObservation = { runtimeAvailable: runtime !== null } if ((runtime === null) === (config.runtime === 'unavailable')) { const result = await responseJson( - await fetchImpl(`${config.directorOrigin}/v1/admin/cell-status`, { - method: 'POST', - headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, - body: JSON.stringify({ v: 1, cellId: config.cellId }), - signal: AbortSignal.timeout(30_000) - }), + await fetchAdminOnceMore( + fetchImpl, + `${config.directorOrigin}/v1/admin/cell-status`, + { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, cellId: config.cellId }) + }, + { wait } + ), 'cell status' ) const status = result.status diff --git a/cloud/dev/scripts/verify-relay-capacity-transition.test.mjs b/cloud/dev/scripts/verify-relay-capacity-transition.test.mjs index fb865257c4a..e596ffbced7 100644 --- a/cloud/dev/scripts/verify-relay-capacity-transition.test.mjs +++ b/cloud/dev/scripts/verify-relay-capacity-transition.test.mjs @@ -1094,3 +1094,77 @@ test('does not retry a rejected cell admin token', async () => { ) assert.equal(waits, 0) }) + +test('retries a transient 503 on the director cell-status read', async () => { + const base = harness() + const statusCalls = [] + const result = await verifyCapacityTransition(config, { + token: 'masked-token', + wait: async () => {}, + fetch: async (url, options) => { + const path = new URL(url).pathname + if (path !== '/v1/admin/cell-status') return await base(url, options) + statusCalls.push(path) + if (statusCalls.length === 1) return new Response('warming up', { status: 503 }) + return await base(url, options) + } + }) + assert.equal(statusCalls.length, 2) + assert.equal(result.cellId, config.cellId) +}) + +test('fails when both director cell-status attempts return a transient 503', async () => { + const base = harness() + let statusCalls = 0 + await assert.rejects( + verifyCapacityTransition(config, { + token: 'masked-token', + wait: async () => {}, + fetch: async (url, options) => { + const path = new URL(url).pathname + if (path !== '/v1/admin/cell-status') return await base(url, options) + statusCalls += 1 + return new Response('warming up', { status: 503 }) + } + }), + /cell status returned 503/ + ) + assert.equal(statusCalls, 2) +}) + +test('retries a transient 503 on the director health preflight', async () => { + const base = harness() + let healthCalls = 0 + const result = await verifyCapacityTransition(config, { + token: 'masked-token', + wait: async () => {}, + fetch: async (url, options) => { + const path = new URL(url).pathname + if (path !== '/health') return await base(url, options) + healthCalls += 1 + if (healthCalls === 1) return new Response('warming up', { status: 503 }) + return await base(url, options) + } + }) + assert.equal(healthCalls, 2) + assert.equal(result.cellId, config.cellId) +}) + +test('fails when both director health attempts return a transient 503', async () => { + const base = harness() + let healthCalls = 0 + await assert.rejects( + verifyCapacityTransition(config, { + token: 'masked-token', + wait: async () => {}, + fetch: async (url, options) => { + const path = new URL(url).pathname + if (path !== '/health') return await base(url, options) + healthCalls += 1 + return new Response('warming up', { status: 503 }) + } + }), + /director health returned 503/ + ) + assert.equal(healthCalls, 2) +}) diff --git a/cloud/docs/relay-improvement-checklist-2026-09.md b/cloud/docs/relay-improvement-checklist-2026-09.md new file mode 100644 index 00000000000..91f1cc742ef --- /dev/null +++ b/cloud/docs/relay-improvement-checklist-2026-09.md @@ -0,0 +1,189 @@ +# Relay improvement: implementation checklist, lanes, and disruption + +Companion to [`relay-improvement-roadmap-2026-09.md`](./relay-improvement-roadmap-2026-09.md) (item numbers +match). This file answers three questions per item: what are the concrete steps, what can run in parallel, +and will a user notice. + +## Status as of 2026-09-04 22:30Z + +Three buckets. "Merged" means the code is on `main` and nothing in production has changed yet. "Deployed" means users are already getting it. "Awaiting owner" means I will not touch production without a go. + +**Deployed to production** +- Auth instance cap 20 + dead-family audit fix (orca-cloud #474) as revision `orca-cloud-auth-00031-tox`. +- Dynamic NAT ports in both regions (stablyai/orca #18693). Zero drops and zero proxy dial errors since. +- Nine alert policies with log metrics: 4 auth (#475), 3 relay Cloud SQL/NAT (#18693), 1 cell process-exit (#18717), all on the relay Slack channel. + +**Merged, ships with the next relay cell image roll (Roll 1 carries `519f4914`; Roll 2 needs a fresh image build)** +- Per-cell inventory locks, delta counters, pool `statement_timeout` (#18722). Roll 2. +- Cells dial Cloud SQL with `--private-ip` when configured (#18720). Inert until 2.1 applies. +- Phone shows a clear "sign in on the desktop again" state when the desktop is signed out (#18698). + +**Merged, ships with the next auth deploy** +- Refresh rotation grace window (orca-cloud #478). Startup adds one nullable column (brief exclusive lock on `refresh_tokens`). +- Pruning job code (orca-cloud #476) is in the image; the job itself is Terraform-disabled until 1.2. + +**Merged, ships with the next desktop release** +- Never replay a refresh token after a timeout; ±10 % jitter on relay lease renewal (#18719). +- Renderer learns when a cloud session is revoked (#18694). + +**Merged, not applied** +- Incident dashboard (#18717) blocked behind the runtime-metric label drift (5.x first item). +- Monitor probe fix (#18723) is live in the workflow; the same-cap roll gate has not yet produced a green dry-run since. + +**Awaiting owner go (production mutations)** +1. Roll 1 cell image roll (1.1): dry-run gate, then c8 canary, then batches. +2. Auth deploy carrying #478 (3.1): quiet minute for the column add. +3. orca-cloud #477 private IP (2.1): merge arms an instance restart and a one-way door. Recommendation: hold. +4. Runtime-metric `region` label drift (5.x): intentional replacement of 21 metrics, or drop the label. +5. Enable pruning (1.2): first budget 20k rows; needs a Terraform apply. +6. Paging channel for auth alerts (5.2): needs the destination from you. + +**Open code follow-ups (no gate, nobody assigned)** +- Monitor summary Markdown does not render `tolerated: true` continuity events (added by #18798); the state artifact has them, the checkpoint table does not. +- Relay container boot races the `cloud-sql-proxy` sidecar: c13's fresh container exited twice (`applyPostgresSchema` connection timeout, 2 s each) before the proxy was listening. Make schema apply wait for the proxy or order the containers. +- `cloud-deploy-relay-production-capacity-job.yml` (~line 416) has the same wave-0 single-shot preflight carve-out that #18778 removes from the same-cap job; its single-evidence path never retries freshness-only failures. +- `cloud/package.json` `test` names every dev-script test file explicitly; an unregistered `*.test.mjs` is silently never run in CI (found by #18769). Needs a glob or a ratchet that fails on an unlisted test file. +- Same-cap job's verify step uses bare `curl --fail-with-body` against the just-rolled cell; one 503 at the LB warm-up edge failed c8 canary #2 (run 33935407461) after the transition verifier had already passed. Needs a bounded retry, same rule as #18723/#18740. +- `verify-mutation` in `cloud-deploy-relay-production.yml`, the multi-target workflow, and the capacity workflow still binds to an exact commit; same exposure #18754 fixed for the same-cap and rehome paths. +- `incident-live-preflight-cli.ts` reports only `source/code` (`active-probe/threshold_max`) with no signal name or observed value, so a failed mutation preflight (c27 recovery #3, run 33986948522) cannot be attributed to an endpoint without an out-of-band probe. Print the signal and observed/threshold pair. Related: the 2 000 ms `endpointLatencyMs` bar is shared by US and Asia cells while Asia /health round trips from a US runner sit at 0.7–1.3 s idle; consider a per-region bar or the p50 of the gate window instead of one shot. Gates #44 and #45 (2026-09-05) both froze on `cell.production-gce-c27.latency_ms` at 2.6–2.7 s with c28 showing the identical tail under operator probes; the bar is now blocking Asia rolls. **Fix: stablyai/orca #18877** (per-region `cellEndpointLatencyMs`, us-central1 2 000 / asia-east2 4 000, plus signal/observed/threshold in preflight messages). Residual: `probeEndpointHealth` in `resource-inventory.ts` still uses the flat 2 000 bar to decide whether to retry after the 10 s readiness-cache wait, so a healthy Asia cell over 2 s costs one extra probe per sample (latency, not verdict); thread the region bar into the retry decision. +- The root oxlint config ignores `cloud/**`, so `check:code-quality:changed` never inspects relay-ops or the cloud dev scripts; typecheck + vitest is the only gate there. +- Monitor bars that froze on non-health today: `directorInstancesMin: 5` with `latest-sum` (one-minute instance recycle), `endpointLatencyMs: 2000` on a US-runner probe to asia-east2, `cloudDataMaxAgeMs: 180000` vs Cloud Monitoring publish lag up to 255 s. Recalibrate with a week of data. +- `parsed()` in `resource-inventory.ts` still returns null on a 200 with a malformed MIG body; a second path to `runtime_power_unknown`. +- Deploy script strips `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` on every release (3.1 first item). +- `assignOnce` placement lock still global (4.1 remainder). +- Region preference (4.2), retries-bar recalibration after a week of Roll 2 data (4.4), pruner `stopReason` alert (1.5). +- Full apps-root apply for 4 unrelated drifts (1.4), from a host with the 1Password account. + +## Uplift ranking (reliability gained per unit of effort) + +| Rank | Item | Why it ranks here | +|---|---|---| +| 1 | 1.1 cell image roll | Removes the only crash mode we have seen in production. 22 of 23 cells still have it. One afternoon. | +| 2 | 3.1 refresh rotation grace window | Turns the entire "slow auth → mass sign-out" class into a slowdown. One day. | +| 3 | 4.1 inventory lock contention | The floor under every 503 and slow phone accept, every day, not just incidents. One week. | +| — | 2.2 relay/auth database split | **Deferred 2026-09-04** to ~2026-11-01. Biggest structural fix, but the concrete cause is fixed and alerts now page; see roadmap 2.2 for re-open triggers. | +| 4 | 1.2 + 1.3 pruning and reclaim | Defuses the 63 M-row time bomb. Low effort, mostly waiting. | +| 5 | 5.1 + 5.2 crash alert, page a human | Cheapest detection uplift; today's incident ran 4 h unpaged. | +| 6 | 2.1 private IP | Durable version of a fix that already landed (dynamic NAT ports). Do it on the existing instance. | +| 7 | 4.3 + 3.2 desktop hardening | Small, ride the normal desktop release. | +| 8 | 4.2, 4.4, 5.4, 1.4, 1.5 | Housekeeping and quality-of-life. | + +## The shared bottleneck: cell rolls + +Every change to what runs on a cell (image, proxy flag, env, relay code) needs a same-cap roll: drain → +recreate → verify, one wave at a time, gated by the 15-minute monitor, about an afternoon. Each wave forces +the desktops on that cell to re-dial (c7 canary: 807 controls re-dialed in ~10 s) and phones on those +desktops reconnect on their normal retry. Users see a few seconds of "reconnecting" per wave. + +So batch. Two rolls, not five: + +- **Roll 1 (now):** current image only (1.1). Do not wait for anything else. +- **Roll 2 (week 2–3):** proxy `--private-ip` (2.1) + relay pool `statement_timeout` (2.3) + lock-contention + fix (4.1), all in one image/template. Prerequisite: 2.1's peering and private IP exist first. + +## Lanes (independent; different people can own them) + +``` +Lane A data plane 1.1 roll ──────────────────► Roll 2 (2.1 flag + 2.3 + 4.1) ──► 4.4 recalibrate +Lane B auth/DB 1.2 enable pruning ──(10 d)──► 1.3 reclaim 3.1 grace window (any time) +Lane C network 2.1 peering + private IP ─────┐ (feeds Roll 2) (2.2 DB split deferred) +Lane D desktop 3.2 no same-token retry, 4.3 lease jitter (any release; wire-compatible) +Lane E observability 1.5, 5.1, 5.2, 5.4 (Terraform only, any time) +Lane F director 4.2 region preference (Cloud Run deploy, any time) +Misc 1.4 full apps-root apply (any time; see its check) +``` + +Hard dependencies: Roll 2 waits on 2.1's network work; 1.3 waits on 1.2 finishing. Everything else is +independent. (2.2 deferred; if revived, do it after 2.1 so the new instance is private from day one.) + +## Disruption summary + +| Item | User-visible? | What they see | Mitigation | +|---|---|---|---| +| 1.1 / Roll 2 | **Yes, transient** | Per wave, desktops on that cell reconnect within seconds; phones follow on retry. | Waves gated by the monitor; run in the US night. Already rehearsed on c7. | +| 1.2 pruning | No | Background deletes, 5k rows per batch. | Small first budget; watch `stopReason` and Cloud SQL write throughput. Stop the scheduler if checkpoint alerts fire. | +| 1.3 reclaim | **Depends on tool** | `VACUUM FULL` takes an exclusive lock on `refresh_tokens`: sign-in and refresh block for its duration (minutes to tens of minutes on 16 GB). `pg_repack` holds only brief locks. | Use `pg_repack`. If VACUUM FULL, announce a maintenance window. | +| 1.4 full apps apply | Should be none, **verify** | Terraform will create a new auth revision (env added). Traffic is pinned to `00031-tox` by name, so the new revision should receive 0 %. | Confirm in the plan that no `traffic` change appears. If it does, stop: the Terraform image variable is not the serving image. | +| 1.5, 5.x alerts | No | | | +| 2.1 private IP | **Yes, certain** | Google: "Configuring an existing Cloud SQL instance to use private IP causes the instance to restart, resulting in downtime." No in-place path, HA does not avoid it. Expect 1–2 min DB unavailability: sign-in fails, relay renewals retry. **One-way door**: private IP cannot be disabled and the VPC link cannot be removed once set. The proxy flag change rides Roll 2. | Off-peak; only after Roll 1 (old image dies on a 2 min DB blip). Owner decision required before the foundation apply. | +| 2.2 DB split (deferred) | **Yes, scheduled** | Relay unavailable for the cutover (drain all cells → copy relay tables → flip `DATABASE_URL` → restart). Minutes if rehearsed. Desktops and phones reconnect automatically after. | Rehearse on staging; do it in the US night; announce. | +| 2.3 statement timeout | No beyond Roll 2 | | | +| 3.1 grace window | No | Auth deploys are no-traffic candidate → smoke → promote. | Security trade-off: a stolen token replayed inside the window is served once instead of revoking. 60 s is the usual choice. | +| 3.2, 4.3 desktop | No | Normal app update. | | +| 4.1 lock fix | No beyond Roll 2 | | Verify against real Postgres on 55440 with concurrent probes before shipping. | +| 4.2 region preference | **Minor, Asia users** | Phones that start being placed in Asia reconnect once to a nearer cell. | Roll out behind the existing region-preference flag. | +| 4.4 | No | | | + +## Checklists + +### 1.1 Cell image roll (Roll 1) +- [x] Confirm fleet is quiet: 15-min monitor dry-run passes. #19 green 23:07:53Z (run 33927238469). Canary then failed the evidence provenance check because main moved during the gate; re-gating with a same-commit chain. +- [x] Confirm director is on 519f4914 and c7 on 85bf6799 (confirmed 2026-09-04 via instance-template census; 20 serving cells still on `5aedbca5`) (`verify` mode of the same-cap workflow). +- [x] Dispatch `cloud-deploy-relay-production-same-cap` waves per the plan in the findings doc; one wave, verify, next. Done 2026-09-05 01:14Z–22:27Z: c8 canary, US batches c9–c10, c13–c16, c19–c26 at protocol 1, then Asia c27 (recovered via `mode=rollback` re-entry after gate freezes on the flat latency bar, fixed by #18877), c28, c29 as single-cell canaries at protocol 0. +- [x] After each wave: the transition verifier passed at migration-only and again at general on every cell (assignments carried, heartbeat fresh, hard cap 3 000); no `container die` fleet-wide across the whole roll. The 4408/1006 burst per wave was not measured separately; the verifier's assignment count before and after each restart is the recovery evidence recorded. +- [x] Record image census in the findings doc. 2026-09-05 22:27Z: all 19 general cells on `519f4914` except c7 on `85bf6799`; existing-only c1–c6, c11, c12 and migration-only c17, c18 untouched on their older images by design. Selector at gen 148. + +### 1.2 Enable pruning +- [x] `auth_token_pruner_image` = digest of `orca-cloud-auth-00031-tox` (`343a0915…`; it contains the entrypoint). orca-cloud #479 merged. +- [x] `auth_token_pruner_enabled = true`, `auth_token_pruner_max_rows_per_run = 20000` for the first day (orca-cloud #479). +- [x] Targeted plan asserted 9 create / 0 change / 0 destroy. Applied 2026-09-05 02:06Z. +- [x] Trigger one run by hand; read the summary event. 02:18Z: `time-budget`, 73 batches, 365k scanned, 1 040 deleted (1 021 revoked, 19 expired), no errors. Scan-bound. +- [ ] Raise the budget to the default 200k after a clean day; watch Cloud SQL write MB/s and the checkpoint alert. +- [ ] 1.5: log metric + policy on `stopReason != complete`. + +### 1.3 Reclaim +- [ ] Wait for steady-state runs deleting ~0 rows. +- [ ] `pg_repack -t refresh_tokens` off-peak (needs the extension; check `pg_available_extensions`). Not `VACUUM FULL` without a window. +- [ ] Confirm table + index size and `disk/utilization` dropped. + +### 1.4 Full apps-root apply +- [ ] Run from CI or a host with the 1Password account (local plan fails on the Cloudflare data source). +- [ ] Plan shows exactly the four known drifts and **no traffic change** on `google_cloud_run_v2_service.auth`. +- [ ] Apply; confirm `status.traffic` still pins `00031-tox` at 100 %. + +### 2.1 Private IP (PRs open: orca-cloud #477 foundation, stablyai/orca #18720 relay flag) +- [ ] **Owner decision**: the foundation apply restarts the instance and is irreversible on Google's side. Merging #477 arms the next foundation apply; hold the merge until the window is chosen. +- [ ] Director is out of scope: it uses the Cloud Run built-in connector (managed Google path, not the relay VPC NAT), so it consumed none of the exhausted ports; moving it needs Direct VPC egress + a separate DSN secret. Own PR if ever wanted. +- [ ] Step 7 (`ipv4_enabled=false`) is blocked until humans have IAP/bastion access and the director is moved; it breaks both today. +- [ ] Allocate a `/24` private services range on the relay VPC; `google_service_networking_connection`. +- [ ] Add `ip_configuration.private_network` to `google_sql_database_instance.auth` (foundation root). Plan must show update, not replace. +- [ ] Apply off-peak; expect a possible restart. Watch auth 5xx alert and relay `sqlFailures`. +- [ ] Cell template: proxy args add `--private-ip` (code merged #18720; flag not set). Director: Direct VPC egress or connector, then the same flag. Both ride Roll 2. +- [ ] After Roll 2: NAT `port_usage` for relay gateways drops to ~0; then consider `ipv4_enabled = false` (removes the public IP; breaks the local `cloud-sql-proxy --token` workflow unless it also goes private). + +### 2.2 Database split (deferred to ~2026-11-01; checklist kept for when it is revived) +- [ ] New `google_sql_database_instance.relay` (private IP from day one, its own size and flags). Staging first. +- [ ] Relay schema applies cleanly to an empty instance (it does at startup). +- [ ] Rehearsal on staging: drain → `pg_dump` relay tables → restore → flip `relay_database_url` secret → restart director + cells → phones/desktops reconnect. Time it. +- [ ] Production: announce a window; same steps; verify `orca_relay_runtime_metrics` controls recover to pre-cutover count. +- [ ] Update `production-cloud-sql-app-consumers` budget test and both alert policies' `database_id`. + +### 2.3 Relay pool statement timeout (merged stablyai/orca #18722; ships Roll 2) +- [x] `statement_timeout` on the relay `pg.Pool` (5 s, env-configurable; schema pool untimed; `57014` retryable), below the control-renewal deadline; DDL on an untimed connection (same pattern as auth #476). +- [x] Postgres test on 55440: a held lock fails the query fast and the bounded retry takes over. + +### 3.1 Refresh rotation grace window (orca-cloud #478 merged 2026-09-04; deploy pending owner go) +- [ ] Fix the deploy-script env strip for `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` (pre-existing; found by #478). +- [x] `rotateRefreshToken`: if `rotated_at` within 60 s and not revoked, return the existing successor (idempotent), no revoke, no audit. +- [x] Outside the window or a third presentation: unchanged (revoke + audit). +- [x] Tests: replay inside window returns same successor; outside revokes; concurrent double-present yields one successor. +- [x] Deploy via `deploy-auth-production` (candidate → smoke → promote). Deployed 2026-09-04 23:15Z as `orca-cloud-auth-00035-gos`, cap 20 kept, 0 5xx; `successor_material` column present; sealed successors being written. (candidate → smoke → promote). + +### 3.2 / 4.3 Desktop (merged stablyai/orca #18719; ships next desktop release) +- [x] 3.2: on refresh timeout, re-read stored session before retrying; do not re-send a token already rotated locally. +- [x] 4.3: ±10 % jitter on control lease renewal; unit test on the distribution; wire-compatible (server accepts early renewals already). + +### 4.1 Lock contention (partial: stablyai/orca #18722 merged; ships Roll 2) +- [x] Replace the global `FOR UPDATE` over `relay_cells` with per-cell row locks; counters delta-only. Remaining: `assignOnce` placement lock is still global (optimistic snapshot follow-up). with per-cell row locks or `pg_advisory_xact_lock(cell)`; counters delta-only. +- [x] Postgres tests on 55440 with concurrent probes (in #18722). Staging load run still owed; `postgres_retries` per hour drops in staging load run. +- [ ] Ships in Roll 2; then 4.4 recalibrates the retries bar from a week of data. + +### 4.2 Region preference +- [ ] Director: honor requested region when the preferred region has headroom, else sticky. Behind the existing flag. +- [ ] Measure with `orca_relay_runtime_metrics` region counters before/after. + +### 5.x Observability +- [x] **Relay-root runtime-metric drift**: resolved by dropping the `region` label to match live state (stablyai/orca #18734). Applied 2026-09-04 23:11Z: 8 never-applied `control_*` renewal metrics + the incident dashboard created, 0 destroyed, 21 live metrics untouched. +- [x] 5.1 `container die` log metric per cell (`relay_cell_process_exit`, applied 2026-09-04 via #18717), > 3 / 15 min, relay channel. +- [ ] 5.2 Add a paging channel (**needs owner input**: destination) to `auth_alert_notification_channels` for refresh rejections + latency. +- [x] 5.4 One dashboard (applied 2026-09-04 23:11Z): `orca_relay_cloud_sql_wal_checkpoint`, NAT drops, `orca_auth_refresh_401`, summed `controls`. diff --git a/cloud/docs/relay-improvement-roadmap-2026-09.md b/cloud/docs/relay-improvement-roadmap-2026-09.md new file mode 100644 index 00000000000..64f33c69a70 --- /dev/null +++ b/cloud/docs/relay-improvement-roadmap-2026-09.md @@ -0,0 +1,67 @@ +# Relay improvement roadmap (written 2026-09-04, after the auth/relay outage) + +Owner-facing list of what is left to make the relay more robust, in priority order. Evidence and history +for every item is in [`relay-reconnect-2026-09-findings.md`](./relay-reconnect-2026-09-findings.md) +(Findings 1–13). Everything already landed on 2026-09-04 is listed at the end so this file is complete on +its own. + +## 1. Finish what 2026-09-04 started (this week) + +| # | Item | Why | How | Size | +|---|---|---|---|---| +| 1.1 | **Roll all 23 cells onto the current relay image** | Every cell still runs the image that exits the whole process on a Postgres connect timeout (Finding 6). The fixed image runs only on the director and c7. Any future DB stall repeats the 200-crashes-in-48h pattern. | `cloud-deploy-relay-production-same-cap` waves, gated by the 15-min monitor. Roll inputs and canary results are in the findings doc ("Roll inputs", "Canary blast radius"). | one afternoon | +| 1.2 | **Enable the refresh_tokens pruning job** (orca-cloud #476, merged, off) | `refresh_tokens` is 63 M rows / 26 GB and grows forever; its size is what turned a slow disk into a sign-out storm (Finding 13). | Build an auth image from main (the 21:04Z deploy already contains the entrypoint: `orca-cloud-auth-00031-tox`, digest `343a0915…`), set `auth_token_pruner_enabled = true` and the image digest in `infra/terraform-apps/environments/production.tfvars`, apply targeted. First run with a small `auth_token_pruner_max_deleted_rows`. Watch the run summary's `stopReason`, not the exit code. ~48 M rows drain in ~10 days at 200k/hour. | 1 hour + 10 days of watching | +| 1.3 | **Reclaim the disk after pruning** | Deletes leave dead tuples; the 16 GB table does not shrink on its own. | `pg_repack` (or `VACUUM FULL` in a maintenance window; it takes an exclusive lock) on `refresh_tokens` off-peak, after 1.2 finishes. | 1 evening | +| 1.4 | **Full Terraform apply of the orca-cloud apps root** | The production plan carries four drifts from other merged work: `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` env on the auth service (#476), a skill-share log exclusion filter change, skill pressure threshold 16→8, an artifacts bucket lifecycle rule. Locally it also fails on the 1Password Cloudflare data source. | Run from CI or a machine with the 1Password account; review the four drifts as ordinary changes. | 30 min | +| 1.5 | **Alert on the pruning job** | A run that only ever times out exits 0 and reads as green. | Log metric on the job's summary event where `stopReason != "complete"`, policy on the relay channel. | 1 hour | + +## 2. Remove the shared fate between auth and relay (2.1 and 2.3 this quarter; 2.2 deferred) + +| # | Item | Why | How | Size | +|---|---|---|---|---| +| 2.1 | **Private IP for Cloud SQL, `--private-ip` on the cell proxies** (do this on the existing shared instance; do not wait for 2.2) | Cells reach the database's public IP through Cloud NAT. Dynamic port allocation (landed) raised the ceiling from 64 to 4096 ports per VM, but the NAT is still in the path and its logs are still the only place port exhaustion shows up (Finding 11). | Add a private IP to `orca-cloud-auth-db` (foundation root, orca-cloud), peer the relay VPC, switch the proxy flag in the cell template, roll. | 1–2 days | +| 2.2 | **Split the relay database from the auth database** — *DEFERRED 2026-09-04 (owner decision): revisit ~2026-11-01 once pruning is done and there is a month of alert history* | One Cloud SQL instance serves `orca_auth`, `orca_relay`, `orca_push`, `orca_skills`. The auth table's growth stalled the relay for a day (Findings 10, 13). Deferral rationale: the concrete cause is fixed (disk 250 GB, WAL 16 GB, index, pruning), 2.3 + 1.1 turn a future stall into retries, and the checkpoint/disk/headroom alerts now page. Re-open if the checkpoint-loop or connection-headroom alert fires, or a large new auth-side table is planned. | New instance for `orca_relay`; migrate with a short relay drain. Relay state is small so the cutover is minutes. | 1–2 weeks incl. rehearsal on staging | +| 2.3 | **Statement timeouts on the relay pool** (the auth pool got one in #476) | A relay query stuck behind a checkpoint fsync should fail fast and let the bounded retry take over rather than hold a pool slot for seconds. | `statement_timeout` on the relay `pg.Pool` in `cloud/apps/relay`, tuned under the lease renewal deadline. | half a day | + +## 3. Make the desktop refresh path forgiving (next 2 weeks) + +| # | Item | Why | How | Size | +|---|---|---|---|---| +| 3.1 | **Refresh-token rotation grace window** | The server revokes the whole family the first time a just-rotated token is presented again. On 2026-09-04 that turned a 30 s server slowdown into 21,605 sign-outs. A short window (e.g. 60 s) where the immediately-previous token is still accepted, returning the same new token, is standard practice. | In `apps/auth/src/tokens/refresh-tokens.ts`: accept `rotated_at` within the window, return the successor instead of revoking. Keep true reuse (outside the window, or a third presentation) as revocation. | 1 day incl. tests | +| 3.2 | **Do not retry `/refresh` with the same token on timeout** | Desktop's 30 s `CLOUD_REQUEST_TIMEOUT_MS` expiring is treated like a network error and retried with a token the server may already have rotated. | In `src/main/orca-profiles/profile-cloud-session-refresh.ts`: on timeout, re-read the stored session first, and prefer a longer single attempt for the refresh call specifically. | half a day | +| 3.3 | **Un-revoke is impossible; make sign-out recovery obvious instead** | Server-side un-revoke does not help because the desktop deletes its local token on the 401. Landed: desktop notices immediately (#18694) and the phone says "desktop signed out" (#18698). | Nothing more unless we want a re-auth deep link from the phone to the desktop. | — | + +## 4. Chronic relay issues already characterised + +| # | Item | Why | How | Size | +|---|---|---|---|---| +| 4.1 | **Cell-inventory lock contention** (partial: PR #18722 narrowed the remaining non-placement sites; `assignOnce` placement lock is the follow-up) | `postgres_retries` is a global `FOR UPDATE` over the 23-row `relay_cells` table with a 1 s `lock_timeout`; it is the floor under every 503 and every slow phone accept (Findings 2, 5; memory `relay-cell-inventory-lock-contention`). | Per-cell row locks or an advisory lock keyed by cell; move capacity counters to delta writes. Verify against real Postgres on 55440. | 1 week | +| 4.2 | **Region preference is mostly inert** | Phones request an Asia cell on ~19 % of attempts and get one ~6 % of the time; the sticky lane wins silently, so Asia users ride the US path more than intended (memory `relay-region-preference-mostly-inert`). | Let a region preference override stickiness when the preferred region has headroom; measure with `orca_relay_runtime_metrics` region counters. | 2–3 days | +| 4.3 | **Desktop lease-rotation waves** | A cell recreate seeds a fleet-wide 1006/4408 reconnect burst ~54 min later, every ~54 min (Finding 3). | Jitter the desktop control lease renewal by ±10 % so the cohort spreads out. | half a day, desktop + wire-compatible | +| 4.4 | **Raise `postgres_retries` gate calibration** | The 300 bar was recalibrated (PR #18580) but should track the post-lock-fix baseline once 4.1 lands. | Re-derive from a week of `orca_relay_postgres_transaction_retry` counts. | 1 hour | + +## 5. Observability still missing + +| # | Item | Why | How | +|---|---|---|---| +| 5.1 | **Cell crash-rate alert** | 201 process exits in 48 h with no page (Finding 6). | Log metric on `container die` for `resource.type="gce_instance"` relay cells, > 3 per 15 min per cell. In `cloud/infra/terraform/relay-observability.tf`. | +| 5.2 | **Page a person for auth alerts** | Today's four auth policies (orca-cloud #475) route to the relay Slack channel only. A repeat of 2026-09-04 deserves a page. | Add a PagerDuty/phone notification channel to `auth_alert_notification_channels` for refresh rejections and latency. | +| 5.3 | **Pruning job alert** | See 1.5. | | +| 5.4 | **Dashboard that puts the four signals side by side** | Diagnosis took hours because checkpoint state, NAT drops, auth 401 rate, and fleet controls live in four consoles. | One Cloud Monitoring dashboard: `orca_relay_cloud_sql_wal_checkpoint`, NAT `dropped_sent_packets_count`, `orca_auth_refresh_401`, summed `controls`. | + +## Landed on 2026-09-04 (for completeness) + +- Auth service cap 2 → 20 (service-level manual scaling removed); Cloud SQL disk 49 → 250 GB PD-SSD; + `max_wal_size` 16384; partial index `refresh_tokens_family_unrevoked` built concurrently by hand. +- orca-cloud #474: the above in Terraform + deploy workflow; replayed dead token answers 401 without + re-revoking or re-auditing. Deployed as `orca-cloud-auth-00031-tox` 21:04Z. +- orca-cloud #475: auth alerts (refresh 401 > 100/5 min, 429 > 20/5 min, 5xx > 10/5 min, p99 > 10 s). Applied. +- orca-cloud #476: batched `refresh_tokens` pruner (disabled), auth pool `statement_timeout` 10 s, schema + DDL on an untimed connection. +- stablyai/orca #18693: both relay NATs on dynamic port allocation 64..4096 (applied US 21:01Z, Asia 21:05Z); + alerts for Cloud SQL WAL-checkpoint loop, disk > 70 %, NAT `OUT_OF_RESOURCES` drops. Applied. +- stablyai/orca #18694: desktop learns of a revoked session immediately, panes re-fetch on mount, pairing + notice says "Sign in again to use Orca Relay". +- stablyai/orca #18698: phone shows "Desktop signed out — sign in to Orca on your desktop to reconnect" via + the WebSocket close reason (only additive slot old phones tolerate). +- Director on image 519f4914; c7 on 85bf6799; other 22 cells still on the old image (see 1.1). diff --git a/cloud/docs/relay-incident-monitor.md b/cloud/docs/relay-incident-monitor.md index 870c95dd413..8a8dfda1495 100644 --- a/cloud/docs/relay-incident-monitor.md +++ b/cloud/docs/relay-incident-monitor.md @@ -73,6 +73,13 @@ for a committed forward-recovery gate. Durable files default to gap resets the active window at the next fresh sample and preserves the prior window evidence. A threshold freeze never clears automatically. +A signal that reads missing or stale may miss up to two consecutive samples +without restarting the window. The sample still counts and is still checked +against every threshold it can read, and each tolerated gap is recorded in +`continuityEvents` with `tolerated: true`. A third consecutive miss of the same +signal, a failed collector, a runner gap, or any threshold breach restarts or +freezes as before. + A production candidate or multi-target mutation must download the exact dry-run artifact by workflow run ID and attempt. It verifies the artifact hashes and provenance, requires a green completed 15-minute state no older @@ -89,7 +96,8 @@ durably marked consumed before mutation and cannot authorize another run. | Signal | Freeze condition | | --- | ---: | | Active probe age | over 60 seconds | -| Cloud/log data age | over 180 seconds | +| Cloud Monitoring data age | over 330 seconds | +| Relay log and director admin data age | over 180 seconds | | Cell heartbeat age | over 45 seconds | | Endpoint latency | over 2,000 ms | | Cloud SQL CPU | over 80% | @@ -155,6 +163,25 @@ heartbeats, and matching live admission. separate it from today's baseline; the exhausted-retry bar (incident peak 467 vs bar 300), director concurrency, and the pool bars carry that role. Re-tighten after the fleet is on the 500 ms lock wait. +- Raised the Cloud Monitoring freshness bar from 180 s to 330 s and let a + freshness-only failure miss up to two consecutive samples without restarting + the window (2026-09-05). Basis: Google's metric list documents Cloud Run + `request_count`, `container/instance_count`, `container/cpu/utilizations`, + `container/memory/utilizations` and `container/max_request_concurrencies` as + "Sampled every 60 seconds. After sampling, data is not visible for up to 120 + seconds", and Cloud SQL `database/cpu/utilization`, + `database/memory/utilization`, `database/postgresql/num_backends`, + `database/postgresql/backends_in_wait` and `database/postgresql/deadlock_count` + as "up to 165 seconds", so the newest visible point is up to 180 s and 225 s + old respectively. Window-sum signals age further: `observedAt` is the newest + point in the 5-minute query window, so a label series that stops emitting + reads as 300 s old while its summed value is complete. The old bar sat under + all three. Production on 2026-09-04/05 restarted healthy 15-minute windows at + 181 s and 255 s (`auth.errors`, run 33928912676) and at 189 s + (`cloud_sql.lock_waits`, run 33944873727), and the last of those then blew the + 25-minute lineage cap at 1 500 004 ms, so a green fleet produced no verdict. + The director admin bar stays at 180 s and the nonzero lock-wait carry window + stays at 180 s; both publish on our own cadence. - Recalibrated the exhausted-PostgreSQL-retry freeze from 0 to 300 per five minutes (2026-09-04). Basis: #18521 cut the request-path cell-inventory lock wait from the 1 s pool `lock_timeout` to 500 ms, so contended waiters diff --git a/cloud/docs/relay-reconnect-2026-09-findings.md b/cloud/docs/relay-reconnect-2026-09-findings.md new file mode 100644 index 00000000000..580a4da84d8 --- /dev/null +++ b/cloud/docs/relay-reconnect-2026-09-findings.md @@ -0,0 +1,1002 @@ +# Relay reconnect investigation: findings and evidence + +Working notes for the 2026-09-04 mobile relay reconnect incident and the cell roll that follows. +Kept current across context compactions. Newest section first. All times UTC. Host ids are log digests, +never raw ids. Nothing here is a production mutation record unless the "Mutations" section says so. + +## Status board + +| Item | State | Where | +|---|---|---| +| PR #18565 relay accept abandonment + lease jitter + desktop rotation spread + phone probe fail-fast | Open, CI fully green again after the doc move (05:45Z), CodeRabbit + Pullfrog cleared, 3 review rounds; not merged (owner has not asked) | https://github.com/stablyai/orca/pull/18565 | +| PR #18569 monitor `relayPostgresRetryExhausted` 0 -> 300 | **Merged** 2026-09-04 ~04:20Z as 4101505b6b | https://github.com/stablyai/orca/pull/18569 | +| Same-cap `verify` of c7 (read-only) | **Passed** run 33836527159 | confirms identities, selector gen 110, rehome gen 12, protocol 1, digests | +| Monitor dry-run #1 | Froze min 5: `relay.postgres_retries` 380 > 300 | run 33836470590 | +| Monitor dry-run #2 | Green to min 13, froze 04:49Z: `director.concurrency` 76.7 > 64 (six-cell crash storm, Finding 6) | run 33837160275 | +| Monitor dry-run #3 | Froze min 3 at 05:01Z: `relay.postgres_retries` 339 > 300; no crash, concurrency 5–8 | run 33838698725 | +| Owner decision 2026-09-04 ~05:10Z | **Option B approved**: "you can raise the bar. or remove it altogether ... whats the most logical move". Kept the bar (removal would leave contention unwatched during the roll) and recalibrated from measured data. | this thread | +| PR #18580 monitor `relayPostgresRetries` 300 -> 2000 | Open, awaiting CI; mutation-checked (300 fails the new test) | https://github.com/stablyai/orca/pull/18580 | +| PR #18565 CI | Was red on `root directory guard` because this findings file sat at repo root; moved to `cloud/docs/` in 8ebff89106 | | +| PR #18580 | **Merged** 2026-09-04 05:23Z as 79d5fb469a (Pullfrog cancelled by the merge; independent Opus review requested instead, per owner) | | +| Monitor dry-run #4 | Froze min 12 at 05:37:35Z: `cell.production-gce-c27.health`/`.ready` = 0. Retries green all 12 samples under the new 2000 bar. Cause: c27 (asia-east2) container died 3x 05:37:00–05:38:01Z, Finding 6 crash class. | run 33840364323 | +| Monitor dry-run #5 | Froze at sample 1 (05:41Z): c27 health/ready still 0. MIG autoheal `recreateInstance` on c27 fired 05:38:12Z after the 3 crashes; instance RECREATING, process up with 0 controls (was ~395). Second c27 recreate in 7 h (Finding 3 seed pattern). Waiting for c27 to settle before dry-run #6. | run 33841327879 | +| Monitor dry-run #6 | **Passed** 06:06:31Z: 16 samples, no freeze (started 05:47:42Z) | run 33841783747 attempt 1 | +| c7 `canary-apply` | **Succeeded.** Dispatched 06:07:15Z; drain 06:10Z; MIG recreate 06:16–06:23Z; new image listening 06:23:42Z; verify + trust proof passed; restored to `admission=general` 06:25:21Z; canary authority sealed. c7 is on `85bf6799…`. | run 33843071283 | +| PR #18581 doc reconcile (Aug 23 figure: 2,200–3,000 raw log lines vs 1,510 on the gate metric) | **Merged** | https://github.com/stablyai/orca/pull/18581 | +| Same-cap `verify` c7 target=519f4914 rollback=85bf6799, gen 112 | **Passed** (read-only) | run 33856355648 | +| Monitor dry-run #7 (gen 112) | Froze at sample 1 (09:05:31Z): `director.errors` 4 > 0, the four 2.0 s pg-connect 500s from the 09:00 cascade still inside the 5-min delta window. Dispatched 4 min too early. | run 33856521278 | +| Monitor dry-run #8 (gen 112) | Green for 15 of 16 samples (09:09:38–09:24), froze on the final sample 09:25:22Z: `director.errors` 1 > 0. The one 500 was `/v1/admin/evacuation-status` at 09:23:50Z, 2.01 s latency = director pg-connect timeout, called by **the monitor's own collector** (`incident-monitor-sources.ts:492`). First evacuation-status 500 since Sep 1. The gate froze on a request it made itself. | run 33856905229 | +| Monitor dry-run #9 (gen 112) | Froze: c13/c23 crashed 50 s after dispatch, then c14/c20/c9 at 09:34. | run 33858650691 | +| Monitor dry-run #10 | Dispatched 09:46:13Z; froze at sample 5 (09:56:59Z): `director.errors` 12. All twelve at 09:55:17–21Z, 0.8–2.1 s latency, 10 on `/v1/regions` + 2 on `/v1/assign`; c16 and c8 crashed at 09:55:19 in the same second. A single 4-second Postgres connect stall hit director and cells together. | run 33859947207 | +| Monitor dry-run #11 | Froze at sample 2 (10:08:07Z): `director.concurrency` 79.8 > 64, the c8/c20 re-dial. They crashed 10:05:54, 3 s before the waiter's quiet check passed (log ingestion lag). | run 33861578009 | +| Monitor dry-run #12 | Dispatched 10:17:38Z after 10 quiet min; froze at sample 2 (10:19:24Z): `cell.production-gce-c16.health` 0. c16 did **not** crash (no container die, MIG NONE/HEALTHY, readiness=true throughout, `/health` 200 in 230 ms at 10:21). At 10:19:07–16 it logged "control activity renewal failed" x4 and a burst of 1006 closes, sqlFailures 1 -> 14, sqlLatencyMsMax 2588: a pg stall on the old image that did not reach the unhandled path. The probe's single fetch (30 s timeout) came back unavailable during that stall and `unavailableIsZero` turned it into health=0. | run 33862504601 | +| Monitor dry-run #13 | Green 14 of 16 samples (10:48:38–11:03), froze 11:04:43Z: c9 crashed 11:04:23, c28 11:04:25 (then looped 11:05:04, 11:05:41); c15 probe also read 0 (stall, no crash). Missed by ~90 s. **Dispatched by hand 10:48:15Z** into a 43-min crash lull (last die 10:05:54; last director 500 10:31:49). The re-armed waiter never fired: its MIG-stable check used `grep -vc True`, which exits 1 when nothing matches, so `&&` short-circuited on the *healthy* case. Waiter armed 10:20Z: 10-min quiet + every MIG stable + 60 s recheck, then dispatch, then canary c7 on green. Held at 10:24 and 10:31 by lone director `/v1/assign` 500s (2 s pg-connect stalls, no cell crash). Director 500 events since 08:46: 6 (gaps 2.7/21/31/29/7.6 min). At 10:39 the waiter was re-armed with a 6-min director-500 window (the monitor's own delta is 5 min) instead of 10, since the gate only needs the 15 min *after* dispatch to be clean. Cell crashes have stopped since 10:05 (33+ min, longest gap since 08:40). 12 dry-runs: 1 pass (#6), 11 freezes, none on a real fleet-health regression. | Cascade gaps since 09:00: 31, 2.9, 5.1, 16.1, 4.0 min (median 5); a 15-min clean window is ~28% per attempt at this rate. | | +| Monitor dry-run #14 | Dispatched 11:26:53Z by the fixed waiter (first autonomous dispatch); c14, c23, c25, c15, c24, c19 died 11:30:59–11:31:08 (six cells, 13 min after the last cascade). Froze on c8 (and others) health/ready probes. Waiter re-armed 11:06Z (grep bug fixed: `grep -c` under `|| true`), same chain; held through the 11:17 cascade and c14/c28 recreates. 13 dry-runs: 1 pass, 12 freezes. Since 08:40: 10 cascades, 75 container dies, gaps 20/31/3/5/16/4/6.5/58/13 min; only 3 windows of >=17 clean minutes existed in 2.6 h, and dry-runs hit two of them (#6 passed, #13 lost the third by 90 s). | +| Monitor dry-run #15 | Waiter armed 11:33Z (6-min director-500 window, 8-min crash window, all MIGs stable), chained canary; still holding at 12:04Z. Since 11:00: 8 cascades, 98 dies, gaps 13/13.6/3.6/14.5/4.4/6.1/3.0 min, **max gap 14.5 min**, so no 15-min clean window has existed in the last hour. 14 dry-runs: 1 pass, 13 freezes. | +| Monitor dry-run #15 verdict | Dispatched 12:28:49Z; froze at sample 2 (12:30:41Z): **12 cells** health/ready = 0 at once (c4, c5, c7, c10, c15, c16, c18, c20, c22, c25, c27, c28), including c4/c5 (0 controls all day, `/health` 200 in 190 ms a minute later) and c7 (new image). Six old-image cells also crashed 12:30:02–21. This was a fleet-wide SQL stall, not a cascade: every cell's `sqlLatencyMsMax` hit 4–6 s (c7 4865, director 5140), director pool waiting 1258, 15 cell pg-connect timeouts, director sqlFailures 92. Cloud SQL CPU 0.73, backends 160, new connections normal, memory 0.46, so the *instance* was not saturated; something held the database for ~5 s. Postgres log 12:31:23–28 shows a burst of `could not obtain lock on row in relation "relay_cells"` from NOWAIT (single-row and full-inventory) sweeps, i.e. the row locks were held during recovery. Cloud SQL transactions/min flat (~30k), reads flat, +network flat: the database was neither busy nor saturated, it was *waiting*. The stall bracket +(12:30:02–12:30:41) is where every cell's SQL max hit 4–6 s at once. Lock retries in that window were +ordinary (49/29/13 per min). Best reading: a ~5 s Postgres-side wait event shared by every session +(lock on a hot row held across a long transaction, or an instance-level pause), not CPU/IO. Cell +`sqlLatencyMsMax` was already 1.5–2.2 s fleet-wide in the four minutes before, i.e. the old cells' 1 s +`lock_timeout` plus queueing. | run 33872946111 | +| Monitor dry-run #16 | Dispatched 12:38:57Z; froze at sample 1 (12:40:11Z): `cell.production-gce-c27.latency_ms` 2071 > 2000, a fifth distinct freeze signal, the probe's own round-trip absorbing a checkpoint sync. **Loop stopped by me at 12:41Z**: with the disk in the checkpoint loop (Finding 10) no bar can hold for 15 min, so further dry-runs only burn the shared rollout lease. 16 dry-runs: 1 pass, 15 freezes. Re-arm after the disk change lands. | +| Cloud SQL checkpoint loop | **Broke on its own 12:39–12:45Z**: disk writes 48 -> 4 MB/s at 12:39 with transactions and network flat and no Cloud SQL operation; 12:40:17 checkpoint synced 0.047 s; 12:45:53 checkpoint was `time`-triggered again (first since 11:55) with sync 0.096 s and write spread over 269 s. Cause of the break unknown (most likely WAL fell back under `max_wal_size` once a burst of full-page writes aged out). It can re-enter the loop on the next large checkpoint; the disk-size fix remains the durable one. | +| Monitor dry-run #17 | Dispatched ~12:49Z (all guards clean); froze at sample 1 (12:52:05Z): `director.errors` 4, from the c9/c22 crash loop that began 12:50:34, ~90 s after dispatch. Checkpoints stayed healthy (85 ms), so this is the old image's baseline crash rate, not the disk. 17 dry-runs: 1 pass, 16 freezes. | +| Monitor dry-run #18 | **Dispatched by mistake 13:48:56Z into the outage**: my gcloud credentials expired ~13:45Z, every guard query returned empty, and the waiter's `grep -c . || true` read empty as "quiet". Froze at sample 1 (13:49:43Z) on `director.ready=0`, `auth.health=0`, and cell probes; no canary dispatched, no production mutation. All waiter loops killed at 13:51Z. Lesson: a quiet-window check must fail closed when its data source errors. Waiter had been re-armed 12:53Z. | +| Gate decision | Owner asked at 09:36Z to choose: A keep looping / B recalibrate `directorErrors` 0 -> small n / C human bypass. Ten dry-runs, four froze on this bar. Recommendation B+A. Note: B alone would not have passed #9 or #10 (cell health probes and a 12-error burst); it fixes the single-500 false freezes (#7, #8) only. | | +| Batch roll | **Deferred by plan**: roll once with the lock-fix image instead of twice. | | +| PR #18606 lock removal (root cause) | **Merged** 09:2xZ as 7b108abf71 after review, fix, re-verify; CI green | https://github.com/stablyai/orca/pull/18606 | +| Image publish for 7b108abf71 | **Done** 08:36:49Z run 33854111305: `sha256:519f4914217f08cabcdcd34825965db8473ec37c6591553a3af0d65dcdeeb183` | | +| Director deploy on 519f4914 | **Succeeded** 08:45Z run 33854355791; serving `orca-cloud-relay-00570-siv`, rollback tag on 00569-ret (also 519f4914), 00565-fes (85bf6799) still deployable. Dispatched 08:37:45Z (blue/green; prior revision 00565-fes on 85bf6799 kept as rollback). Note: `predecessor-image-digest` is a required input even with bootstrap=false; pass the serving digest. | `cloud-deploy-relay-production-director.yml` | +| c7 on new image, 2 h in | 817 controls, **0 container die** since restore (was ~1 per 15 min on old image); `sqlLatencyMsMax` still 1.0 s = lock wait unchanged, which #18606 targets | | +| Terraform alert `relay_postgres_retry_exhausted` at `> 0` | Firing continuously since #18521; recalibration not done (own change) | `cloud/infra/terraform/relay-observability.tf:447,469` | + +## Mutations performed (complete list) + +1. Merged PR #18569 to main (code/docs only). +2. Merged PR #18580 and #18581 to main (monitor bar + docs). +2b. Merged PR #18606 to main (relay lock change; no serving effect until the image is deployed). +2c. Dispatched `cloud-publish-relay-production` for 7b108abf71 (builds and pushes an image; changes nothing serving). Done: 519f4914. +2d. Dispatched `cloud-deploy-relay-production-director` on 519f4914 (preserve placement, no prune, rehome gen 12). Succeeded 08:45Z; serving revision 00570-siv. Rollback: `gcloud run services update-traffic orca-cloud-relay --region us-central1 --to-revisions orca-cloud-relay-00565-fes=100` (85bf6799, still Ready). Not needed so far. +3. 2026-09-04 06:07:15Z: dispatched `cloud-deploy-relay-production-same-cap` `canary-apply` for production-gce-c7 only (run 33843071283). Completed successfully 06:26Z: c7 isolated, drained (807 controls re-dialed), template + MIG rolled to 85bf6799, verified, restored to general admission. Selector generation advanced 110 -> 112 (isolate + restore). +4. Nothing else. Both monitor dispatches were `mode=dry-run` (read-only). The same-cap dispatch was `mode=verify` (read-only, confirmed by step gates `if: inputs.mode != 'verify'` on every mutating step). + +## Finding 6 (2026-09-04 ~05:00Z): the old cell image crashes the whole process on a Postgres connect timeout + +**This is the most important open finding.** The 23 GCE cells run image `sha256:5aedbca5…` = orca-cloud +commit e3e92d95d3 (2026-08-14). In that build `beginProof` is called as `void this.beginProof(...)`. +When `verifyCellAssignment` inside it throws (pg-pool `timeout exceeded when trying to connect`, 2 s +`connectionTimeoutMillis`), the rejection is unhandled and Node exits 1. Docker restarts the container +in ~1 s, but every control on that cell (~800 hosts) drops and re-dials `/v1/assign` at once. + +Evidence, cell c7 instance 4545742188814054238, 2026-09-04: + +``` +04:46:47.951 stderr [orca-relay] control activity renewal failed (x5) +04:46:49.527 stderr Error: timeout exceeded when trying to connect + at pg-pool/index.js:45:11 + at async PostgresPoolPressure.connect (postgres-pool-pressure.js:30:20) + at async PostgresDatabase.query (database.js:645:24) + at async RelayAssignmentStore.verifyCellAssignment (assignment-store.js:2024:22) + at async HostSessionRegistry.beginProof (host-session-registry.js:376:15) +04:46:49.527 stderr Node.js v24.19.0 +04:46:49.835 dockerd: container die … exitCode=1 image=…relay@sha256:5aed… +04:46:50.258 dockerd: container start +04:46:52.761 stdout [orca-relay] listening on https://c7.relay.onorca.dev +``` + +2026-09-04 05:36:59–05:38:01Z: c27 died 3x in 62 s plus one other instance (5464389947731541178); this froze dry-run #4 on c27's health probe. + +Fleet-wide `container die … exitCode=1` on the relay image, last 48 h: **201 events on 19 instances** +(c28 x38, c29 x37, c27 x19). Hourly counts track the lock-contention curve (peak 23/h at 21Z Sep 3). +Every one has the same `Node.js v24…` crash banner. On 2026-09-04 04:46:35–04:47:41Z six cells +(c7, c8, c19, c21, c22, c25) died within 66 s: ~4,800 hosts re-dialed, `/v1/assign` returned 16,321 +503s in one minute (baseline ~20), director concurrency hit 85 (Cloud Run cap 80), Cloud SQL +`new_connection_count` 119 -> 287/min. Fleet recovered by 04:51Z. That is what froze dry-run #2. + +Fix status: `guardSessionTask` wrapping `beginProof` landed in orca-cloud #436 (2026-08-27) and is in +the target image `sha256:85bf6799…` (main 11aace8dec). The roll is the fix. Not caused by anything in +this session: the same-cap verify finished ~04:25Z and never reached a mutating step; no compute +operations exist for those instances; heap/event-loop were flat before the crash. + +Autoheal amplifier: MIG health check is `/health` every 10 s, timeout 5 s, unhealthy after 3, so a +crash loop of ~30 s+ triggers `compute.instances.repair.recreateInstance`. All ~20 recreates in the +48 h to 2026-09-04 05:40Z were the three Asia cells (c27 x6, c28 x7, c29 x8; gcloud prints local +-07:00 times). c27 recreated 05:38:12Z after 3 crashes in 62 s; its ~395 controls went to 0 and the +monitor's `cell.production-gce-c27.health/ready` probe read 0 for the whole recreate (~several min), +freezing dry-runs #4 and #5. Each recreate also seeds a Finding 3 rotation cohort. Rolling the Asia +cells early in the batch phase should be weighed against the canary-first rule; c7 stays the canary. + +Implication for the gate: the monitor's `director.concurrency` freeze is *correctly* detecting these +crash storms. A dry-run only passes in a 15-minute window with no cell crash, roughly 1 in 3 windows +at current rates. Retrying in quiet hours is legitimate; the bar is not wrong. + +## Finding 5: `relay.postgres_retries` at 300 is 3x under today's baseline + +Retries per 5 min, cells + director, last 24 h: p50 579, p90 1039, p99 1398, max 1505; **65% of +windows over 300**. Quiet hours (03–08Z) p50 235, max 512. When the 300 bar was set (2026-08-26) +healthy bursts reached 234. Baseline has roughly tripled in 10 days. Skill notes say do not raise this +bar; I have not. Best odds for a clean 15 min are 02–04Z and 17–18Z (9/12 five-minute windows under +300 in each). + +## Finding 4: exhausted-retry bar was the wrong single blocker (fixed) + +`relayPostgresRetryExhausted: 0` never cleared after #18521 reached the director (22:12Z Sep 3): 236/236 +five-minute windows non-zero; post-#18521 p50 42 / p90 147 / max 220; Aug 23 incident peak 467. +Recalibrated to 300 in #18569 (merged). Dry-run #1 immediately revealed Finding 5 behind it. + +## Finding 3: the 00:50Z control-close wave was desktop lease rotation, not a rollout + +2026-09-04 00:49–00:51Z: 2,745 control closes on 19 instances; 1157/1632 code 1006 and 973/1030 code +4408 `control rebound` had ageMs in the 53-minute bin. Relay grants a flat 55 min lease; desktops +rebind 60–120 s early; so every host that (re)connected in the same minute rebinds as one cohort +forever. Seed: c27 MIG autoheal recreate 23:23Z (`compute.instances.repair.recreateInstance`) dumped +~420 controls. Harmonics at 23:55, 00:04, 00:25, 00:49Z. Each rebind is an `activateControl` +transaction that can take the inventory lock. Fix in #18565: relay lease 55 min ± 5 min (symmetric, +so mean rebind rate unchanged), desktop early window 1–6 min. + +## Finding 2: fleet-wide lock contention, worse on Sep 3 + +| window | 55P03 retries/h (cells) | cell sqlFailures/h | +|---|---|---| +| Sep 2 18Z – Sep 3 07Z | 660–1470 | 680–1620 | +| Sep 3 08Z–16Z | 3600–7100 | 3700–7700 | +| Sep 3 23Z | 7468 | 7585 | + +100% of sampled retries are 55P03; director phase is `cell-inventory`. Every cell pins +`sqlLatencyMsMax` at 1.0–1.2 s = the pre-#18521 1 s pool `lock_timeout`. Not load (controls flat +~26k, Cloud SQL CPU 46–53%). No `cloud-*` workflow explains the 08Z step. The lock is a global +`SELECT * FROM relay_cells FOR UPDATE` (23 rows) taken by assignment, control activation, activity +acquire, and sweeps, held to COMMIT. + +## Finding 1: root cause of the phone's 24 s hang (the original symptom) + +`acceptClient` runs four serialized Postgres calls; the fourth (`acquireActivity`) contends for the +global lock. Under contention the cell finishes after the phone's 12 s bound, then +`PendingHostDataReservation.bind` throws `host_data_reservation_already_bound` because the phone's +close already released the reservation. Every "first frame handler failed already_bound" line is that +post-mortem (31 events 23:06–01:01Z across 12 instances). Fix in #18565: abandon the accept after each +DB step once the socket is closed; new event `orca_relay_client_accept_abandoned {stage, elapsedMs}` +and metric fields `clientAcceptsAbandonedByStageDelta` / `clientAcceptAbandonedMsMax`. Phone side: +direct probe now fails fast on `reconnecting` so relay recovery is not queued behind three doomed +LAN redials (~3.5 s saved per foreground). #18518 (merged, not yet on the phone) covers the +stage-aware dial bound. + +Host 666077865f2e: stable throughout. 4408 rotation 00:27:45Z; 1006 quit 00:52:24Z on old adhoc; +sticky reassignment to c27 00:52:35Z on new build; rotation closes 01:44:55Z and 02:23:15Z with +splices intact. No drain/4404/wrong-cell. + +## Finding 7 (2026-09-04 ~05:10Z): retries bar recalibration basis (PR #18580) + +Chose 2000 over removal. The metric is the gate's own source (`orca_relay_postgres_retries` +log metric, director + cells summed per five minutes, ALIGN_DELTA 300 s): + +| window | p50 | p90 | p99 | max | > 300 | +|---|---|---|---|---|---| +| 2026-09-01 | 56 | 105 | 206 | 456 | 0% | +| 2026-09-02 | 109 | 186 | 294 | 377 | 1% | +| 2026-09-03 | 430 | 924 | 1320 | 1504 | 55% | +| 2026-09-04 to 05Z | 285 | 1012 | 1211 | 1211 | 44% | + +15-minute pass rate, last 24 h: bar 300 -> 22%, 800 -> 66%, 1000 -> 86%, 1500 -> 99%, 2000 -> 100%. +Aug 23 incident on this metric: 1510 then 646 (single windows), so retries no longer separate an +incident from baseline; exhausted (467 vs bar 300; healthy 72 h max 184), director concurrency, +and pool bars carry that role. Note: my earlier "p99 1398 / 65% over 300" in Finding 5 came from +raw log line counts; the metric-based numbers above are what the gate actually evaluates. +Baseline tripled between Sep 2 and Sep 3 with no deploy; still unexplained (Finding 2). + +## Decision needed from the owner (resolved: B) + +The same-cap roll is blocked only by the monitor gate, and the gate is blocked by `relayPostgresRetries: 300` +(Finding 5: 65% of windows breach it; even the 04:55Z quiet window hit 339). Three options: + +- A. Keep waiting for a naturally quiet 15 min. Odds per attempt ~1 in 3 in quiet hours, lower by day. + Each attempt is free and read-only. Could take hours. +- B. Recalibrate `relayPostgresRetries` from measured data, same method as #18569: 24 h p99 is 1398, the + Aug 23 incident ran 2200–3000, so ~1500 clears healthy windows with ~1.5–2x incident separation + (less margin than the exhausted bar had). Overrides the "do not raise" note in the skill facts. + Argument for: the roll being gated is the thing that reduces retries. Argument against: the bar is + doing its job of saying contention is high. +- C. A human dispatches the roll with a different gate policy. Not something I can or should do. + +My recommendation: B, with the number chosen from the table in Finding 5 and the roll following +immediately so the bar can be re-tightened after the fleet is on the 500 ms lock wait. + +## Finding 12 (2026-09-04 13:12Z): **INCIDENT IN PROGRESS. The auth service is at its 2-instance cap and rejecting 90% of desktop token calls with 429; the relay fleet has emptied.** + +Timeline: 13:04–13:06 the old-image cascades and NAT stalls drove ~1,400 desktops to re-dial. Their relay +JWTs (5-min TTL) expired mid-storm, so they hit `orca-cloud-auth` `/v1/desktop/auth/refresh` and +`/v1/desktop/auth/relay-token` together. The auth service is Cloud Run `maxScale=2`, `concurrency=80`, +1 vCPU throttled (`auth_max_instances = 2` in orca-cloud `infra/terraform-apps/environments/production.tfvars`, +applied by `deploy-auth-production.yml`). Both instances pinned at concurrency 85 from 13:02; from 13:07 +Cloud Run's front door returns **429 "no available instance"** (0 s latency, never reaches the container): +12,045 at 13:07, 54,292 at 13:08, 46,025 at 13:08, 42,529 at 13:09. Sep 3 total auth 429s: **0**. +Without a fresh relay token every desktop's `/v1/assign` gets 401 (1,433 distinct hosts 401'd, 0 got 200 +since 13:07) and every cell closes its control with `4401 relay authorization expired`. Fleet controls: +13,375 (12:55) -> 7,633 (13:08) -> **249 (13:12)**, splices 1. Auth container CPU 0.15–0.5, so the cap is +the limit, not the code. Every desktop is now in its refresh-retry loop hammering the same 2 instances: +this is a self-sustaining thundering herd and will not clear on its own. At 13:14Z: fleet **30 controls** +across 23 cells; successful relay-token issuance 5,000–6,500/min until 13:05, then 1,059 / 734 / 733 / +443 / 220 / 214 / 148 / **4** per minute through 13:13; auth 429s 54k -> 25k/min only because desktops +are backing off, not because the service recovered. Note `AUTH_MAX_INSTANCES: 2` is also hardcoded in +orca-cloud `.github/workflows/deploy-auth-production.yml` (lines 33–34), so a redeploy would re-pin it; +change both the workflow env and the tfvars. + +**Immediate mitigation (owner action, not applied):** raise the auth service's max instances. Fastest: +`gcloud run services update orca-cloud-auth --region us-central1 --max-instances 20` (or `10`, matching +the other apps' `max_instances = 10`), then land the same in `auth_max_instances` so Terraform does not +revert it. Auth is stateless behind Cloud SQL (`refresh_tokens` table); backends 210 of 400, so 20 +instances x a small pool is within budget. Also consider the desktop's refresh backoff: it re-dials on +401 immediately with no jitter, so a 429 storm sustains itself. + +**13:51Z status: my gcloud session lost auth at ~13:45Z; all production monitoring from this session is +blind until re-authenticated (`gcloud auth login`, interactive). Last confirmed state 13:40Z: fleet 0 +controls, auth maxScale 2, 7,600 auth 429/min. All autonomous dispatch loops are stopped.** + +**17:19Z–17:21Z MITIGATION APPLIED (owner said "fix it NOW").** State at 17:19Z, four hours in: all 23 +cells at 0 controls, auth 429 ~2,000/min, auth 2xx ~40/min, and the 2xx that got through took 13–28 s +(both instances saturated). Mutation 1: `gcloud run services update orca-cloud-auth --max-instances 20` +created revision `orca-cloud-auth-00018-4jc` (same image `auth@sha256:1710ff6c`, same env/concurrency, +only maxScale 2 -> 20) but the service pins traffic to `00023-qud` **by revision name**, so the new revision +was immediately `Retired` and nothing changed. Mutation 2 (17:21:30Z): `gcloud run services update-traffic +--to-revisions orca-cloud-auth-00018-4jc=100`. Lesson: the auth service's traffic block is name-pinned +(the deploy workflow does an explicit traffic switch), so a bare `services update` never reaches users. +Terraform still says `auth_max_instances = 2`; the next `deploy-auth-production.yml` run will revert this +unless the tfvars and the workflow's `AUTH_MAX_INSTANCES` are changed first. + +## Finding 13 (2026-09-04 17:19Z–18:10Z): **the auth outage is a database problem, not (only) a Cloud Run cap; `refresh_tokens` has 63 M rows and reuse-revokes scan whole families** + +Mutations this window (all online, no restarts, all by hand in project onorca-cloud): +1. 17:19Z `gcloud run services update orca-cloud-auth --max-instances 20` → new revision `00018-4jc`, but traffic is + pinned by revision name so it was `Retired`; 17:21:30Z `update-traffic --to-revisions 00018-4jc=100`. +2. Still 2 instances at 17:31Z: the SERVICE has its own `scaling.maxInstanceCount=2` in **manual scaling mode** + (`run.googleapis.com/maxScale: '2'` on service metadata, set by Terraform `infra/terraform-apps/auth.tf`), which + overrides the revision cap. `--scaling=auto` then `--max 20` at 17:31:45Z. Instances 2→20 by 17:38Z; 429s fell + 6,000/2 min → 60/2 min at 17:36Z and controls briefly reached 11. +3. Then latency, not capacity, became the wall: every refresh took 100+ s inside Postgres (desktop client timeout + is 30 s, `CLOUD_REQUEST_TIMEOUT_MS`), so 20 instances × 80 concurrency filled again with requests nobody was + waiting for, and 429s returned (~1,500/2 min from 17:40Z). +4. 17:27Z Cloud SQL disk 62 GB → 250 GB (IOPS ceiling 1,470 → ~7,500). 18:00Z `max_wal_size` 1.5 GB → 16 GB + (the checkpoint loop: `checkpoint starting: wal` every 45–60 s since 13:06Z). +5. 18:07Z `CREATE INDEX CONCURRENTLY refresh_tokens_family_unrevoked ON refresh_tokens(family_id) WHERE + revoked_at IS NULL` (an earlier attempt with `AND rotated_at IS NULL` was wrong for the revoke predicate; its + invalid remnant `refresh_tokens_family_live` was dropped). + +Evidence: `refresh_tokens` = 63.3 M live tuples, 16 GB table + 10 GB indexes; every refresh inserts a row and +nothing ever deletes (30-day TTL rows are never pruned). Query Insights 17:33–17:39Z: `UPDATE refresh_tokens SET +revoked_at = $1 WHERE family_id = $2 AND revoked_at IS NULL` = 21,000 s of execution per 6 min, ~90–120 k rows +updated per minute; io_time 15,000 s read; pg_stat_activity 180+ backends in `IO/DataFileRead` on that statement, +200 backends total for orca_auth (20 instances × pool max 10). `session-refresh-reuse-detected` audit events per +hour: ~100 all day → 8,805 (13Z), 15,511, 19,486, 24,897, 26,935 (17Z). Mechanism: a desktop's refresh times out +client-side at 30 s, the server had already rotated the token, the desktop retries with the same token, the +server calls that reuse and revokes the family (Bitmap scan on `refresh_tokens_family` + heap filter over every +row the family ever had), then the desktop retries the dead token again, and each retry re-runs the same +full-family scan (already-revoked families short-circuit nowhere). Reuse-detected 401 also **signs the user out** +on the desktop (`isOrcaCloudAuthFailure` → `clearCloudSessionIfUnchanged`), so every user who hit this during the +outage must sign in again. + +Durable fixes (orca-cloud PR in preparation on branch `auth-revoke-only-live-tokens`): `AUTH_MAX_INSTANCES` and +`auth_max_instances` → 20; Terraform disk 250 + `max_wal_size=16384`; the partial index in the schema; an +`already-revoked` short-circuit in `rotateRefreshToken` that skips the family UPDATE and the audit insert. Still +open after that: prune `refresh_tokens` (expired or revoked rows older than N days), a server-side statement +timeout shorter than the desktop's 30 s so the client and server agree on failure, and an alert on auth 429s. + +**19:11Z RESOLVED at the database layer.** `refresh_tokens_family_unrevoked` went valid at 19:11:17Z (build +18:07–19:11, two full table scans of 2.1 M blocks under load). Within 60 s: refresh latency 100 s → 0.1 s, auth 429 +→ 0, active orca_auth backends 200 → 2, checkpoints back on the 5-min timer (`checkpoint starting: time` at 18:35, +18:41, 19:00, 19:11). Director `/v1/assign` returning 200. Fleet controls 0 → 17 by 19:14Z. + +**Residual: mass sign-out.** 19:11–19:14Z: 3,857 refresh 401s from 3,829 distinct IPs, then near zero. Every one is +a desktop whose family was revoked by reuse-detection during the outage; the desktop clears its cloud session on +401 (`clearCloudSessionIfUnchanged`) and stops retrying. Those users must sign in again before the relay sees +them. Fresh `/session` sign-ins: 1, 5, 3 per minute at 19:10–19:12. Recovery of controls is now paced by users +signing in, not by infrastructure. Total `session-refresh-reuse-detected` events 13:00–19:00Z ≈ 100k, against a +~100/hour baseline. +**Affected-user count (19:22Z, from `refresh_tokens`):** 23,318 live token families revoked in the window, +**21,605 distinct users**. Only ~3,800 desktops had seen their 401 by 19:15Z; the rest were closed or asleep +and will find themselves signed out on next launch, so sign-ins will trickle for days. + +**Desktop UX finding (owner's own Mac, 19:22Z):** a revoked desktop keeps showing the account card as +"Connected" and the pairing pane as "Orca Relay: Unavailable" / `relay_control_not_active` indefinitely; the +local trace writes no relay events. Only quit + relaunch surfaced the sign-out prompt, after which sign-in → +relay-token → `/v1/assign` 200 (0.15 s) → working pairing, all within 10 s. Follow-ups: the relay coordinator's +401 path should flip the account card to reconnect-required immediately, and the pairing error should say "Sign +in again to use Relay" when the cause is an auth failure. Announcement wording: "If Relay shows Unavailable, quit +and reopen Orca, then sign in when prompted." + +orca-cloud PR #474 (branch `auth-revoke-only-live-tokens`): caps → 20, disk 250 / max_wal_size 16384 in +Terraform, partial index in the schema, `already-revoked` short-circuit. Do not deploy auth to any environment +with a large `refresh_tokens` before building the index concurrently there. + +**Wave 1 of the roadmap (2026-09-04 21:35Z onward):** five Opus agents in isolated worktrees: 3.1 grace window +(orca-cloud), 4.1+2.3 relay locks + pool timeout, 3.2+4.3 desktop refresh/jitter, 5.1+5.4 observability, +2.1 private IP (plan only, both repos). First back: stablyai/orca PR #18717 (crash alert + dashboard). Its key +finding: cell exits log to `cos_system` with uppercase `jsonPayload.MESSAGE` and `SYSLOG_IDENTIFIER=docker`, +so every earlier `jsonPayload.message:"container die"` count in this doc that read 0 was querying the wrong +field. Verified: 87 exits 12–13Z on the agent's filter, 0 in the last 6 h. Monitor dry-run 33922255205 +dispatched 21:41Z as the Roll 1 gate. +Dry-run 33922255205 froze at 21:46Z on `signal_missing cloud_sql.backends`. Cause: Cloud Monitoring published +no `num_backends` point for the auth instance between 21:40 and 21:46 (every other minute of the last 100 has +one; measured directly via the timeSeries API). A Google-side publish gap, not a database or monitor defect; +the monitor's freeze-on-missing rule is correct. The 12–13Z monitor failures were a different cause (active +probes reading 0 during the crash cascade). Re-dispatched at 21:50Z. +Dry-run #2 (33922844671) froze at 21:52:21Z on `auth.health observed 0` — verdict read from the state.json +artifact, not the log (the log only prints checkpoints). Auth served `/health` 200 continuously, including the +21:52:05 probe. Cause: the probe requires `/health` AND `/ready` on the first attempt; auth has no `/ready` +(404 by design), so every auth sample takes the forced 11 s retry, and on the third sample the retry fetch threw +at the network layer on the runner (no request reached Cloud Run) and `check()` recorded the exception as +health=false. Neither freeze was fleet health. Fix delegated (relay-ops: a thrown fetch is not a reading; auth +does not require `/ready`). **Sequencing constraint for Roll 1:** monitor evidence must be < 5 min old at +canary dispatch, so the owner's go must precede the dry-run, and a green dry-run must be followed by the +canary dispatch immediately. + +stablyai/orca PR #18719 (3.2 + 4.3, desktop): the replay engine was not the refresh function but +`RelayAuthCoordinator.scheduleRetry`, since `shouldRetryRelayConnectionError` treats any non-HTTP error +(including a refresh `TimeoutError`) as retryable and re-reads the same stored token on backoff. Fix: refresh +gets one 60 s attempt; an ambiguous failure (no status line) records the token and blocks re-sending it for +30 s (bounded, not permanent); definitive 5xx gets exactly one retry after re-reading the store; a 401 on an +ambiguously-attempted token logs `orca_cloud_refresh_possible_replay`. Lease renewal gets ±10 % full jitter +(base shrunk so the latest sample stays ≥ 90 s before expiry); server resets the full 55-min TTL on any rebind +(`host-session-registry.ts:736-743`) so early renewal is free. Verified the retry-path claim and both server +cites against main. + +2.1 private IP: orca-cloud PR #477 (foundation: servicenetworking API, /24 peering range 10.42.128.0, private +network on the instance, `prevent_destroy`; real production plan 3 add / 1 in-place change, staging unchanged) +and stablyai/orca PR #18720 (relay: `relay_cloud_sql_private_ip` variable, conditional `--private-ip` in the +cell startup template; default false renders byte-identical to main). Findings that change the plan: Google +states the private-IP change **restarts the instance** with no in-place path, and it is a one-way door (cannot +disable private IP or remove the network link). The director uses the Cloud Run built-in connector, not the +relay VPC NAT, so it never consumed the exhausted ports and is out of scope. Disabling public IP later breaks +the local proxy workflow and the director. #18720 merges (inert); #477 held for owner decision. + +4.1 + 2.3 relay: stablyai/orca PR #18722. Premise correction: #18521 and #18606 had already bounded and +narrowed most of the fleet-wide lock before today; what remained were the sticky-refresh retry (all 23 rows → +the one pinned row), reservation reconciliation (23 → the 2 involved rows), a dead pool-default fallback, and +an absolute counter write (→ delta with capacity guard). Placement (`assignOnce`) deliberately keeps the +ordered inventory lock: least-loaded selection is fleet-wide and dynamic target-only locking previously caused +cross-cell cycles; converting it to optimistic snapshot + conditional delta is the remaining 55P03 floor and a +follow-up. Pool `statement_timeout` was already 5 s but hardcoded; now env-configurable, `57014` added to the +retryable set (it was terminal before), schema DDL on an untimed max:1 pool. Independently re-ran the new and +adjacent suites here against 55440: 66/66. Harness note: 55440 is not idempotent across full runs (2 +pre-existing failures on a second run); reset the schema between runs. Rollout: director first, watch +`orca_relay_postgres_transaction_exhausted` and `cellInventoryHoldMsP95` before cells. + +#18719 first CI run failed only on `windows-host-job.win32.test.ts` (EPERM on temp-dir cleanup), a Windows +PTY test the PR does not touch and which no other recent run failed on; rerun dispatched rather than waved. + +3.1 grace window: orca-cloud PR #478 merged (not yet deployed; deploy is an owner gate because the startup +schema apply adds a nullable column to `refresh_tokens` with a brief ACCESS EXCLUSIVE). Semantics: within +`ORCA_CLOUD_REFRESH_ROTATION_GRACE_MS` (60 s default, 300 s cap, 0 = off) a re-presented rotated token gets the +SAME successor refresh token + a fresh access token, no revoke, no audit, provided the successor is still the +live head. Third presentation / outside window / revoked family: unchanged (revoke + audit). Successor plaintext +is stored sealed (AES-256-GCM, key = HKDF of the predecessor token; the DB never holds the key). Cost stated +plainly: a stolen token replayed inside 60 s is served once instead of tripping detection; DB-read + stolen +predecessor recovers the successor offline until pruned. Rotation now runs in one transaction (proved by a +forced-INSERT-failure rollback test; the 8-way race alone did not kill the non-transactional mutant). Verified +locally 27/27 incl. the Postgres suite against 55440, and CI ran it on PG 16 and 17 (4/4 each, not skipped). +Deploy wiring: env is set by BOTH Terraform and the deploy workflow, with a test pinning all three sources to +one value. **Pre-existing bug surfaced:** the deploy script strips every env var it does not own, so the +Terraform-set `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` (from #476) silently reverts to the compiled default on each +release. Latent only because both defaults are 30. Follow-up: add it to `authEnvironment` + the workflow env. + +Monitor probe fix: stablyai/orca PR #18723. A thrown fetch (DNS/TCP/TLS/8 s abort) is now "no reading" and is +re-asked once after 1 s; only a second throw is `false`. A non-ok HTTP answer is still `false` with no extra +retry. `latencyMs` is the slowest answering round trip, never a sleep. `requiresReady` is per endpoint: auth +(no `/ready` by design) is judged on `/health` + latency; director and cells unchanged. No threshold or rule +touched; `auth.ready` had no consumer. 81/81 relay-ops tests and 9/9 evidence-script tests locally. The monitor +runs at `main` head, so once merged the next dry-run uses it. + +Applying #18717 (22:10Z): the cell-exit log metric `orca_relay_cell_process_exit` is created; the alert policy +raced descriptor propagation (404) and is being retried. **Not applied, deliberately:** the dashboard. Its +targeted plan drags in `google_logging_metric.relay_snapshot[*]`, and that plan is `32 to add, 21 to destroy`: +the Terraform source adds a `region` label to every runtime metric (`EXTRACT(jsonPayload.region)`) which the +live metrics do not have, and a label change on a log metric is a delete+create. Replacing 21 live metrics +resets their history and would blank the 14 existing relay alert policies during the swap. That is +pre-existing drift in the relay root (unapplied since the region work), not something #18717 introduced. It +needs its own reviewed apply in a quiet window, ideally with the runtime-metric replacement acknowledged as +intentional. Dashboard apply waits on that. + +**Wave 1 closed 22:20Z.** Merged: orca-cloud #478 (grace window); stablyai/orca #18717 (crash alert + +dashboard TF), #18719 (desktop no-replay + jitter), #18720 (private-IP flag, off), #18722 (relay per-cell +locks + pool timeout), #18723 (monitor probe fix). Applied to production: cell-exit log metric + alert policy. +Held for owner: orca-cloud #477 private IP (restart, one-way); the dashboard apply (behind the runtime-metric +label drift); the auth deploy carrying #478; Roll 1. Every wave-1 code change now sits on main un-deployed: +the next relay image build carries #18722 + #18723's monitor runs at main head already; the next auth deploy +carries #478. + +**Landing (2026-09-04 20:50Z–21:02Z, owner: "if you are confident the cloud changes are valid, you can land them"):** + +- Merged: orca-cloud #474, #475, #476; stablyai/orca #18693, #18694, #18698. Neither repo has branch + protection or environment reviewers; `verify` / `cloud-verify` green on main after each. +- Applied to production by targeted saved plans (each plan asserted create-only / exact-attribute before + apply, via `terraform show -json`): 4 relay resources (WAL-checkpoint log metric + 3 alert policies), 8 auth + resources (3 log metrics, propagation sleep, 4 alert policies), and the us-central1 NAT + (`enable_dynamic_port_allocation` false→true, ports 64..4096). Google's docs: switching to dynamic does not + break existing connections when max ≥ 1024 and max ≥ old min; only lowering max or reverting to static is + disruptive. asia-east2 NAT deliberately left for after a US soak. +- Not applied: the untargeted apps-root plan also carries 4 unrelated drifts (`ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` + env on the auth service from #476, a skill log exclusion filter change, skill pressure threshold 16→8, an + artifacts bucket lifecycle rule) and fails on the 1Password Cloudflare data source locally. The foundation + root plans clean (disk 250 / max_wal_size already match). Those drifts belong to whoever runs the next full + apps apply in CI. +- `deploy-auth-production` on main 8034955 (run 33919143723) **succeeded 21:04Z**: serving revision + `orca-cloud-auth-00031-tox` at 100%, previous `00018-4jc`, cap 20, smoke passed on both URLs. First 15 min on + the new revision: 31×200 / 1×401 on `/refresh`, max latency 56 ms, no 5xx. The new + `refresh_token_prune_cursor` table exists, so the new schema applied. +- US NAT soak (21:01–21:06Z): 0 drops, 0 proxy dial errors, 0 cell exits, port_usage 11, sqlMax ~1.07 s. + Asia NAT then applied 21:05:28Z from the pre-verified saved plan (same three attributes). The deploy script strips env vars it does not own, so the Terraform + TTL var will not be on the new revision until the full apps apply lands; the auth code defaults to 30 d. +- Terraform locally needs `GOOGLE_OAUTH_ACCESS_TOKEN="$(gcloud auth print-access-token)"`; ADC is stale. + +**Alerting + NAT follow-ups (19:58Z, superseded by the landing block above):** + +- stablyai/orca PR #18693 (`relay-nat-ports-and-sql-alerts`): both relay NATs switch to dynamic port + allocation (64–4096 per VM); new relay-channel alerts for the Cloud SQL WAL checkpoint loop (log metric on + `checkpoint starting: wal`, > 3 per 5 min), Cloud SQL disk > 70%, and NAT `OUT_OF_RESOURCES` drops. No + existing workflow applies these resources; the PR body carries the targeted plan. +- orca-cloud PR #475 (`auth-observability-alerts`): log metrics + policies for auth refresh 401 (> 100 per 5 + min; Sep 3 baseline 20–80 per hour), 429 (> 20 per 5 min; baseline 0), 5xx (> 10 per 5 min), and Cloud Run + p99 latency > 10 s. Production routes to the relay Slack channel. +- Desktop stale auth-status fix: stablyai/orca PR #18694 (`desktop-cloud-session-revoked-status`). Main pushes + an auth-status-changed IPC when a 401 clears the session; panes re-fetch on mount; the pairing notice says + "Your Orca account session expired. Sign in again to use Orca Relay" and hides Retry. StrictMode regression + test verified red on the old guard. Does not help desktops already revoked today (session cleared before + this code); it fixes every future revocation. +- orca-cloud PR #476 (`auth-refresh-token-pruning`): batched `refresh_tokens` pruner as a scheduled Cloud Run + job (revoked rows kept 30 d, rotated rows 60 d against a 30 d TTL, 5k-row batches, 200 ms pauses, persisted + cursor, per-run budget) plus a 10 s `statement_timeout` on the auth request pool with schema DDL on an + untimed connection. Merges cleanly onto #474 and does not need its index (walks the primary key; + EXPLAIN-asserted no seq scan). CI ran the Postgres integration tests for real on PG 16 and 17. Ships + `auth_token_pruner_enabled = false` in both environments: enabling needs an image digest from a build that + contains the new entrypoint. Operating rules once enabled: monitor the run summary's `stopReason` and + `deletedRows`, not the exit code (a run that only ever times out exits 0); ~48 M rows drain in ~10 days at + 200k/hour; deleting them leaves dead tuples, so the 16 GB is not reclaimed without a separate VACUUM FULL or + pg_repack pass, which is its own change. +- Phone-side copy when the desktop is signed out: stablyai/orca PR #18698 (`phone-desktop-signed-out-reason`). + Real path traced: the director resolves the phone to the host's last cell (durable assignment row), and the + cell's `acceptClient` rejects with 4404. The only additive slot every shipped peer tolerates is the WebSocket + close *reason* (relay-hello and resolve schemas are zod strict; a new close code drops old phones off the + host-offline cadence). Desktop closes its control with reason `signed-out` only when the cloud session is gone + (null context after a 401, or explicit sign-out); quit and relaunch stay reasonless. Cell remembers it per + host for the dormant-assignment TTL, forgets on re-auth, and echoes it as the 4404 close reason; phone + renders "Desktop signed out — sign in to Orca on your desktop to reconnect" with the same retry cadence. + Old×new matrix in the PR body; nothing changes for any old peer. Merges cleanly with #18694. + +## What actually blocks the roll now (12:58Z summary for the owner) + +0. **Cloud NAT ports** (Finding 11, found 12:55Z): every us-central1 cell reaches Cloud SQL's public IP + through a NAT with the default 64 ports/VM; port_usage pinned at 64 and 1,514 dropped SYNs to + Cloud SQL:3307 in one 4-min window. This is the 2 s connect stall that kills old-image cells and is + still active after the disk loop broke. Fix: `min_ports_per_vm = 1024` (or dynamic allocation) on + `google_compute_router_nat.relay_gce` in `cloud/infra/terraform/relay-gce-foundation.tf`, targeted + apply; durable fix is a private IP on the Cloud SQL instance. Online, no VM restart. +1. **Cloud SQL disk** (Finding 10): 49 GB PD-SSD saturated since 11:58Z, checkpoint loop, fleet-wide + 4–6 s stalls every ~45 s. Fix: bigger disk and/or `max_wal_size`. Owner: `stablyai/orca-cloud` + `infra/terraform-foundation/database.tf` `google_sql_database_instance.auth` (no `disk_size`, + `disk_autoresize`, or `database_flags` set today, so Terraform is at defaults: 10 GB initial, autoresize + grew it to 49 GB). Add `disk_size = 200` (+ `disk_autoresize = true`) and optionally + `database_flags { name = "max_wal_size" value = "4096" }`; production tfvars are + `infra/terraform-foundation/environments/production.tfvars`; applied by `deploy-production.yml` in + that repo. Online, no restart for disk; `max_wal_size` is also a non-restart flag. Note Terraform + `disk_size` below the live 49 GB would be a destructive shrink, so 200 is safe and 49 is the floor. **This is now the first thing to do**; nothing else can pass a + 15-min gate while it persists, and it is also what is killing the old-image cells several times an hour. +2. **Old cell image** (Finding 6): dies on every stall. Fixed by rolling 519f4914 (canary inputs ready). +3. **Gate policy**: `directorErrors: 0` and per-cell health probes freeze on any single stall. Recalibrate + after 1 and 2, or bypass by hand for the canary. + +## Plan agreed with the owner (2026-09-04 ~06:45Z), in execution order + +Owner: "feel free to improve operations to make things more effective ... continue driving everything e2e +until this process is complete." Owner has had multi-day experiences with cell rolls and does not want a +9-hour sequential roll. + +1. **Lock-removal PR** (root cause). *Status 08:55Z: pushed as branch `relay-single-row-reservation` + (2 commits). Opus adversarial review found one real defect: `acquireActivity` moving a client-chosen + activity id across cells locked the old cell's row before the new one, cycling with placement's + ascending inventory lock (reviewer reproduced it as paired 55P03s on real Postgres; no 40P01 because + lock_timeout == deadlock_timeout == 1 s). Fixed with `lockCellRows` (ordered, 500 ms bound); census now + fails on any inline `relay_cells FOR UPDATE` outside the named helpers. Three-cell Postgres test moves + an activity high->low while the target row is held; 5/5 revert-mutants fail it. 480 SQLite tests + + tsc green. Also fixed a pre-existing test leak (`relay_cell_connection_snapshots`) that made + `assignment-control-supersession-postgres` fail on reruns. Reviewer re-verified 65569be3de: cycle + repro completes in 7 ms (was 1022 ms + paired 55P03); no remaining out-of-order pair in the store; + flagged two evasions in the new census guard, closed in the third commit (whole-statement scan, + covers query() too, mutation-checked with both evasions). Headroom Postgres test's one failure is + pre-existing on main (verified by swapping in main's store).* Make `activateControl` superseded-control cleanup, `acquireActivity` + existing-lease branch, and `changeActivity` use the existing single-row + `adjustCellReservationAtomically` instead of the 23-row `lockCellInventory`. Keep the global lock only + for placement (`resolve`/assignment) and sweeps. Real-Postgres contention test on port 55440. +2. **Faster same-cap rollout workflow.** (a) paced drain instead of `graceMs: 0` so a cell's ~800 hosts + re-dial over minutes, not one second (director cap is 5 x 80 = 400 in-flight); (b) cells in a batch run + in parallel once drains are paced; (c) post-canary batches use a short freshness check instead of a new + 15-min dry-run, since the in-job safety recheck already runs before each drain; (d) job timeout > 75 min. + Target: 22 cells in ~6 batches x ~25 min. +3. **Build image** with (1) merged, then one roll of the fleet with (2). Asia cells c27/c28/c29 first. +4. Re-tighten the monitor retries bar; recalibrate the Terraform exhausted alert. +5. Consider deleting the 55-min control lease rebind entirely (no recorded reason; liveness is the 75 s + watchdog + 90 s activity lease). Separate PR after (1) so its effect is measurable. + +## Faster same-cap rollout: design (step 2 of the plan), from reading the real limits + +What actually bounds parallelism today (measured on the c7 canary, run 33843071283): + +| step | c7 duration | bound by | +|---|---|---| +| prechecks (recheck, backend init, resolve, verify) | 43 s | none | +| isolate + drain + transition wait | 7 min | drain is `graceMs: 0`; `verify-relay-capacity-transition --activity restart-safe` polls until leases drain | +| Terraform template + MIG recreate + wait-until stable | 8 min | GCE recreate; per cell, independent | +| verify new incarnation + trust proof + restore | 1.5 min | none | + +Real constraints: (1) the director is 5 x 80 = 400 in-flight `/v1/assign`; a `graceMs: 0` drain of ~800 +hosts pins it at cap for ~2 min (observed 79.75/84.75 p99). (2) `production-cloud-sql-rollout` lease and +workflow concurrency group serialise the whole run, by design, and the per-cell job shares it via +`holder-key`. Nothing else forbids parallel cells. + +Changes, smallest first: +1. **Paced drain.** `HostSessionRegistry.drain(graceMs)` already sends `drain {graceMs}` and closes each + session after `graceMs`, but the desktop's `handleDrain` re-dials immediately regardless of graceMs + (`relay-origin-pool.ts:150-162`), so graceMs only delays the *close*, not the stampede. Fix on the + cell: stagger the drain *send* across sessions over a window (e.g. 800 sessions over 120 s = ~7/s), + which needs no desktop change and works for every desktop version in the field. New admin body field + `spreadMs` (optional, default 0 keeps today's behaviour); canary script passes `spreadMs: 120000`. + Requires the cell to be on an image with the change, so it applies to batches after the first + post-lock-fix roll, not to this one. +2. **Parallel cells in a batch.** In `cloud-deploy-relay-production-same-cap.yml` make `cell_2..cell_4` + `needs: [gate]` instead of chaining, gated on the same evidence (drop the `+75 min x wave-index` + allowance, it exists only because of chaining). Each job already takes the rollout lease with the + run's `holder-key`, so they re-enter it rather than fail. With paced drains, 4 cells x ~800 hosts + over 120 s is ~27 dials/s, well under the director cap. Raise `timeout-minutes` to 90. +3. **Post-canary batches skip the 15-min dry-run.** The in-job "Recheck aggregate SQL, pool, + reconnect, migration, and selector safety" step (`pnpm incident:relay-preflight`) already runs a + live one-shot check before each drain. For `batch-apply` with a sealed `canary-run-id` from the + same commit, accept a dry-run of any age (the canary's) plus that live recheck; keep the 15-min + requirement for `canary-apply`. Change lands in `relay-monitor-evidence.mjs verify-authority` + + `relay-production-same-cap-wave.mjs` + their node:test suites. + +**Correction after reading the cell job (07:35Z):** (2) parallel cells is not a flag flip. Each cell job +asserts the exact selector generation `expected + 2 x wave-index` and exact memberships derived from +predecessors having completed (`ISOLATED_*`/`RESTORED_*` in the job, `applyExactAdmissionSelector` +compare-and-swap), and all cells share one Terraform state lock. Making that concurrent means a batch-level +isolate/restore in the gate and a rewrite of the 650-line job's expectations. That is the multi-day trap +the owner described. Deferred. + +What is cheap and removes most of the wall-clock: (3). The per-batch 15-min dry-run costs 15 min each +*and* fails ~50% of the time on old-image crashes, which is where hours go. Implement: `batch-apply` with a +verified canary authority accepts a passed dry-run up to 6 h old and may re-use one already consumed +(the consumed-marker check exists to stop replaying stale evidence; the canary binding plus the in-job +live preflight at drain time replace it). Files: `relay-monitor-evidence.mjs` (`--after-canary`), +`incident-live-preflight-cli.ts` (same flag), the same-cap workflow + job, and both test suites. +Revised expectation: 22 cells = 6 sequential batches x ~70 min = ~7 h wall-clock but *unattended-safe* +and with one dry-run total, versus today's 6 dry-runs at ~50% each. (1) paced drain rides the lock-fix +image. + +## Recommended next steps (superseded by the plan above; kept for history) + +1. Resolve the gate decision above, then: monitor dry-run -> c7 `canary-apply` only -> verify -> stop. + Each rolled cell leaves the Finding 6 crash class. +2. Merge #18565; publish; a later same-cap roll carries it to cells. +3. Remove the global inventory lock from per-connection paths (`acquireActivity` existing-lease + branch, `activateControl` superseded-control cleanup, `changeActivity`) by using the existing + `adjustCellReservationAtomically` single-row update. Own PR, after the roll. +4. Recalibrate the Terraform alert `relay_postgres_retry_exhausted` to 300/300 s (observability root). +5. Whether to raise `relayPostgresRetries` is a human call; the data is in Finding 5. + +## Canary blast radius (read before dispatching c7) + +- What `canary-apply` does to c7, in order: isolate (selector -> migration-only, no new + assignments), `/v1/admin/drain graceMs:0` (every control on c7 re-dials the director and is + reassigned), Terraform template + MIG update to the target image, wait stable, verify new + incarnation + exact digest + protocol, prove per-host trust, restore c7 to general admission. + On any failure c7 is left isolated (migration-only) with rehome disabled; nothing else is touched. +- c7 at 05:20Z: 788 controls, 5 splices, 800 connections. So ~790 desktops re-dial once. The fleet + already absorbs this exact event 201 times / 48 h uncontrolled (Finding 6); the controlled version + isolates first, so no new assignment lands on c7 mid-roll. Expect a director concurrency blip, not + a freeze-class one (six cells at once gave 85; one cell should stay well under 64). +- Precedent: the identical workflow (pre-move, in orca-cloud) ran 9 successful `apply` canaries and + batches on 2026-08-27 (last: c20 -> 5aedbca5). Its failures that day all stopped at the read-only + "Recheck aggregate SQL..." or "Require durable rehome disabled" step, before `MUTATION_STARTED`. + The moved copy in this repo has one run: the read-only `verify` of c7 (passed, including WIF auth). +- c7 side note: MIG autoheal recreated the c7 instance four times on 2026-09-01 08:02-08:42 PDT + at ~13 min spacing. Same crash class as Finding 6 (health check failing during restart loops). + +### Canary observed effect (c7 drain, 2026-09-04 06:10Z) + +- c7 807 controls -> 0 between 06:08:52Z and 06:10:52Z. Director `/v1/assign`: 200s 32 (06:09) -> 2628 (06:10) + -> 340 (06:11); 5xx 1969 (06:10) -> 31 (06:11). Director max-concurrency p99 7.9 -> 79.75 (06:10) -> 84.75 + (06:11), i.e. at the Cloud Run cap of 80 for ~2 min. My pre-dispatch estimate ("well under 64") was wrong. +- Confounder: c10 (us-central1, instance 2803000337345335589) crashed 06:09:56Z on the old-image class + (Node.js banner + container die), so ~1,600 hosts re-dialed in the same minute, not ~800. Coincidental; + the fleet has one of these every ~15 min. +- Recovery: 06:13 903 / 06:14 1471 assign 200s from 640 distinct desktop IPs; 503s 78 -> 183 -> 29/min. + No cell crash 06:12–06:16Z. Drain step passed ~06:16Z; template/MIG apply started. +- 06:16:03–06:17:08Z, during c7's template apply (not its drain): c27 (x4) and c29 (x3) crash-looped on the + old-image pg-pool connect timeout in `beginProof`, both MIGs autoheal-recreated (c27's second recreate in + 40 min). Fleet 23 -> 21 reporting cells, controls 13286 -> 12462, assign 503s 1000/min at 06:17, director + concurrency p99 74.8. Cloud SQL CPU 0.70 max, backends 174 max (bar 250). Same multi-cell pattern occurred + at 01:31Z (4 cells) and 04:47Z (5 cells) with nothing rolling; the c7 drain's SQL load 6 min earlier may + have nudged the pool timeouts but the class is pre-existing. c7 MIG RECREATING onto new template + `…20260904061618…` = the expected image swap. +- 06:20Z: 849 assign 503s. Closes 06:19:30–06:21: 162x1006 age<5min (hosts bouncing off the recreating + c27/c29), 73x4408 + 53x1006 in the 50-min age bin (Finding 3 rotation cohort). Not roll-caused. + c7 MIG `recreating=1` on the new template since 06:16:18Z; c27 and c29 MIGs also RECREATING (autoheal). +- 06:23:16Z c7 instance restarted in place (MIG RECREATE keeps name/id relay-c7-bwjc / 4545742188814054238), + pulled `relay@sha256:85bf6799…` 06:23:37Z, listening + readiness true 06:23:42Z. Apply step passed 06:24Z; + verify step running. Isolate -> ready on new image took ~14 min end to end. +- Post-restore c7 on new image (06:25:42–06:26:42Z): controls 143 -> 273 -> 377 refilling, sqlQueries + ~1,500/30 s, `sqlLatencyMsMax` 518 -> 1003 -> 1155 ms, still 55P03 `cell-inventory` retries. So the new + image alone does not remove lock waits; the request-path 500 ms cap from #18521 applies to the director's + paths, and cell-side `acquireActivity`/`activateControl` still ride the global lock (step 3 in next steps). + Watch: does c7's sqlLatencyMsMax settle below the old 1.0–1.2 s pin once refill finishes, and does c7 stop + appearing in `container die` (the real win: guardSessionTask). +- 08:25Z (2 h after restore): c7 817 controls, 0 crashes since 06:25Z. Fleet crashes last 2 h: c27 x6, + c28 x5, all old-image Asia cells. The new image stops the crash class as predicted; it does not move + lock latency (c7 sqlLatencyMsMax 1005 ms), which is #18606's job. +- Implication for the batch phase: every drain will push director concurrency past the monitor's 64 bar + for ~1-2 min. The batch job rechecks safety *before* it drains (read-only step), so that is fine per wave, + but never run a monitor dry-run concurrently with a wave, and prefer batches of 2 over 4 until the fleet + is on the new image and the crash class is gone. + +## Post-merge dispatch plan for #18606 (image -> director -> cells) + +1. `gh workflow run cloud-publish-relay-production.yml --ref main -f mode=publish` (after the squash lands + on main). Resolve the digest by tag, never by parsing the log (it mixes relay and fence-broker digests): + `gcloud artifacts docker images describe us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay:sha- --format='value(image_summary.digest)'`. +2. Director: `gh workflow run cloud-deploy-relay-production-director.yml --ref main -f image-digest= + -f regional-placement-mode=preserve -f prune-incompatible-revisions=false -f expected-rehome-generation=12 + -f bootstrap-runtime-identity=false -f predecessor-image-digest=` + (no monitor evidence needed; requires rehome disabled at gen 12, which it is). Last run 33826514754 used + the same shape. Watch director `orca_relay_postgres_transaction_retry` per minute before/after. +3. Cells: same-cap `verify` c7 with target=, rollback=85bf6799; fresh dry-run; `canary-apply` c7; + then batches (3 per batch, Asia c27/c29/c28 first). Each batch: new dry-run unless the batch-reuse + change (design section above) has shipped. + +## Finding 8 (2026-09-04 08:40Z): ten-cell crash cascade during the director deploy, not caused by it + +Timeline: candidate revision 00570-siv created 08:38:39Z, first log 08:39:20Z; traffic still 100% on +00565-fes through 08:43 (assign logs by revision). Cell crashes: c28 (5031087219978409220) looped 08:37:55– +08:40:07 (9x), then at 08:40:20–08:40:45Z **ten** instances died within 25 s (c10 2803…, 5110…, 532…, 5464…, +7536…, 7726…, 8671…, 8928…, 8966…). All old-image `beginProof` pg-pool timeouts. Fleet controls 13,423 -> +6,157 by 08:43; assign 503s 3,912 (08:42) and 4,624 (08:43) per minute, director concurrency 85 (cap 80), +Cloud Run autoscaled 5 -> 10 instances, Cloud SQL CPU 0.55 -> 0.99. Deploy finished cleanly at 08:45Z with +the new director taking the tail of the storm; by 08:46 503s were ~30/15 s, controls 7,913 and rising, +director lock retries 29/min (vs 105–157/min pre-deploy) and exhausted 2/min (vs 65/min at 08:36). +Same class as 01:31Z (4 cells) and 04:47Z (5 cells) today; this was the biggest. c7, on the new image +since 06:25Z, did not crash. What triggered the pool timeouts fleet-wide at 08:40 is not established; Cloud +SQL CPU was 0.78–0.88 in the minutes before, the highest of the day, so the cells' 2 s connect timeout is +the plausible tipping point under a busy database. Every cell still on 5aedbca5 remains exposed to this. + +## Finding 9 (2026-09-04 08:56Z): #18606 on the director cut lock retries ~10x + +`orca_relay_postgres_retries` per 5 min, director only: 08:21–08:41 windows 419–689 (old image, incl. the +crash storm); 08:46/08:51/08:56 (new image 519f4914, refilling ~7k hosts): **61 / 69 / 54**. Exhausted: +104–178 -> **11 / 14 / 12**. Inventory hold p95 ~200 ms, max 255 ms, ~366 holds/min. Cells (still old +image) 17–44 -> 0–3, because the director no longer holds the 23-row lock on their behalf. This is the +first direct measurement of the root-cause fix under real load. Cloud SQL CPU peaked 0.99 during the +cascade and is decaying (0.86 at 08:55); the monitor freezes above 0.80, so no dry-run until it clears. + +Fourth cascade 09:00:12–09:00:18Z: c23, c8, c16, c26, c22 (five cells, 11 container-die events in 6 s, +all `5aedbca5`, exitCode 1, Node banner, pg-pool `client closed the connection` burst right before). Cloud +SQL CPU 0.84 -> 0.78 in the preceding minutes, director concurrency 18–22 (idle), so this one fired +*without* a database or director spike. Fleet had just recovered to 13,015. Cadence today: 01:31 (4), +04:47 (5), 08:40 (10), 09:00 (5), 09:31 (c13, c23), 09:34 (c23 again, c14, c20, c9; c14/c20 crash-looping), +09:39 (c21, c24), 09:55 (c16, c8), 09:59 (c20), 10:05 (c8, c20), 10:19 (c16 stalled, no crash), then a 58-min +lull, 11:04 (c9; c28 died 13x in 4 min, autoheal recreate 11:09Z, its 3rd recreate today), 11:17 (c10, c28 +again, c22, c23, c14 x9 looping; 23 dies in ~90 s; fleet 13.3k -> 10.8k), 11:31 (c14, c23, c25, c15, c24, c19), 11:34 (c20, c26, c29 x4, c14, c27 x3, c25; fleet 13.1k -> 10.3k). +Three cascades in 17 min. 11:38–11:45 c27 crash-looped 17x and c28 4x (Asia cells), c29 recreating. +11:59 (c21, c9, c10, c23), 12:02 (c19; 4,109 assign 503s that minute, mostly hosts bouncing off the +recreating cells, code 1006 age<5min x217), 12:09–12:12 (c19, c27 x6, c28 x5, c13, c22, c15, c26, c14; +8 cells, c27/c28 recreating again). Cloud SQL CPU 0.62–0.85 through it. 12:20 (six more cells). Cascade +cadence since 11:00 is now ~every 8 min; the waiter has held correctly the whole time and there has been +no dispatchable window. Loop continues unattended; findings stop logging each cascade from here unless the +class changes. Every cell that has died today is on 5aedbca5; c7 (85bf6799, +5.5 h) has not. Cell dies per hour today: +01Z 5, 02Z 7, 03Z 4, 04Z 7, 05Z 4, 06Z 9, 07Z 11, 08Z 30, 09Z 26, 10Z 2, 11Z 68+ (to 11:42). +Director concurrency pinned at 85 for 09:32–09:33; 503s 4,141 and 4,396 per minute. 09:39: c21, c24 +(2,870 503s). Crashes per instance 08:10–09:40Z: c28 x14, c27 x5, c23 x5, c22 x4, c14 x4, c13/c20 x3, +then c26/c9/c24/c16/c8 x2. Mean gap between cascades since 08:40: ~12 min. Every 15-min gate attempt +now has well under even odds; the c7-style canary that ends this needs a gate it can pass. The old image is now cascading roughly hourly regardless of load; the +only cell on a fixed image (c7) has 0 crashes in 2.5 h across all four. + +Director 500s: 4 in the 09:00 window, all 2.0 s latency on `/v1/assign` or `/v1/resolve` = pg-pool connect +timeout surfacing as a 500. Pre-existing (Sep 3: 03h/08h/16h one each, same 2.0 s shape; 06:09Z today on +the old image during the c7 drain). The monitor's `directorErrors: 0` bar freezes on any of these, so a +dry-run needs a 15-min window with none; at ~1 per cascade that is a real but modest constraint. + +**Gate observation (09:26Z):** `directorErrors: 0` counts every non-503 5xx on the director, including +the monitor's own admin calls. The director on 519f4914 still sees an occasional 2.0 s pg-pool connect +timeout (~1 per 20 min under today's Cloud SQL load), which surfaces as a 500 on whichever request drew +it. Two consecutive dry-runs (#7, #8) froze on exactly this: one, isolated, 2 s 500. That bar was set for +"unexpected director 5xx"; a single connect timeout that the client retries is not an incident. Candidate +recalibration (own PR, not done): `directorErrors` 0 -> 2 per 5 min, or exclude the monitor's own +user-agent. Not changing it unasked; noting that at ~3 per hour the 15-min gate passes ~1 in 2 attempts. + +**Did the director deploy make cells crash more? (checked 09:45Z)** Cell `container die` per 30 min: +06:00 9, 07:00 2, 07:30 9, **08:30 30** (director candidate 08:38, traffic 08:43–08:45; the 10-cell burst +was 08:40:20, before the move), 09:00 11, 09:30 12. Per hour today 05:4 06:9 07:11 08:30 09:23 vs Sep 3 +same hours 2/7/8. So today is 2–3x worse than yesterday and was rising before the deploy; after the deploy +it is ~11–12 per 30 min, in line with 06:00–07:30. Cloud SQL backends (~230 max) and new connections +(~5k/30 min) are flat across the deploy. Latest crash (c21 09:39:11) is `Connection terminated due to +connection timeout` with cause `Connection terminated unexpectedly` in `verifyCellAssignment` <- +`beginProof`, the same unhandled path. Conclusion: no evidence the deploy worsened it; the old image's +crash rate simply climbed all day. Director lock retries stayed ~10x lower after the deploy. + +**Checkpoint-phase check (10:00Z, negative result):** Postgres checkpoints complete every 5 min at ~:07. +Cell crashes bucketed by phase within that 5-min cycle show a mild :00–:29 s cluster today (22 of 103) +that is absent on Sep 3 (7 of 114), so checkpoints are not the trigger. Disk write bytes in cascade +minutes are at or below the median except 09:00. Cloud SQL memory 0.47, transaction rate flat. The +09:55 stall (11 director + 4 cell pg-connect timeouts in the same 4 s) came with `could not obtain lock +on row in relation "relay_cells"` from a NOWAIT sweep at 09:55:36, i.e. someone was holding the full +inventory at that moment. On the new director that can only be placement or a sweep; on the old cells it +is still every rebind. What stalls *connections* (not locks) for 2 s fleet-wide remains unexplained; +Cloud SQL is `db-custom-4-15360` REGIONAL PD_SSD 49 GB at 0.5–0.75 CPU when it happens. + +**Stall census (10:01Z):** 33 pg-connect-timeout stall events today (clusters of timeouts < 20 s apart). +Before 08:35 they were 1–9 timeouts each and 10–60 min apart; from 08:35 the big ones are 16, 22, 21, +17 timeouts and 5–30 min apart. No second-of-minute phase (start seconds spread across all buckets), so +not a fixed timer. Cloud SQL backends by state at 09:55: active peaked 42 at 09:52, idle-in-transaction +≤ 10, nothing near the 400 ceiling; memory 0.47; disk normal. Each stall is a few seconds where *new* +connections to Cloud SQL (via the auth proxy socket) time out at the 2 s `connectionTimeoutMillis`, +hitting every process that happens to need a fresh pool connection in that window. Old-image cells die +on it (unhandled), new-image director logs a 2 s 500 and continues. Root cause of the stall itself is +outside the relay code (Cloud SQL proxy or instance); not chased further here. + +## Finding 10 (2026-09-04 12:40Z): Cloud SQL disk write saturation since 11:58Z is driving the stalls + +`orca-cloud-auth-db` is `db-custom-4-15360` on a **49 GB PD-SSD** (81% used). PD-SSD performance scales +with size: 49 GB gives roughly 1,470 write IOPS and ~23 MB/s write throughput. Measured: + +| | before 11:58Z | 11:59Z onward | +|---|---|---| +| disk write MB/s | 4–6 | **30–50** (over the ~23 MB/s cap) | +| disk write IOPS | 500–800 | 800–1,475 (at the ~1,470 cap in 11:59, 12:15, 12:24, 12:34) | +| checkpoint `sync=` | 0.07–0.2 s (Sep 3 max 0.65 s, 290 checkpoints) | 2–20 s; 27 of 39 checkpoints in 12Z were >= 2 s | +| checkpoints per hour | 12 (timed, every 5 min) | 39 (WAL-triggered, every ~45 s; `write=` fell from 270 s to 30 s) | +| Cloud SQL CPU / memory | 0.5–0.8 / 0.47 | same (not the bottleneck) | + +Every 4 s+ fleet-wide SQL stall since 11:04 (11:04, 11:17, 11:31, 11:34, 12:09, 12:10, 12:18, 12:20, +12:30) sits inside a slow checkpoint `sync` window; the 12:30:49 checkpoint synced 5.88 s (longest file +5.47 s), matching the 12:30:02–41 stall. During fsync the WAL writer stalls and every session waits, which +is why the stall hit all 23 cells and the director at once regardless of the relay lock changes. The +old-image cells then die on the pool timeout; the new image survives. What raised write volume ~8x at +11:58Z is not established (autovacuum ran on every relay table 11:55–11:57 and checkpoints are being +forced by WAL volume, so a write amplifier inside Postgres is the leading candidate; relay transaction +rate and Cloud SQL network bytes were flat). This is the first cause found today that is *upstream* of +the relay code and it explains the afternoon acceleration (11Z 68 dies, 12Z 47 by 12:34). + +Corrections after digging (12:45Z): relay query volume, renewals, reconnects, and assignments per 5 min +were **flat** across 11:58 (sqlQ ~330k, renewals ~115k), so the relay did not start writing more. WAL +recycling per checkpoint went 7 -> 10–11 files (16 MB each) at 45 s intervals, i.e. WAL output rose from +~0.4 MB/s to ~4 MB/s while data-file writes rose to 30–50 MB/s; checkpoints switched from `time` to `wal` +triggered at 11:58:24. No Postgres slow-statement or "checkpoints too frequently" lines. This is +write amplification inside Postgres (full-page writes after each of the now-frequent checkpoints on +hot pages, plus autovacuum on every relay table each minute) on a disk too small for its IOPS ceiling, +not new relay load. Instance label `managed_by=terraform`, created 2026-07-09; the instance resource is +**not** in `cloud/infra/terraform` (only the database, user, and secret are, via +`local.relay_database_instance_name`), so it lives in the other Terraform root (orca-cloud, per +[[orca-cloud-terraform-split-findings]]). `storageAutoResize=true` with limit 0, so Cloud SQL will grow +the disk only when it fills, not when IOPS saturate; disk is 81% full. + +Onset precisely: the 11:55:37 `time` checkpoint wrote 67,258 buffers (10.5% of shared_buffers, the +day's largest) over 163 s and completed 11:58:24. Every checkpoint since has been `wal`-triggered at +~45 s spacing (`max_wal_size` reached), each writing 13–20k buffers with 9–11 WAL files recycled. This is a +self-sustaining loop: a checkpoint completes -> every subsequent write to a hot page emits a full-page +image into WAL -> WAL fills `max_wal_size` in ~45 s -> next checkpoint -> repeat. The relay's hot rows +(`relay_cells`, `relay_assignments`, activity leases, cell runtime) are updated tens of thousands of +times a minute, so full-page-write amplification is large. Before 11:58 the 5-min timed checkpoints kept +WAL well under the limit; a one-off larger checkpoint tipped it over and the disk's write ceiling keeps +it there. Query Insights: io_time +30% in the 12:00 bucket, lock_time flat. + +**Owning workflow / mitigation (not applied):** raise the Cloud SQL data disk (PD-SSD IOPS and MB/s scale +linearly with GB; 49 -> 200 GB roughly quadruples the ceiling, online, no restart) in the Terraform root +that owns `google_sql_database_instance` for `orca-cloud-auth-db`, applied through that root's workflow. +A second, flag-level lever is raising `max_wal_size` (default 1 GB) so timed checkpoints resume; that is +also a Cloud SQL instance setting in the owning Terraform root. Per the standing rule, not applied from +this session. Until then the fleet-wide 4–6 s stalls recur on +every slow checkpoint sync, the old-image cells die on each one, and no 15-min gate window will exist. + +## Finding 11 (2026-09-04 12:55Z): **Cloud NAT port exhaustion** on the us-central1 cells is the second stall class + +`google_compute_router_nat.relay_gce` (us-central1, `AUTO_ONLY` IPs, no `min_ports_per_vm`, no dynamic +port allocation, i.e. the default **64 ports per VM**). `router.googleapis.com/nat/port_usage` per VM +hit **64 = the cap** in exactly the minutes the cells' Cloud SQL proxies logged `dial tcp +35.188.82.89:3307: i/o timeout` (12:20–12:22, 12:41–12:43, 12:51–12:53), and +`nat/dropped_sent_packets_count` went 0 -> 56/552/590, 82/272/133, 395/1565/1842 in those same minutes. +Hourly: port_usage max was 25–50 all of Sep 3 and until 10Z today, 64 in 11Z and 12Z; dropped packets 0 +until 11Z (219), then 5,491 in 12Z. Open NAT connections rose 400–600 -> 815–874. Every cell's Cloud SQL +traffic egresses through this NAT to the instance's public IP (the instance has no private IP: +`ipv4Enabled=true`, `privateNetwork` unset). When a VM's 64 ports fill, new TCP SYNs to 3307 are dropped, +the proxy's dial times out, and the relay pool's 2 s `connectionTimeoutMillis` fires: that is the exact +2 s stall the old image dies on and the new director surfaces as a 500. The dial timeouts hit c7 and c8 +hardest because they carry the most controls and open the most DB connections. + +What raised port demand today: each old-image crash re-opens a full pool through fresh NAT ports, the +autoheal recreates do the same, and the 55P03 retry storms keep more connections mid-transaction, so +crashes and NAT exhaustion feed each other. This is why the afternoon accelerated even after the disk +loop broke at 12:39. + +**Owning change (not applied):** `cloud/infra/terraform/relay-gce-foundation.tf` +`google_compute_router_nat.relay_gce` (this repo): set `min_ports_per_vm = 1024` (or enable +`enable_dynamic_port_allocation = true` with `max_ports_per_vm = 4096`) and, if needed, add manual NAT IPs +(each IP supplies 64,512 ports across VMs). Online change, no VM restart. The durable fix is giving the +Cloud SQL instance a **private IP** and pointing the proxy at `--private-ip`, which takes DB traffic off +NAT entirely; that is a Cloud SQL instance change in the orca-cloud foundation root plus a startup-script +flag here. Per the standing rule, not applied from this session. + +Direct proof: `resource.type="nat_gateway" AND jsonPayload.allocation_status="DROPPED"` shows **1,514 +dropped allocations to 35.188.82.89:3307** in 12:50–12:54 alone, every one of them the Cloud SQL public +IP. The NAT has zero manual IPs (AUTO_ONLY) and no port settings in Terraform, so it is at Google's +default 64 ports/VM. No workflow in this repo applies `relay-gce-foundation.tf` broadly (the roll +workflows apply cell templates with `-target`), so the NAT change needs a targeted apply of +`google_compute_router_nat.relay_gce`, which is an owner-run Terraform step. + +Original write-up of the symptom before the NAT correlation follows. + +The 12:50:30–12:50:50 stall (every cell 3.7–3.9 s SQL max, six old-image cells died) happened with +checkpoints healthy (85 ms) and disk at 6 MB/s, so it is not Finding 10. The cells' Cloud SQL Auth Proxy +logged `failed to connect to instance: dial error: dial tcp 35.188.82.89:3307: i/o timeout`. Count of +those per hour today: 08Z 1, 11Z 15, **12Z 416**; all of Sep 3: 4. Cloud SQL `up`/backends/connections +did not blip. So new TCP connections to the instance's public IP on 3307 are timing out from the cells' +proxies in bursts, which is exactly the "2 s connect timeout" the old image dies on. Query Insights for +12:49–12:54 attributes 1,380 s of lock wait to the placement CTE (`WITH assignment_state AS +MATERIALIZED …`) and 469 s to the single-row reservation UPDATE: the lock queue is the *consequence* of +connections stalling mid-transaction, not the cause. Not chased further; candidates are the proxy's +connection churn under the crash loops (each recreated cell opens a fresh pool) and the instance's +public-IP path. Relay code cannot fix this; it is Cloud SQL / network. Dial timeouts by minute today: 12:20 24, 12:21 +66, 12:41 22, 12:42 6, 12:51 160, 12:52 137, i.e. bursts of 20–160 s each, and they hit c7 (new image, +89 today) and c8 (93) hardest, so it is not the old image's connection churn either. Cloud SQL `up`=1 +throughout. The proxy dials the instance's public IP `35.188.82.89:3307`; a burst of i/o timeouts to a +healthy instance points at the path (public-IP egress / NAT / proxy connection limits), not at Postgres. +That is the same 2 s that the old image dies on and that the new director surfaces as a 500. + +## Roll inputs (verified by the read-only `verify` run) + +**Image census from instance templates, 2026-09-04 21:45Z (authoritative, read from `gcloud compute +instance-templates`):** 20 serving cells on `5aedbca5` (c8, c9, c10, c13–c16, c19–c29) — the image that exits +the process on a Postgres connect timeout (Finding 6); c7 on `85bf6799`; c4, c5, c17, c18 (draining / +migration-only) on `0e83408b` / `36a56b10`; c1, c2, c3, c6, c11, c12 (existing-only) on Jul/Aug images. Target +for Roll 1 is `519f4914` (director already on it). Monitor dry-run dispatched 21:45Z as the roll gate; waves +require owner go. + + +- target-image-digest `sha256:519f4914217f08cabcdcd34825965db8473ec37c6591553a3af0d65dcdeeb183` (lock fix; supersedes 85bf6799 as target) +- previous target `sha256:85bf67993869a769642995d0863f4c2b6b569c3850c2d8390ec2ca5f2b179e28` (c7 is on this; use as c7's rollback) +- rollback-image-digest `sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563` +- target/rollback rehome protocol 1 / 1; expected-rehome-generation 12; selector generation **112** (110 before the c7 canary) +- existing-only c1,c11,c12,c2,c3,c4,c5,c6; migration-only c17,c18; general c10,c13–c16,c19–c29,c7,c8,c9 +- confirmation for canary: `ROLL_RELAY_SAME_CAP production-gce-c7` +- monitor evidence is single-use and must be < 5 min old at dispatch (plus 75 min per predecessor wave) +- monitor dry-run dispatch (read-only, runs at `main` head so a merged bar change applies immediately): + `gh workflow run cloud-monitor-relay-production.yml --ref main -f mode=dry-run -f expected-selector-generation=110 + -f expected-existing-only-cells= -f expected-migration-only-cells=production-gce-c17,production-gce-c18 + -f expected-general-cells= -f migration-policy=strict -f recovery-source-cell-id=none -f capacity-cell-id=none` + +## Queries that worked (copy-paste) + +- Cell metrics: `resource.type="gce_instance" AND jsonPayload.event="orca_relay_runtime_metrics"` +- Container crashes: `resource.type="gce_instance" AND jsonPayload.MESSAGE:"container die" AND jsonPayload.MESSAGE:"relay@sha256"` +- Crash banner: `resource.type="gce_instance" AND jsonPayload.message:"Node.js v24"` +- Retries: `jsonPayload.event="orca_relay_postgres_transaction_retry"` (no resource filter to get both) +- Director lines are `textPayload`; cell lines are `jsonPayload.message` +- Cloud Run concurrency: Monitoring API `run.googleapis.com/container/max_request_concurrencies` +- Dry-run final state: download artifact `relay-monitor-dry-run--`, read `*.state.json` (the log's `schemaVersion` lines are only checkpoints, not the final verdict) + +## 2026-09-04 22:50Z onward: owner go received; driving the gates + +Owner: "sure, feel free to drive these." Sequence chosen: Roll 1 first (highest uplift), auth deploy with +#478 second, pruner enable third, label drift resolved by matching Terraform to live state, #477 still held. + +| Step | Result | +| --- | --- | +| Monitor dry-run #19 (gen 112, strict) | **Passed** 23:07:53Z, run 33927238469 attempt 1. First green since the probe fix (#18723). 16 samples, no freeze. Dispatched 22:51:33Z after confirming: 0 `container die` in 3 h, director 5xx in the last 4 h were all 503s (excluded by the `director.errors` filter). | +| c8 `canary-apply` onto 519f4914 (rollback 5aedbca5) | **Failed at 23:09:07Z before any mutation**: `relay monitor evidence provenance does not match` in `verify-authority`. Run 33928330631. Gate job passed, `cell_1 / rollout` failed on the manifest check, `seal_canary` skipped, lease released. Cause: the manifest binds `commitSha`; the dry-run ran at main `264c9ed8d2`, the canary dispatched at `--ref main` resolved to `4fab8e2f15` because unrelated PRs merged to main during the 15-minute gate. Verified no side effects: c8 MIG still on template `…c8-20260827…` (5aedbca5), stable, 25 controls; no `/v1/admin/drain` or isolate calls in the director log. | +| Constraint learned | Both workflows must run at the **same main commit**. The production environment's deployment branch policy allows only `main`, and the job gates on `github.ref == 'refs/heads/main'`, so a pinned tag/branch is not an option. Any merge to stablyai/orca main during the 15-minute dry-run invalidates the evidence. Mitigation for the retry: dispatch the canary within seconds of the green, and do not merge anything to stablyai/orca main myself during the window. A durable fix (accept evidence whose commit is an ancestor with identical workflow/script content) is a follow-up, not a same-day change to a safety check. | +| Label drift (5.x) | Resolved by dropping the `region` label from Terraform to match the 21 live metrics (stablyai/orca #18734, merged). Targeted plan asserted `27 no-op, 9 create, 0 destroy`; applied 23:11Z: 8 `orca_relay_control_*` renewal metrics that had never been applied, plus `google_monitoring_dashboard.relay_incident`. `orca_relay_controls` createTime unchanged (2026-07-13), label extractors unchanged. | +| Pruner enable (1.2) | orca-cloud #479 merged: `auth_token_pruner_enabled = true`, image digest of `00031-tox`, `max_rows_per_run = 20000`. Targeted plan asserted 9 create / 0 change / 0 destroy (job, scheduler at `41 * * * *` UTC, two service accounts, five IAM grants). **Not yet applied**: waiting until the roll canary has landed so the first hourly run does not overlap a drain. | +| Auth deploy with #478 (3.1) | Dispatched 23:13Z from orca-cloud main `f0fa4b5` (run 33928663526). Candidate startup adds nullable `successor_material` under a brief ACCESS EXCLUSIVE lock. | +| Auth deploy result | **Succeeded** 23:15:37Z: `orca-cloud-auth-00035-gos` serving 100 %, cap 20 preserved, 0 5xx. `refresh_tokens.successor_material` present (nullable text); 298 sealed successors written in the first 15 min against 924 rotations; `session-refresh-reuse-detected` at baseline (5 / 15 min). Grace window is live. | +| Monitor dry-run #20 | Froze 23:35:38Z on `runtime_power_unknown cell.production-gce-c11.powered`. Two window restarts earlier (23:24, 23:25) on `signal_stale auth.errors` (Cloud Monitoring publish lag 181–255 s vs 180 s bar). Cause: one transient rejection of the per-cell MIG GET in `readResourceInventory` yields `targetSize: null` → `runtimeKnown=false` → hard freeze. c11 is a parked existing-only cell (MIG size 0, stable) and was fine. Not fleet health. Fix delegated: stablyai/orca #18740 (retry the MIG read once, mirroring #18723). Run 33928912676. | +| Monitor dry-run #21 | **Green** 23:54Z at main `8064d1f991`, but main had moved to `0a821e5bc8` during the window; the chain re-gated instead of dispatching (the canary would have failed provenance again). Run 33930229711. | +| Monitor dry-run #22 | **Green** 00:10Z at `0a821e5bc8`; main moved to `2e80972450`. Re-gated. Run 33931177390. | +| Monitor dry-run #23 | Froze 00:18:31Z on `cell.production-gce-c29.latency_ms` 2635 > 2000, the probe's own round-trip from a US runner to asia-east2; c29 controls 17→19 and `sqlLatencyMsMax` flat ~1050 through the minute, no crash, no checkpoint stall. c29 probe max was 0 in the three previous gates, so a one-off. Run 33932092775. | +| Blocking constraint | Main receives unrelated merges every 5–10 min (23:08, 23:15, 23:17, 23:40, 23:42, …). A 15-min gate bound to an exact commit cannot be consumed under that traffic. Delegated a durable fix: `verify-authority` accepts evidence whose commit is an ancestor of the canary commit **and** has no diff on the monitor/deployer trusted paths; fails closed on shallow clones or unknown commits. Chain re-armed on dry-run #24 (run 33932679796) meanwhile. | +| Monitor dry-run #24 | Froze 00:28:00Z on `director.instances` 4 < 5. Cloud Run active-instance count read 4 for exactly one minute (00:27), 5 in every other minute for 3 h; min/max scale is pinned at 5; no new revision. A routine single-instance recycle. Not fleet health. Bar `directorInstancesMin: 5` with `latest-sum` cannot tolerate that; recalibrate to 4 or use a 3-min window minimum (follow-up, not same-day). Run 33932679796. Chain dispatched #25 (run 33933193511) at `86cd327749`. | +| Monitor dry-run #25 | **Green** 00:46Z at `86cd327749`; main moved to `8096cb2803`. Fourth green gate lost to unrelated main traffic (#19, #21, #22, #25). Run 33933193511. Chain's re-gate #26 (run 33934079533) cancelled by me. | +| Fixes merged 00:55Z | stablyai/orca #18740 (MIG inventory read retried once before `runtime_power_unknown`; 2 tests) and #18754 (`verify-authority` and the batch canary authority accept evidence sealed at an **ancestor** commit when every trusted monitor/deployer path is byte-identical; fails closed on shallow clones and unknown commits; deploy/rehome jobs now check out with `fetch-depth: 0`; 5 new tests, 18/18 pass). Reviewed both diffs; trusted-path set verified to exist on main. | +| Monitor dry-run #27 | Dispatched 00:56Z at `74ad08ec66` (first gate whose evidence the new rule can consume). Run 33934541092. Chain re-armed with the same ancestor + identical-trusted-code rule so an unrelated merge no longer forces a re-gate. | +| Monitor dry-run #27 | **Green** 01:11:35Z at `74ad08ec66`; main had moved to `38bde20121` with identical trusted code, so the new rule (#18754) let the chain dispatch. Run 33934541092. | +| c8 `canary-apply` #2 (run 33935407461) | Provenance check **passed** (first consumption of ancestor evidence). Isolate → gen 113, drain, template+MIG applied 01:14–01:22, new c8 came up on `519f4914` and `relay_capacity_transition_verified` (migration-only, image exact, heartbeat fresh) at 01:23:50. Then the step's next call, `curl --fail-with-body` to c8 `/v1/admin/runtime-status`, got a **503 with a 27-byte body** at 01:23:51 and the step exited 22. Director `cell-status` at 01:23:50.8 returned 200; c8's own logs show nothing at that second; c8 health/ready both 200 seconds later; backend HEALTHY (the health check had just flipped TIMEOUT→HEALTHY at 01:22:16 and UNKNOWN→HEALTHY at 01:23:47 as the new instance warmed). Read: a single 503 at the load-balancer/warm-up edge on a curl with no retry, on a cell that was already verified healthy one line earlier. Failsafe ran: c8 kept **migration-only**, rehome control disabled, selector gen 113. c8 is serving (40 controls at 01:39, sqlLatencyMsMax ~30 ms) on the target image, just not admitted for general traffic. Nothing to roll back. | +| Recovery | The job has an explicit resume path: `mode=rollback` with `rollback-image-digest` = the image the cell already runs skips isolate/apply, verifies, and restores general admission (`ROLLBACK_RESUME=true`). Dispatched gate #28 (run 33936966508) at gen 113 with c8 in migration-only; on green the chain dispatches that resume for c8 with rollback digest `519f4914` and target `5aedbca5` (the validator only requires them to differ). | +| Follow-up | The verify step's bare `curl --fail-with-body` needs the same "no reading is not a verdict" retry the monitor got (#18723/#18740); a 503 immediately after `verify-relay-capacity-transition` passed is not evidence of a bad cell. | +| Monitor dry-run #28 | **Green** 01:58:59Z at gen 113 with c8 in migration-only. Run 33936966508. | +| c8 recovery (run 33937756402, `mode=rollback`, rollback digest = 519f4914) | **Succeeded** 02:02Z. `ROLLBACK_RESUME=true` path: isolate/apply skipped, converged-Terraform check passed, verify passed (`relay_capacity_transition_verified` general, image `519f4914`, heartbeat fresh), activate → **gen 114**, c8 general. No restart, no drain. c8 at 43 controls, sqlLatencyMsMax 36 ms. **c8 is the second cell on 519f4914** (with c7 on 85bf6799). Because the recovery ran as `rollback`, `seal_canary` was skipped, so no canary authority exists for a `batch-apply`; the next cell runs as another `canary-apply`. | +| Merged 02:05Z | stablyai/orca #18769: bounded retries on every admin-endpoint curl/fetch in the same-cap job and the rehome/canary/verify scripts (`--retry 3 --retry-delay 2 --retry-connrefused`, per-attempt bodies to a file; script helper 2 attempts on network error or 500/502/503/504 only; 4xx never retried; 650/650 tests). Trusted-path change, so the next gate runs at a commit containing it. | +| Pruner enabled (1.2) | Terraform applied 02:06Z (8 creates, then the deploy-identity job IAM grant after a propagation 404, 9/9). Job `orca-cloud-auth-token-pruner`, image `343a0915…`, scheduler `41 * * * *` UTC, budget 20 000 rows/run. First run by hand (exec `sf5ct`): cold start 3m20s, then `stopReason: time-budget` at 480 s: 73 batches, 365 000 scanned, **1 040 deleted** (1 021 revoked, 19 expired, 0 rotated), ~6.4 s/batch of 5 000, `completedFullPass: false`. No errors, no lock-wait or checkpoint alert. Scan-bound, not budget-bound: at this pace a full pass over the table takes many hourly runs, and the row budget is never the limiter. Leave the budget alone; watch hourly runs for `stopReason` and a rising `deletedRows` as the cursor reaches the rotated backlog. | +| Monitor dry-run #29 | **Green** 02:20:58Z at gen 114, main `e2b70a5eba` (contains #18740, #18754, #18769). Run 33938052374. | +| c9 `canary-apply` (run 33938818286) | **Succeeded end to end** 02:21–02:34Z: isolate → gen 115, drain, template+MIG to `519f4914`, verify passed on the first try (retry-hardened step), trust proof, activate → **gen 116**, general. `seal_canary` **succeeded**: batch authority now exists. c9 at 38 controls, sqlLatencyMsMax 33 ms. No `container die` in 30 min. Three cells on new images (c7 `85bf6799`, c8 and c9 `519f4914`); 17 serving cells still on `5aedbca5`. | +| Monitor dry-run #30 | Dispatched 02:36Z at gen 116 (run 33939533990). On green the chain dispatches **batch 1**: `batch-apply` c10,c13,c14,c15 bound to canary run 33938818286 (sealed at gen 116, same commit `e2b70a5eba`). Preflight: all four on `5aedbca5`, MIGs stable, no crash in 20 min. Sequential cells inside the job (wave-index 0..3), each with its own isolate/drain/apply/verify/restore, so ~12 min per cell, ~50 min total. | +| Monitor dry-run #30 verdict | **Green** 02:52:15Z at gen 116, `e2b70a5eba`. | +| Batch 1 (run 33940290163) | Dispatched 02:52:27Z: `batch-apply` c10,c13,c14,c15, canary authority run 33938818286, same commit. | +| Batch 1 attempt 1 (run 33940290163) | **Failed at 02:54:39Z in the live preflight, before any mutation**: `relay live preflight failed: cloud-monitoring/signal_stale`. The step's `--retry-freshness` (5 attempts, 15 s apart, freshness-only codes) is passed only for `WAVE_INDEX != 0`; the first cell takes a single sample, so one Cloud Monitoring publish lag > 180 s at that instant fails the batch. Every candidate series was current again by the time I checked. c10 untouched (template `…c10-20260827…`, 47 controls), no selector write, gen still 116, failsafe no-op. Gate #31 dispatched 02:57Z (run 33940508865); chain re-dispatches the same batch (canary authority 33938818286 still valid: same gen 116, same commit). Fix delegated: wave 0 gets the same freshness retry. | +| Monitor dry-run #31 | **Green** 03:13:26Z at gen 116; main at `cb7f7dd11a` with identical trusted code. Run 33940508865. | +| Batch 1 attempt 2 (run 33941253533) | Dispatched 03:13:38Z: c10,c13,c14,c15, canary authority 33938818286. Runs at `cb7f7dd11a` (batch authority is accepted across the ancestor since trusted paths are unchanged). | +| Merged 03:14Z | stablyai/orca #18778: `--retry-freshness` on every same-cap wave including the first, and the retry loop now stops before the next wait would push evidence past the wave's age bound (it was checked only at entry before). Twin carve-out in the capacity job filed as a follow-up. | +| Batch 1 cell 1 (c10) | **Succeeded** 03:14–03:27Z (preflight, drain, apply, verify, restore). c13 started 03:27Z. | +| Batch 1 cell 2 (c13) | **Succeeded** 03:27–03:38Z. c14 started 03:38Z. | +| Batch 1 cell 3 (c14) | **Succeeded** 03:38–03:50Z. c15 started 03:50Z. | +| Batch 1 complete (run 33941253533) | **All four succeeded** 03:13–04:00Z: c10, c13, c14, c15 on `519f4914`, selector **gen 124**. Fleet at 936 controls, 23 cells. Two `container die` at 03:35:41/44 were **c13's new container** exiting during boot (`applyPostgresSchema` → `Connection terminated due to connection timeout`, exit 1, 2 s runtime each) because the `cloud-sql-proxy` sidecar had not finished starting; the third start at 03:35:45 succeeded and c13 has been serving since (57 controls). A boot-order race in the container spec, not a serving-cell crash. Follow-up: schema pool should wait for the proxy socket, or the container should depend on the proxy's readiness. **8 cells on new images** (c7 85bf6799; c8, c9, c10, c13, c14, c15 519f4914), 12 on `5aedbca5`: c16, c19–c26 (US), c27–c29 (Asia). | +| Monitor dry-run #32 | **Green** 04:19:50Z at gen 124, main `436ef827dd` (contains #18778). Run 33943539025. | +| c16 `canary-apply` (run 33944255902) | Dispatched 04:20:02Z. On success it seals the authority for batch 2 (c19,c20,c21,c22). | +| c16 canary (run 33944255902) | **Succeeded** 04:20–04:32Z, activate → gen 126, batch authority sealed. 9 cells on new images. | +| Monitor dry-run #33 | Failed 04:58:56Z on `continuity_deadline_exceeded` (1 500 004 ms > 1 500 000 ms). One `signal_stale cloud_sql.lock_waits` at 04:46 (189 s vs 180 s bar, Cloud Monitoring publish lag) restarted the 15-min window at sample 12; the restart could not complete inside the 25-min continuity cap. No health failure at any sample; no `container die` since c16's own boot race at 04:30. Run 33944873727. Chain re-gates. Note for recalibration: `cloudDataMaxAgeMs: 180000` vs observed Cloud Monitoring publish lag of 181–255 s has now cost three gates (#20 twice, #33). | +| Freshness recalibration | stablyai/orca #18798 (open, merge after batch 2 dispatch): `cloudDataMaxAgeMs` 180 s → 330 s, derived from Google's documented visibility delays (Cloud Run 60+120 s, Cloud SQL 60+165 s) and the 5-min window-sum query (a label series that stops emitting reads as up to 300 s old while its sum is complete, which is the 255 s `auth.errors` case) plus ~30 s collect latency. Director-admin and the lock-wait carry keep their own 180 s pins. A freshness-only failure may miss 2 consecutive samples without restarting the window; the sample still counts and is still threshold-checked; a 3rd miss, collector failure, runner gap, or any breach restarts/freezes as before. 92/92 tests. | +| Monitor dry-run #34 | **Green** 05:17:31Z at gen 126, `436ef827dd`. Run 33946093029. | +| Batch 2 (run 33946819345) | Dispatched 05:17:43Z: c19,c20,c21,c22, canary authority 33944255902 (c16). | +| Merged 05:19Z | stablyai/orca #18798 (freshness bar 330 s + two-sample tolerance). Next gate runs at a commit containing it. | +| Batch 2 cell 1 (c19) | **Succeeded** 05:19–05:32Z. c20 started. | +| Batch 2 cell 2 (c20) | **Succeeded** 05:32–05:43Z. c21 started. | +| Batch 2 cell 3 (c21) | **Succeeded** 05:43–05:59Z. c22 started. | +| Batch 2 complete (run 33946819345) | **All four succeeded** 05:17–06:12Z: c19, c20, c21, c22 on `519f4914`, selector **gen 134**. Fleet at 1 090 controls, 23 cells, refresh 401s at baseline (1–4 per 3 min). One `container die` at 06:08:45 was **c22's new container** exiting during boot (exit 1, 2 s runtime; started 06:08:43, restarted 06:08:46 and serving since), the same proxy-sidecar boot race seen on c13 and c16. No serving-cell crash. **Census: 15 of 23 serving cells on new images** (c7 `85bf6799`; c8–c10, c13–c16, c19–c22 `519f4914`), 7 on `5aedbca5`: c23–c26 (US), c27–c29 (Asia). Next: gate at gen 134 → canary c23 → batch c24,c25,c26; then canary c27 → batch c28,c29. | +| Monitor dry-run #35 | **Green** 06:32:01Z at gen 134, `b33d1972bc` (contains #18798, first gate at the 330 s freshness bar). Run 33949334606. | +| c23 `canary-apply` (run 33950075843) | Dispatched 06:32:13Z at main `b0c67eaf88` (ancestor gate SHA, identical trusted code). On success it seals the authority for batch 3 (c24,c25,c26). | +| c23 canary (run 33950075843) | **Succeeded** 06:32–06:46Z, activate → gen 136, batch authority sealed. No `container die` during boot. 16 of 23 serving cells on new images; 6 on `5aedbca5` (c24–c26 US, c27–c29 Asia). | +| Monitor dry-run #36 | Dispatched 06:46Z at gen 136, run 33950746574 (`58553bfe1c`). On green the chain dispatches batch 3 (c24,c25,c26) under canary authority 33950075843. | +| Monitor dry-run #36 result | **Green** 07:02:49Z at gen 136, `58553bfe1c`. | +| Batch 3 (run 33951468008) | Dispatched 07:03Z: c24,c25,c26, canary authority 33950075843 (c23). | +| Batch 3 cell 1 (c24) | **Succeeded** 07:04–07:18Z. c25 started. | +| Batch 3 cell 2 (c25) | **Succeeded** 07:18–07:31Z. c26 started. | +| Batch 3 complete (run 33951468008) | **All three succeeded** 07:03–07:44Z: c24, c25, c26 on `519f4914`, selector **gen 142**. Fleet at ~1 230 controls, 23 cells, refresh 401s at baseline. **Zero `container die`** during the batch (no boot race on c24–c26). **All 20 US serving cells now on new images** (c7 `85bf6799`; c8–c10, c13–c16, c19–c26 `519f4914`). Remaining on `5aedbca5`: c27, c28, c29 (asia-east2, probe hard cap 3000 ms). | +| Monitor dry-run #37 | Dispatched 07:48Z at gen 142, run 33953555224 (`4c5077d57a`). On green the chain dispatches the c27 canary (first Asia cell). | +| Monitor dry-run #37 result | **Green** 08:04:24Z at gen 142, `4c5077d57a`. | +| c27 `canary-apply` (run 33954264945) | Dispatched 08:04Z, first Asia cell (asia-east2-a). On success it seals the authority for batch 4 (c28,c29). | +| c27 canary (run 33954264945) | **Failed closed before any mutation** 08:07:25Z at "Verify exact current generation, digest, cap, and rollback point": `runtime predecessor mismatch fields=regionalRehomeProtocol`. **Operator input error, not a cell fault**: the chain script hardcoded `target-rehome-protocol=1 / rollback-rehome-protocol=1` for every cell, but `relay_region_rehome_source_cell_ids` lists only the 16 US cells (c7–c10, c13–c16, c19–c26), so the Asia startup template omits `ORCA_RELAY_REHOME_*` and c27–c29 report protocol 0 by design. `MUTATION_STARTED` never set, failsafe no-op, selector stays gen 142, c27 still serving on `5aedbca5`, no `container die`. Gate #37 evidence consumed. Fix: chain script now takes `PROTO`; Asia round dispatches with protocol 0 (the per-host trust proof step is protocol-gated and skips, as designed for non-source cells). Follow-up: the job already reads `relay_region_rehome_source_cell_ids`; it could derive the expected protocol from membership instead of trusting the operator input. | +| Monitor dry-run #38 | Dispatched 08:12Z at gen 142, run 33954621425 (`e95d247be1`). On green the chain dispatches the c27 canary with protocol 0. | +| Monitor dry-run #38 result | **Green** 08:28:36Z at gen 142, `e95d247be1`. | +| c27 `canary-apply` #2 (run 33955359385) | Dispatched 08:28Z with `target/rollback-rehome-protocol=0`. | +| c27 canary #2 (run 33955359385) | **Failed closed, no mutation** 08:31:19Z. Predecessor check passed with protocol 0; the isolate step then died at argument parsing: `production capacity target is not approved`. The same-cap job shells out to `prepare-relay-production-capacity-canary.mjs` for isolate/drain/activate, whose `PRODUCTION_CAPACITY_CELL_IDS` allowlist is the 16 US capacity cells (c7–c26), while the same-cap wave validator (`SAME_CAP_CELLS`) approves all 19 serving cells including c27–c29. The Asia cells have never been through this job (their Aug 14 rollout used the asia-topology workflow). Both the isolate step and the failsafe threw before any HTTP call, so `MUTATION_STARTED=true` was written but nothing was isolated: selector stays gen 142, c27 general and serving on `5aedbca5`, no `container die`. Gate #38 evidence consumed. Fix: stablyai/orca #18811 (`--approved-cells same-cap` on all four invocations, default unchanged for the US capacity job, census test over every `SAME_CAP_CELLS` member × isolate/drain/activate + the job's cell-shape bash block; 525/525 script tests). Sweep of the other job scripts found no further Asia blocker; gate #39 (run 33955668701) dispatched at gen 142 to prove the selector is unchanged before the next attempt. | +| Monitor dry-run #39 | **Green** 08:51:28Z at gen 142: independent proof the selector was untouched by both failed c27 attempts. Not used for dispatch (its commit predates #18811). | +| Merged 08:51Z | stablyai/orca #18811 → main `12e05203a4`. | +| Monitor dry-run #40 | Dispatched 08:51Z at gen 142 on main `12e05203a4` (contains #18811), run 33956408337. On green the chain dispatches the c27 canary, protocol 0, third attempt. | +| Monitor dry-run #40 result | **Green** 09:08:03Z at gen 142, `12e05203a4`. | +| c27 `canary-apply` #3 (run 33957151726) | Dispatched 09:08Z, protocol 0, on main containing #18811. | +| c27 canary #3 (run 33957151726) | **Failed after isolate; failsafe held** 09:17:21Z. Live check 09:26Z: c27 at 0 controls (drained), template still `…20260814235757`, c28/c29 absorbed the hosts (37 each), fleet 1 404 controls / 23 cells, refresh 401s baseline, no `container die` in 60 m. Predecessor check and allowlist passed; isolate → **gen 143** (c27 migration-only), drain sent (graceMs 0, hosts reconnected via director to c28/c29/US). Terraform plan built correctly (template replace + MIG update to `519f4914`), then `validate-relay-capacity-plan.mjs --mode same-cap-cell` rejected it: `cell plan does not contain the reviewed image and capacity`. Its same-cap rule demands exactly one `ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT` and one `ORCA_RELAY_REHOME_AUDIENCE` printf in the startup script; Asia templates omit both because c27–c29 are not rehome sources (same root as attempt 1, third US-only assumption in the job). **No apply ran**: c27 template unchanged, still `5aedbca5`, isolated and draining (drain is one-way in-process; only a restart clears it). Failsafe re-asserted migration-only at gen 143 and rehome disabled. Recovery plan: fix validator (protocol-0 path: require the rehome lines *absent*), merge, gate at gen 143, then `mode=rollback` with rollback-image=`519f4914` (the failed-canary re-entry path; accepts draining + migration-only) to restart c27 onto the target image and restore it; then single-cell canaries for c28 and c29 (batch needs ≥2 cells). | +| Plan-validator fix | stablyai/orca #18818 (merged 09:41Z → main `9f2a9a248e`): `validate-relay-capacity-plan.mjs --regional-rehome-protocol 0|1` in same-cap-cell mode; protocol 0 requires the rehome lines *absent*, protocol 1 unchanged; both plan-validation calls in the job pass `DESIRED_REHOME_PROTOCOL`; census test now validates a correct plan for every `SAME_CAP_CELLS` member at its tfvars-derived protocol. 529/529. Residual: the operator-supplied protocol is still unbound for Asia cells (no `SOURCE_CELLS` cross-check outside us-central1), so a wrong value fails late at plan validation rather than early; deriving it from membership is the checklist follow-up. | +| Monitor dry-run #41 | Dispatched 09:42Z at gen 143 (c27 expected migration-only) on main `9f2a9a248e` (contains #18811 + #18818), run 33958728141. On green: c27 recovery via `mode=rollback`, rollback-image `519f4914`, protocol 0, confirmation `ROLL_BACK_RELAY_SAME_CAP`. | +| Monitor dry-run #41 result | **Green** 09:58:51Z at gen 143, `9f2a9a248e`. | +| c27 recovery #1 (run 33959789773, `mode=rollback`) | **Failed closed, no mutation** 10:09:21Z at `Verify monitor evidence provenance`: `relay monitor dry-run authority is incomplete or stale`. The dry-run authority is valid for 5 min after `completedAt` at wave 0 (`EVIDENCE_MAX_AGE_MS`); the gate completed 09:58:51Z but the operator poller (20 s `gh run view` loop) only observed completion at 10:07:09Z during a local network outage, so the dispatch landed at 10:07:11Z, 8 m 20 s after completion. Failed before the rollout lease, isolate, or any Terraform step; c27 unchanged (migration-only, drained, `5aedbca5`, gen 143). Every prior canary dispatched ≤15 s after gate green, so this is a dispatch-latency miss, not a job defect; the freshness bound behaved as designed. | +| Monitor dry-run #42 | Dispatched 18:39Z at gen 143 on main `af82126058` (trusted paths byte-identical to `9f2a9a248e`), run 33984753269. Recovery script re-armed behind it (same `mode=rollback` onto `519f4914`, protocol 0). | +| Monitor dry-run #42 result | **Green** 18:55:47Z at gen 143, `af82126058`. | +| c27 recovery #2 (run 33985902062, `mode=rollback`) | **Failed closed, no mutation** 19:05:02Z, same `authority is incomplete or stale`. Dispatch landed 19:02:39Z, 6 m 52 s after the gate completed. Root cause of both misses is the operator laptop sleeping during the 15 min gate wait (`pmset -g log`: asleep 18:52:28Z → 19:02:17Z; the morning miss coincided with a sleep/dark-wake cycle too), so the 20 s poller never ran inside the 5 min window. Not a job or evidence defect: the freshness bound did its job. Operator fix: poller now runs under `caffeinate -i`. | +| Monitor dry-run #43 | Dispatched 19:06Z at gen 143 on main `af82126058`, run 33986121849. Recovery armed behind it under `caffeinate`. | +| Monitor dry-run #43 result | **Green** 19:22:53Z at gen 143, `af82126058`. | +| c27 recovery #3 (run 33986948522, `mode=rollback`) | **Failed closed, no mutation** 19:25:48Z. Dispatched 13 s after gate green (authority accepted this time), then the live preflight recheck failed: `relay live preflight failed: active-probe/threshold_max`. That is the 2 000 ms `endpointLatencyMs` bar on one endpoint's slowest /health or /ready round trip from the runner (8 s fetch timeout, one retry). The error names no endpoint and the job log prints none; gate #43 had zero failures across 16 samples, so this was a transient probe slow-down in the ~3 min between gate and preflight. Live probe 19:32Z from the operator: director and auth ~130–190 ms, US cells ≤540 ms, Asia cells 690–1 315 ms (c28/c29 /health ~1.3 s, the closest to the bar; c27 ~0.9 s). Existing-only cells c1–c3, c6, c11, c12 return 503 on both paths as expected (unpowered). Failed before the rollout lease, isolate, or any Terraform step; c27 unchanged. Follow-up (checklist): preflight should print the failing signal and observed value. | +| Monitor dry-run #44 | Dispatched 19:33Z at gen 143 on main `062db77118`, run 33987646501. Recovery re-armed behind it. | +| Monitor dry-run #44 result | **Frozen red** 19:50:01Z after 13 samples: `active-probe/threshold_max cell.production-gce-c27.latency_ms observed=2568 threshold=2000`. No other failure, no continuity event, no `container die` fleet-wide in 60 m. `/health` is a static JSON reply (`app.ts`), so the slow round trip was `/ready` (the probe reports the max of the two) or the path to the cell. Cloud SQL logs for 19:49:38Z–19:51:58Z show six `could not obtain lock on row in relation "relay_cells"` errors and a time-triggered checkpoint completing at 19:50:36Z (write phase 270 s, the spread target, not a stall). c27 is drained with 0 controls, so its `/ready` dependency check was the only thing it was doing. Recovery script stopped as designed (no auto re-gate). Operator probe 19:53Z: c27 and c28 both bimodal, ~0.27 s or ~0.89 s per `/health` from the US, identical shape, nothing c27-specific. Attributing the one 2.6 s sample to the same shared-DB contention that produced the lock errors is the best available reading; the retry at gate #45 tests whether it recurs. | +| Monitor dry-run #45 | Dispatched 19:53Z at gen 143 on main `062db77118`, run 33988383401. Recovery re-armed behind it. | +| Monitor dry-run #45 result | **Frozen red** 19:54:47Z after 3 samples, same signal: `cell.production-gce-c27.latency_ms observed=2668 threshold=2000`. Two gates in a row now attribute a >2 s round trip to c27 while every other cell passes. | +| c27 `/ready` tail analysis | `/health` is static; `/ready` (`relay-readiness.ts`) fetches the auth JWKS (2 s timeout) then runs `SELECT 1`, cached 10 s. Operator probes 19:57Z–20:00Z, 15 each from the US: c27 and c28 have the **same** tail (0.27 s / 0.88 s modes, then 1.3 s, then 2.17–2.27 s at the top); US cells c8/c20 sit at 0.08–0.18 s. Auth JWKS latency over the last hour: 400 requests, max 20 ms, none over 1 s. So the tail is cell→Cloud SQL (US) round trips plus the runner→Asia hop, not auth and not c27-specific; c27 is drained (0 controls) so nothing local competes. Cloud SQL `could not obtain lock on row in relation "relay_cells"` runs at 17–78 per 10 min all day (NOWAIT inventory locks, expected under placement bursts) with no spike in the failing minutes. The bar (`endpointLatencyMs` 2 000 ms, one shot per minute, max of two paths) leaves Asia cells ~10% of samples from tripping; the gate got unlucky twice on c27 and lucky on c28/c29. Not a health finding. | +| Monitor dry-run #46 | Dispatched 20:01Z at gen 143 on main `062db77118`, run 33988810139. Recovery re-armed behind it. If this also freezes on an Asia probe, the next move is a per-region latency bar (or p50 over the window) in `incident-monitor.ts`, reviewed and merged before further Asia gates rather than retrying blindly. | +| Monitor dry-run #46 result | **Frozen red** 20:07:44Z after 7 samples, third time on `cell.production-gce-c27.latency_ms` (observed 2 685). Operator 40-sample `/ready` probe per Asia cell at 20:10Z: c27 p50 0.88 s / p90 2.15 s / max 2.26 s / 6 over 2 s; c28 p50 0.88 / p90 1.25 / max 2.25 / 1 over; c29 p50 0.88 / p90 0.89 / max 1.27 / 0 over. All 200. `/ready` (`relay-readiness.ts`) fetches the auth JWKS in us-central1 then `SELECT 1` on Cloud SQL in us-central1, so an Asia cell's readiness is two trans-Pacific hops plus the runner→Asia hop; the fleet-wide 2 000 ms bar was calibrated on US cells (0.08–0.5 s). c27 being drained and idle has no local load, so this is path latency, not health. **Stopped retrying gates.** Fix in flight: per-region `cell..latency_ms` bar (us-central1 stays 2 000, asia-east2 4 000; hard faults still caught by the health/ready equal-1 checks and the 8 s probe timeout) plus attributable preflight failure messages, via review + CI before the next Asia gate. | +| Merged 20:33Z | stablyai/orca #18877 → main `a3c1d32995`: per-region `cellEndpointLatencyMs` (us-central1 2 000, asia-east2 4 000; director/auth rules and the `endpointLatencyMs` key unchanged), region carried from tfvars onto every cell expectation, preflight failures now print `source/code signal observed= threshold=`. relay-ops 95/95, cloud suite 633 + 529 + 148 green. | +| Monitor dry-run #47 | Dispatched 20:34Z at gen 143 on main `a3c1d32995` (first gate with the per-region bar), run 33989896150. Recovery re-armed behind it. | +| Monitor dry-run #47 result | **Green** 20:38:09Z at gen 143 on `a3c1d32995`: first gate under the per-region bar, 16/16 samples, no Asia latency failure. | +| c27 recovery #4 (run 33990715317, `mode=rollback`) | **Success** 20:51Z. Dispatched 13 s after gate green. Isolate re-asserted migration-only at gen 143 (already isolated, no change), Terraform applied the same-cap template `…20260905204141` and the MIG replaced the instance, new incarnation on `519f4914`, protocol 0, transition verifier passed at migration-only (1 180 assignments carried, hard cap 3 000, heartbeat fresh), then activate → **gen 144**, c27 general, verifier passed again. No `container die` fleet-wide 19:55Z–20:52Z. c27 now runs the target image; c28/c29 remain on `5aedbca5` (template `…20260814235757`). | +| Monitor dry-run #48 | Dispatched 20:53Z at gen 144 (c27 back in general, MIG = c17,c18) on main `61ebffa86e` (trusted paths identical to `a3c1d32995`), run 33991385880. On green the chain dispatches the c28 `canary-apply`, protocol 0. | +| Monitor dry-run #48 result | **Green** 21:08Z at gen 144, 16/16 samples, no Asia latency failure. Main had moved to `5cec2c2dfc`; the chain verified the trusted paths were identical to the gate commit and dispatched 12 s after green. | +| c28 canary (run 33992169289, `canary-apply`) | **Success** 21:27Z. Isolate → migration-only at **gen 145**, drain already clear, verifier passed on the old image (1 220 assignments carried, hard cap 3 000, heartbeat fresh), Terraform applied same-cap template `…20260905211352`, new incarnation on `519f4914` at protocol 0, verifier passed again at migration-only, activate → **gen 146**, c28 general, verifier passed (1 219 assignments). Seal step recorded the canary. No `container die` fleet-wide 21:08Z–21:30Z. Only c29 remains on `5aedbca5`. | +| Monitor dry-run #49 | Dispatched 21:33Z at gen 146 (c28 back in general, MIG = c17,c18) on main `dce5ebd83d` (trusted paths identical to `a3c1d32995`), run 33993075948. On green the chain dispatches the c29 `canary-apply`, protocol 0, the last Roll 1 cell. | +| Monitor dry-run #49 result | **Frozen red** 21:52:24Z, `active-probe/continuity_deadline_exceeded observed=1500005 threshold=1500000`. One continuity event at 21:41:27Z, `cloud-monitoring/collector_failed` (a Cloud Monitoring read failed, not tolerated), which reset the continuous window at sample 14; the restarted window reached 10 samples before the 25-minute lineage cap (`INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS`) expired. No health failure in any of the 25 samples, no Asia latency failure, no `container die`. Monitor-side transient, not a fleet finding. The chain re-gated automatically after its 2-minute back-off. | +| Monitor dry-run #50 | Dispatched 21:54Z at gen 146 on main `51eed5a1bc`, run 33994385666. **Green** 22:10Z, 16/16 samples. Main had moved to `d7767fb196`; trusted paths identical to `a3c1d32995`. Chain dispatched the c29 `canary-apply` (run 33995164002, protocol 0) 12 s after green. | +| c29 canary (run 33995164002, `canary-apply`) | **Success** 22:27Z. Isolate → migration-only at **gen 147**, verifier passed on the old image (1 199 assignments), Terraform applied same-cap template `…20260905221622`, new incarnation on `519f4914` at protocol 0, verifier passed at migration-only, activate → **gen 148**, c29 general, verifier passed (1 199 assignments carried). No `container die` fleet-wide 22:11Z–22:30Z. | +| **Roll 1 complete** | Image census 22:30Z from MIG templates: c8–c10, c13–c16, c19–c29 on `519f4914` (18 cells); c7 on `85bf6799` (the earlier rehearsal image, carries the same fix); existing-only c1–c6, c11, c12 and migration-only c17, c18 untouched by design. No serving cell remains on `5aedbca5`. Selector gen 148, membership unchanged from the start of the roll. Zero relay container exits fleet-wide across the roll (01:14Z–22:30Z). Gates used: #19–#50; freezes were all monitor-side (provenance, freshness, flat Asia latency bar, one Cloud Monitoring collector failure), none a fleet health finding. Roll 2 (fresh image with #18722 + #18720) is the next data-plane step and waits on the owner's private-IP window decision. | + +## Roll 2 (image `4916ed67`, 2026-09-06) + +| Step | Result | Evidence | +|---|---|---| +| Docs split | #18958 merged `3bb038a185` (findings, checklist, roadmap, Roll 2 plan). | | +| Code PR | #18959 merged `61b09b7a02` (rebase of #18565 onto main; desktop rotation change dropped since #18719 shipped a proportional version). Two Opus review rounds: round 1 caught the mobile fail-fast rejecting on any socket close (one AP flap would book the 60 s cooldown) → 2 s grace, re-armed once on `handshaking`; round 2 caught a removed jitter assertion that let a one-sided jitter pass → exact pin on the top of the band. Control lease 55 min → 6 h ± 30 min. | | +| Image publish | run 34002233801 → `sha256:4916ed676d8389f694a648e750f1112d9002d68c84a1e0c7af828d5af129de62`; mirrored to staging (run 34002326150). | | +| Staging cell smoke | **Dropped.** Staging C4 is pinned to the Asia launch digest by `relay-staging-c4-refresh-workflow.test.mjs` (with production c27–c29 tfvars and the C4 recovery workflow) and the only C4 image-refresh path pins its accepted predecessor to an older digest. Re-pinning all of it for a smoke widens into the Asia launch machinery; #18969 closed. Roll 2 follows the Roll 1 path: director first, c7 as the rehearsal cell. | | +| Director deploy | run 34002673626 **success** 01:02Z: serving `orca-cloud-relay-00575-leq` on `4916ed67`, `00574-wag` (same image) tagged `selector-rollback`, `00569-ret` (`519f4914`) still deployable. Baseline before: 1 director Postgres retry in the prior hour, 0 `container die`. | | +| c7 `verify` (read-only) | run 34002885408 dispatched 01:03Z, target `4916ed67`, rollback `85bf6799`, protocol 1, gen 148. | | diff --git a/cloud/docs/relay-roll2-plan-2026-09.md b/cloud/docs/relay-roll2-plan-2026-09.md new file mode 100644 index 00000000000..f84de39163f --- /dev/null +++ b/cloud/docs/relay-roll2-plan-2026-09.md @@ -0,0 +1,154 @@ +# Relay Roll 2 and close-out plan (2026-09-05) + +Owner-approved scope 2026-09-05: finish the relay reliability work with one more cell image roll, +deferring the Cloud SQL private-IP move (2.1, orca-cloud #477) to a separate owner decision. Roll 1 +is complete (see `relay-reconnect-2026-09-findings.md`, "Roll 1 complete"); every serving cell runs +`519f4914` except c7 on `85bf6799`. + +Estimate: about two working days of effort over one week of calendar time. The cell roll itself is +6 to 7 hours of mostly unattended wall clock, run in the US night. + +## Phase 0. Land the code (half a day, no production change) + +### 0a. Split PR #18565 + +The branch mixes three relay/mobile/desktop fixes with the operator record. Split so the record +lands regardless of how the code review goes. + +- **Docs PR** (new branch off main): `relay-reconnect-2026-09-findings.md`, + `relay-improvement-checklist-2026-09.md`, `relay-improvement-roadmap-2026-09.md`, this file. + Docs only, merge on CI green. +- **Code PR** (rebase #18565 onto main, resolve two conflicts): + - `cloud/apps/relay/src/host-session-registry.ts`: conflict with #18698 (signed-out signal). + Keep both; the accept-abandonment and lease changes are orthogonal to the signed-out path. + - `src/main/runtime/relay/relay-origin-pool.ts`: **drop this branch's version**. #18719 already + merged the desktop early-window jitter (1 to 6 min). Also drop + `relay-session-broker.test.ts` additions that only exercise the dropped change. + - Keep: relay accept abandonment (`orca_relay_client_accept_abandoned` event), relay-side lease + jitter, mobile direct-probe fail-fast, and their tests. + +### 0b. Lengthen the control lease (same code PR) + +In `cloud/apps/relay/src/host-session-registry.ts`: + +``` +CONTROL_LEASE_MS = 6 * 60 * 60 * 1000 // was 55 min +CONTROL_LEASE_JITTER_MS = 30 * 60 * 1000 // was 5 min +``` + +Why 6 h: the lease bounds how long a host stays on a cell after a missed drain and is the only +passive rebalancing; 6 h keeps both and cuts control-activation traffic on the inventory lock by +about 6x. Nothing else depends on it: the relay JWT (5 min) is refreshed by the desktop on its own +schedule and liveness is the 75 s silence watchdog. Wire-safe: the relay sends `leaseExpiresAt` in +the hello ack and old desktops schedule from that value. + +Update the comment above the constants and the three assertions in +`host-session-client-accept.test.ts` that pin the lease arithmetic. Check that nothing in +`cloud/apps/relay-ops` or the monitor thresholds assumes a 55 min rotation period (grep +`55`, `CONTROL_LEASE`, `rotation`). + +### 0c. Review and merge + +Review rounds per the standing process (Opus review, then Codex pass). Merge order: docs PR first +(no dependency), then the code PR. Record the merge SHA of the code PR; that is the Roll 2 image +source. + +## Phase 1. Build and stage the image (half a day) + +Roll 2 image = code PR merge SHA. It carries, relative to `519f4914`: + +| Change | PR | Effect | +|---|---|---| +| Per-cell inventory locks, delta counters | #18722 | Removes the global `relay_cells FOR UPDATE` behind the phone accept hang | +| Relay pool `statement_timeout` 5 s | #18722 | A relay query can no longer hang a cell | +| Accept abandonment | #18565 | Cell stops finishing accepts for phones that already closed | +| Control lease 6 h ± 30 min | #18565 | Fewer, spread-out rebinds | +| `--private-ip` proxy flag support | #18720 | Code only; flag stays unset until 2.1 | + +Steps, in order (from the findings doc's post-merge dispatch plan): + +1. `gh workflow run cloud-publish-relay-production.yml --ref main -f mode=publish`. Resolve the + digest by tag, not from the log: + `gcloud artifacts docker images describe us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay:sha- --format='value(image_summary.digest)'`. +2. Staging: `cloud-deploy-relay-staging.yml` with the new digest; paired phone plus desktop smoke + (connect, background, reconnect). Confirm `orca_relay_client_accept_abandoned` appears only when + a client closes early, and that `sqlLatencyMsMax` no longer pins at the lock timeout. +3. Director: `cloud-deploy-relay-production-director.yml -f image-digest= + -f regional-placement-mode=preserve -f prune-incompatible-revisions=false + -f expected-rehome-generation=12 -f bootstrap-runtime-identity=false + -f predecessor-image-digest=`. Blue/green; prior revision stays as rollback. + Watch director `orca_relay_postgres_transaction_retry` per minute before and after. The director + goes first so the per-cell locks are live before any cell restart burst. +4. Same-cap `verify` mode against c7 with target=, rollback=`519f4914`. Read-only. + +Go/no-go for Phase 2: director serving the new image for at least 30 min, retries per minute at or +below the pre-deploy baseline, no `container die`, no auth 5xx. + +## Phase 2. Roll the cells (one US night, mostly unattended) + +Same machinery as Roll 1: `cloud-monitor-relay-production.yml` dry-run gate, then +`cloud-deploy-relay-production-same-cap.yml`. Cells roll one at a time by design (exact selector +assertions, single Terraform state, and one cell's ~1.2k-host reconnect burst per restart). Do not +add parallelism for this roll. + +Inputs: target=, rollback=`519f4914` (c7: rollback=`85bf6799`). Selector membership is +unchanged from the end of Roll 1 (gen 148; existing-only c1–c6, c11, c12; migration-only c17, c18). + +Order: + +1. **c7 canary** (`canary-apply`, protocol 1). c7 is the rehearsal cell and the only one not on + `519f4914`. +2. **c8 canary**, then **batch c9, c10, c13, c14**. +3. **c15 canary**, then **batch c16, c19, c20, c21**. +4. **c22 canary**, then **batch c23, c24, c25, c26**. +5. **Asia c27, c28, c29** as three single canaries at protocol 0 (`PROTO=0`). Batch mode cannot + take Asia cells yet and needs at least two cells. + +Each batch needs a same-commit canary authority; each wave needs a fresh 15 min gate. Use the +chain script pattern from Roll 1 (wait gate green, check trusted-path ancestry, dispatch within 5 min, +log `CANARY `) under `caffeinate -i`. Budget: 11 to 13 min per cell plus 15 min per gate, +about 6 to 7 h total. + +Per wave checks (same as Roll 1): transition verifier passes at migration-only and again at general +with assignments carried; no `container die` fleet-wide; selector generation advances by exactly 2 +per cell. After the Asia cells: image census from MIG templates; every general cell on the new digest. + +Failure handling: a failed canary re-enters through `mode=rollback` with rollback-digest = desired +image (Roll 1 c27 pattern). A gate freeze on an Asia latency probe despite the 4 000 ms bar is a +stop-and-investigate, not a retry. Monitor-side freezes (freshness, continuity deadline) re-gate +after a 2 min back-off; the chain does this on its own. + +Record every gate and wave in the findings doc as in Roll 1. + +## Phase 3. After the roll (spread over the following week) + +- **4.4 Recalibrate the retries bar.** After one week of `orca_relay_postgres_transaction_retry` + on the new image, re-derive the `postgres_retries` monitor threshold from the new baseline + (PR against `cloud/apps/relay-ops/src/incident-monitor.ts` thresholds). About 2 h. +- **1.2 Pruner budget.** Raise `auth_token_pruner_max_rows_per_run` to the default 200k after a + clean day; watch Cloud SQL write MB/s and the checkpoint alert. Then **1.5** log metric plus + policy on `stopReason != complete`. +- **1.3 Reclaim.** Once pruner runs delete ~0 rows: `pg_repack -t refresh_tokens` off-peak (check + `pg_available_extensions` first; not `VACUUM FULL`). Confirm table, index, and `disk/utilization` + dropped. +- **Monitor residuals** already in the checklist: `probeEndpointHealth` retry decision still uses the + flat 2 000 ms bar; operator protocol unbound for Asia; `probe-relay-rehome-trust` regex. +- Update the checklist status header; tick 2.3, 4.1, 4.3 relay-side as deployed. + +## Deferred, owner decision required + +- **2.1 Private IP** (orca-cloud #477). One-way door with a Cloud SQL restart. When chosen: apply the + foundation off-peak, then a template-only change that sets the `--private-ip` proxy flag. That is + another cell roll unless bundled with a future image. +- **5.2 Paging channel** for auth alerts: needs a destination. +- **Parallel cell rolls** (2 or 3 at a time): about 1.5 days (relax exact-selector assertions to + "exact except in-flight", single coordinator Terraform apply, parallel job shape, tests). Only + worth building if more image rolls are planned after Roll 2, and only once the per-cell locks are + live so a multi-cell reconnect burst is safe. +- **2.2 Database split**: deferred to ~2026-11-01. + +## Not in this plan + +Desktop and mobile changes already merged (#18719 desktop early-window jitter and no same-token +refresh retry; #18565 mobile fail-fast once merged) ship with the next desktop and mobile releases +on their own schedules. No relay action needed. diff --git a/cloud/infra/terraform/relay-gce-cells.tf b/cloud/infra/terraform/relay-gce-cells.tf index d6b7f3351f9..a4505ba2e37 100644 --- a/cloud/infra/terraform/relay-gce-cells.tf +++ b/cloud/infra/terraform/relay-gce-cells.tf @@ -242,6 +242,7 @@ resource "google_compute_instance_template" "relay_gce_cell" { artifact_registry_host = "${var.region}-docker.pkg.dev" relay_image = each.value.image cloud_sql_proxy_image = var.relay_gce_cloud_sql_proxy_image + cloud_sql_private_ip = var.relay_cloud_sql_private_ip # Keep cell-only plans independent from unrelated database configuration drift. cloud_sql_connection_name = local.relay_database_connection_name }) diff --git a/cloud/infra/terraform/relay-gce-foundation.tf b/cloud/infra/terraform/relay-gce-foundation.tf index aab3b4579eb..a8d64b3fcea 100644 --- a/cloud/infra/terraform/relay-gce-foundation.tf +++ b/cloud/infra/terraform/relay-gce-foundation.tf @@ -42,6 +42,12 @@ resource "google_compute_router_nat" "relay_gce" { router = google_compute_router.relay_gce[0].name nat_ip_allocate_option = "AUTO_ONLY" source_subnetwork_ip_ranges_to_nat = "LIST_OF_SUBNETWORKS" + # Cells reach Cloud SQL's public IP through this NAT. The static default of 64 ports per VM + # filled during the 2026-09-04 incident and every cell's proxy dial timed out at once. + enable_dynamic_port_allocation = true + enable_endpoint_independent_mapping = false + min_ports_per_vm = 64 + max_ports_per_vm = 4096 subnetwork { name = google_compute_subnetwork.relay_gce[0].id @@ -85,6 +91,12 @@ resource "google_compute_router_nat" "relay_gce_additional" { router = google_compute_router.relay_gce_additional[each.key].name nat_ip_allocate_option = "AUTO_ONLY" source_subnetwork_ip_ranges_to_nat = "LIST_OF_SUBNETWORKS" + # Cells reach Cloud SQL's public IP through this NAT. The static default of 64 ports per VM + # filled during the 2026-09-04 incident and every cell's proxy dial timed out at once. + enable_dynamic_port_allocation = true + enable_endpoint_independent_mapping = false + min_ports_per_vm = 64 + max_ports_per_vm = 4096 subnetwork { name = google_compute_subnetwork.relay_gce_additional[each.key].id diff --git a/cloud/infra/terraform/relay-gce-startup.sh.tftpl b/cloud/infra/terraform/relay-gce-startup.sh.tftpl index f593d94e9e5..a77466f2169 100644 --- a/cloud/infra/terraform/relay-gce-startup.sh.tftpl +++ b/cloud/infra/terraform/relay-gce-startup.sh.tftpl @@ -109,6 +109,9 @@ docker run --detach \ --user 0:0 \ --volume "$${cloudsql_dir}:/cloudsql" \ '${cloud_sql_proxy_image}' \ +%{ if cloud_sql_private_ip ~} + --private-ip \ +%{ endif ~} --unix-socket=/cloudsql \ '${cloud_sql_connection_name}' diff --git a/cloud/infra/terraform/relay-observability.tf b/cloud/infra/terraform/relay-observability.tf index a8fa4276eb2..6bc938100c6 100644 --- a/cloud/infra/terraform/relay-observability.tf +++ b/cloud/infra/terraform/relay-observability.tf @@ -37,6 +37,16 @@ locals { description = "Relay PostgreSQL transactions that exhausted bounded retry." filter = "((resource.type=\"cloud_run_revision\" AND (${local.relay_service_log_filter})) OR resource.type=\"gce_instance\") AND jsonPayload.event=\"orca_relay_postgres_transaction_exhausted\"" } + cell_process_exit = { + # The docker event stream is the only per-exit line: the relay's own crash footer only + # appears for unhandled rejections, and `container start` also counts healthy first boots. + description = "Relay cell container exits, one Docker `container die` event per process exit." + filter = "resource.type=\"gce_instance\" AND logName=\"projects/${var.project_id}/logs/cos_system\" AND jsonPayload.SYSLOG_IDENTIFIER=\"docker\" AND jsonPayload.MESSAGE:\"container die\" AND jsonPayload.MESSAGE:\"name=orca-relay)\"" + } + cloud_sql_wal_checkpoint = { + description = "Cloud SQL checkpoints triggered by WAL volume instead of the timed schedule; a sustained run is the fsync loop that stalled every relay process at once on 2026-09-04." + filter = "resource.type=\"cloudsql_database\" AND resource.labels.database_id=\"${var.project_id}:${local.relay_database_instance_name}\" AND textPayload:\"checkpoint starting: wal\"" + } } relay_runtime_metrics = { @@ -201,7 +211,8 @@ resource "google_logging_metric" "relay_snapshot" { label_extractors = { role = "EXTRACT(jsonPayload.role)" cell_id = "EXTRACT(jsonPayload.cellId)" - region = "EXTRACT(jsonPayload.region)" + # No region label: adding one replaces all 21 live metrics (label change = delete+create), + # which resets history and blanks the relay alert policies during the swap. } metric_descriptor { @@ -220,12 +231,6 @@ resource "google_logging_metric" "relay_snapshot" { value_type = "STRING" description = "Durable relay cell identifier." } - - labels { - key = "region" - value_type = "STRING" - description = "Coarse Relay region." - } } bucket_options { @@ -523,3 +528,280 @@ resource "google_monitoring_alert_policy" "relay_cloud_sql_backends" { mime_type = "text/markdown" } } + +resource "google_monitoring_alert_policy" "relay_cloud_sql_checkpoint_loop" { + project = var.project_id + display_name = "Orca Relay: Cloud SQL checkpoint loop" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "WAL-triggered checkpoints above 3 in 5 minutes" + + condition_threshold { + filter = "resource.type=\"cloudsql_database\" AND metric.type=\"logging.googleapis.com/user/orca_relay_cloud_sql_wal_checkpoint\"" + comparison = "COMPARISON_GT" + threshold_value = 3 + duration = "300s" + + aggregations { + alignment_period = "300s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "Healthy operation is one timed checkpoint every 5 minutes. Repeated `checkpoint starting: wal` lines mean WAL is outrunning `max_wal_size` and every checkpoint fsync stalls all relay SQL for seconds. Check `checkpoint complete` sync= times and disk write throughput against the PD-SSD ceiling; the fix is disk size and `max_wal_size` in the Terraform root that owns the instance (orca-cloud `infra/terraform-foundation`)." + mime_type = "text/markdown" + } + + depends_on = [google_logging_metric.relay_incident] +} + +resource "google_monitoring_alert_policy" "relay_cloud_sql_disk" { + project = var.project_id + display_name = "Orca Relay: Cloud SQL disk utilization" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "Cloud SQL disk above 70%" + + condition_threshold { + filter = "resource.type=\"cloudsql_database\" AND resource.label.\"database_id\"=\"${var.project_id}:${local.relay_database_instance_name}\" AND metric.type=\"cloudsql.googleapis.com/database/disk/utilization\"" + comparison = "COMPARISON_GT" + threshold_value = 0.7 + duration = "600s" + + aggregations { + alignment_period = "300s" + per_series_aligner = "ALIGN_MAX" + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "The shared auth/relay Cloud SQL disk is filling. `refresh_tokens` is the largest table and grows without pruning; grow the disk (IOPS scale with size) before it reaches the WAL checkpoint loop, and prune revoked token rows." + mime_type = "text/markdown" + } +} + +resource "google_monitoring_alert_policy" "relay_cloud_nat_port_drops" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + display_name = "Orca Relay: Cloud NAT port exhaustion" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "NAT packets dropped for lack of ports" + + condition_threshold { + filter = "resource.type=\"nat_gateway\" AND resource.label.\"gateway_name\"=monitoring.regex.full_match(\"${local.relay_gce_name}(-.*)?\") AND metric.type=\"router.googleapis.com/nat/dropped_sent_packets_count\" AND metric.label.\"reason\"=\"OUT_OF_RESOURCES\"" + comparison = "COMPARISON_GT" + threshold_value = 0 + duration = "120s" + + aggregations { + alignment_period = "60s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + group_by_fields = ["resource.label.\"gateway_name\""] + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "Relay cells reach Cloud SQL's public IP through this NAT. Port exhaustion makes every cell's Cloud SQL Auth Proxy dial time out at once, which reads as a fleet-wide SQL stall with a healthy database. Check `nat/port_usage` per VM and raise `max_ports_per_vm` in `relay-gce-foundation.tf`, or move the database to a private IP." + mime_type = "text/markdown" + } +} + +resource "google_monitoring_alert_policy" "relay_cell_process_exit" { + project = var.project_id + display_name = "Orca Relay: cell process exits" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "Cell container exits above 3 in 15 minutes" + + condition_threshold { + filter = "resource.type=\"gce_instance\" AND metric.type=\"logging.googleapis.com/user/orca_relay_cell_process_exit\"" + comparison = "COMPARISON_GT" + threshold_value = 3 + duration = "0s" + + aggregations { + alignment_period = "900s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + group_by_fields = ["resource.label.\"instance_id\""] + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "A Relay GCE cell restarted its container more than three times in 15 minutes. Each exit drops every host and phone on that cell, and 201 exits went unpaged over 48 h on 2026-09-04. The instance hostname is `relay--`; read `jsonPayload.MESSAGE` on `cos_system` for the exit code and the container's own stderr for the stack before blaming MIG autoheal or load. A same-capacity roll is the remedy when the running image is behind." + mime_type = "text/markdown" + } + + depends_on = [google_logging_metric.relay_incident] +} + +# Why: the four signals that had to be assembled by hand during the 2026-09-04 incident. +resource "google_monitoring_dashboard" "relay_incident" { + project = var.project_id + + dashboard_json = jsonencode({ + displayName = "Orca Relay: incident overview" + mosaicLayout = { + columns = 12 + tiles = [ + { + xPos = 0 + yPos = 0 + width = 3 + height = 4 + widget = { + title = "Cloud SQL WAL checkpoints" + xyChart = { + dataSets = [{ + plotType = "LINE" + targetAxis = "Y1" + timeSeriesQuery = { + timeSeriesFilter = { + filter = "metric.type=\"logging.googleapis.com/user/orca_relay_cloud_sql_wal_checkpoint\" AND resource.type=\"cloudsql_database\"" + aggregation = { + alignmentPeriod = "300s" + perSeriesAligner = "ALIGN_SUM" + crossSeriesReducer = "REDUCE_SUM" + } + } + } + }] + yAxis = { + label = "checkpoints" + scale = "LINEAR" + } + } + } + }, + { + xPos = 3 + yPos = 0 + width = 3 + height = 4 + widget = { + title = "Cloud NAT dropped packets" + xyChart = { + dataSets = [{ + plotType = "LINE" + targetAxis = "Y1" + timeSeriesQuery = { + timeSeriesFilter = { + filter = "metric.type=\"router.googleapis.com/nat/dropped_sent_packets_count\" AND resource.type=\"nat_gateway\" AND resource.label.\"gateway_name\"=monitoring.regex.full_match(\"${local.relay_gce_name}(-.*)?\")" + aggregation = { + alignmentPeriod = "60s" + perSeriesAligner = "ALIGN_SUM" + crossSeriesReducer = "REDUCE_SUM" + groupByFields = ["resource.label.\"gateway_name\"", "metric.label.\"reason\""] + } + } + } + }] + yAxis = { + label = "packets" + scale = "LINEAR" + } + } + } + }, + { + xPos = 6 + yPos = 0 + width = 3 + height = 4 + widget = { + title = "Auth refresh 401s" + xyChart = { + dataSets = [{ + plotType = "LINE" + targetAxis = "Y1" + timeSeriesQuery = { + timeSeriesFilter = { + filter = "metric.type=\"logging.googleapis.com/user/orca_auth_refresh_401\"" + aggregation = { + alignmentPeriod = "300s" + perSeriesAligner = "ALIGN_SUM" + crossSeriesReducer = "REDUCE_SUM" + } + } + } + }] + yAxis = { + label = "rejections" + scale = "LINEAR" + } + } + } + }, + { + xPos = 9 + yPos = 0 + width = 3 + height = 4 + widget = { + title = "Standing desktop controls (fleet sum)" + xyChart = { + dataSets = [{ + plotType = "LINE" + targetAxis = "Y1" + timeSeriesQuery = { + timeSeriesFilter = { + # ALIGN_MEAN, not ALIGN_SUM: each process reports its standing control count once per interval. + filter = "metric.type=\"logging.googleapis.com/user/orca_relay_controls\"" + aggregation = { + alignmentPeriod = "300s" + perSeriesAligner = "ALIGN_MEAN" + crossSeriesReducer = "REDUCE_SUM" + } + } + } + }] + yAxis = { + label = "controls" + scale = "LINEAR" + } + } + } + } + ] + } + }) + + depends_on = [google_logging_metric.relay_incident, google_logging_metric.relay_snapshot] +} diff --git a/cloud/infra/terraform/variables.tf b/cloud/infra/terraform/variables.tf index 68f6c555bd3..91f67e8ebe0 100644 --- a/cloud/infra/terraform/variables.tf +++ b/cloud/infra/terraform/variables.tf @@ -468,6 +468,12 @@ variable "relay_gce_fenced_cells" { default = [] } +variable "relay_cloud_sql_private_ip" { + type = bool + description = "Dial Cloud SQL over its private IP inside this VPC instead of its public IP through Cloud NAT. Requires the foundation root's private services access peering to be applied first; a cell that cannot reach the private IP never becomes ready." + default = false +} + variable "relay_gce_cloud_sql_proxy_image" { type = string description = "Digest-pinned Cloud SQL Auth Proxy image used by private relay workers." diff --git a/cloud/package.json b/cloud/package.json index c75d027d3d2..62dbadc7455 100644 --- a/cloud/package.json +++ b/cloud/package.json @@ -21,7 +21,7 @@ "load:relay:recovery-gate": "node dev/scripts/run-relay-recovery-wave-gate.mjs", "ops:relay": "pnpm --filter @orca-cloud/relay-ops dev", "pretest": "node --test dev/scripts/capture-terraform-plan-baseline.test.mjs dev/scripts/operate-relay-asia-admission.test.mjs dev/scripts/prepare-relay-asia-director-cells.test.mjs dev/scripts/prepare-relay-asia-topology-input.test.mjs dev/scripts/production-cloud-sql-rollout-lock.test.mjs dev/scripts/read-relay-serving-regional-placement-version.test.mjs dev/scripts/relay-asia-admission-workflow.test.mjs dev/scripts/relay-asia-rollout-evidence.test.mjs dev/scripts/relay-asia-topology-workflow.test.mjs dev/scripts/relay-cloud-sql-connection-budget.test.mjs dev/scripts/relay-load-reader-evidence.test.mjs dev/scripts/relay-staging-deploy-identity.test.mjs dev/scripts/sanitize-relay-asia-admission-result.test.mjs dev/scripts/terraform-root-partition.test.mjs dev/scripts/validate-relay-asia-topology-plan.test.mjs ../.github/actions/cloud-sql-rollout-lease/action-contract.test.mjs ../.github/actions/cloud-sql-rollout-lease/storage-lease.test.mjs", - "test": "pnpm -r test && node --test dev/scripts/classify-relay-production-capacity-director.test.mjs dev/scripts/classify-relay-staging-bootstrap.test.mjs dev/scripts/deploy-relay-blue-green.test.mjs dev/scripts/deploy-relay-gce-candidate.test.mjs dev/scripts/deploy-relay-gce-multi-target.test.mjs dev/scripts/github-smoke-token.test.mjs dev/scripts/infra.test.mjs dev/scripts/operate-relay-regional-rehome.test.mjs dev/scripts/power-staging-relay.test.mjs dev/scripts/prepare-relay-capacity-canary.test.mjs dev/scripts/prepare-relay-production-capacity-canary.test.mjs dev/scripts/probe-relay-legacy-admission.test.mjs dev/scripts/probe-relay-rehome-trust.test.mjs dev/scripts/production-cell-image-digest-consistency.test.mjs dev/scripts/read-relay-production-capacity-identity.test.mjs dev/scripts/relay-admission-selector.test.mjs dev/scripts/relay-gce-terraform-fence.test.mjs dev/scripts/relay-load-connection-failure.test.mjs dev/scripts/relay-load-control-peer.test.mjs dev/scripts/relay-load-director-capacity-gate.test.mjs dev/scripts/relay-load-model.test.mjs dev/scripts/relay-load-phase-barrier.test.mjs dev/scripts/relay-load-placement-boundary.test.mjs dev/scripts/relay-load-profile.test.mjs dev/scripts/relay-load-rebind-boundary.test.mjs dev/scripts/relay-load-region-behavior.test.mjs dev/scripts/relay-load-request-unit-boundary.test.mjs dev/scripts/relay-load-run-lifecycle.test.mjs dev/scripts/relay-monitor-evidence.test.mjs dev/scripts/relay-production-capacity-wave.test.mjs dev/scripts/relay-production-capacity-workflow.test.mjs dev/scripts/relay-production-identity-boundaries.test.mjs dev/scripts/relay-production-same-cap-wave.test.mjs dev/scripts/relay-public-workflow-contract.test.mjs dev/scripts/relay-recovery-wave-gate.test.mjs dev/scripts/relay-region-observation-evidence.test.mjs dev/scripts/relay-regional-rehome-workflow.test.mjs dev/scripts/relay-rehome-aggregate-evidence.test.mjs dev/scripts/relay-repository.test.mjs dev/scripts/relay-staging-c4-refresh-workflow.test.mjs dev/scripts/relay-staging-capacity-identity.test.mjs dev/scripts/staging-relay-apply-guard.test.mjs dev/scripts/validate-relay-capacity-plan.test.mjs dev/scripts/verify-relay-capacity-transition.test.mjs dev/scripts/verify-relay-legacy-bootstrap.test.mjs dev/scripts/workload-identity-attribute-conditions.test.mjs", + "test": "pnpm -r test && node --test dev/scripts/classify-relay-production-capacity-director.test.mjs dev/scripts/classify-relay-staging-bootstrap.test.mjs dev/scripts/deploy-relay-blue-green.test.mjs dev/scripts/deploy-relay-gce-candidate.test.mjs dev/scripts/deploy-relay-gce-multi-target.test.mjs dev/scripts/github-smoke-token.test.mjs dev/scripts/infra.test.mjs dev/scripts/operate-relay-regional-rehome.test.mjs dev/scripts/power-staging-relay.test.mjs dev/scripts/prepare-relay-capacity-canary.test.mjs dev/scripts/prepare-relay-production-capacity-canary.test.mjs dev/scripts/probe-relay-legacy-admission.test.mjs dev/scripts/probe-relay-rehome-trust.test.mjs dev/scripts/production-cell-image-digest-consistency.test.mjs dev/scripts/read-relay-production-capacity-identity.test.mjs dev/scripts/relay-admin-endpoint-retry-workflow.test.mjs dev/scripts/relay-admin-transient-retry.test.mjs dev/scripts/relay-admission-selector.test.mjs dev/scripts/relay-gce-terraform-fence.test.mjs dev/scripts/relay-load-connection-failure.test.mjs dev/scripts/relay-load-control-peer.test.mjs dev/scripts/relay-load-director-capacity-gate.test.mjs dev/scripts/relay-load-model.test.mjs dev/scripts/relay-load-phase-barrier.test.mjs dev/scripts/relay-load-placement-boundary.test.mjs dev/scripts/relay-load-profile.test.mjs dev/scripts/relay-load-rebind-boundary.test.mjs dev/scripts/relay-load-region-behavior.test.mjs dev/scripts/relay-load-request-unit-boundary.test.mjs dev/scripts/relay-load-run-lifecycle.test.mjs dev/scripts/relay-monitor-evidence.test.mjs dev/scripts/relay-production-capacity-wave.test.mjs dev/scripts/relay-production-capacity-workflow.test.mjs dev/scripts/relay-production-identity-boundaries.test.mjs dev/scripts/relay-production-same-cap-wave.test.mjs dev/scripts/relay-public-workflow-contract.test.mjs dev/scripts/relay-recovery-wave-gate.test.mjs dev/scripts/relay-region-observation-evidence.test.mjs dev/scripts/relay-regional-rehome-workflow.test.mjs dev/scripts/relay-rehome-aggregate-evidence.test.mjs dev/scripts/relay-repository.test.mjs dev/scripts/relay-same-cap-script-census.test.mjs dev/scripts/relay-staging-c4-refresh-workflow.test.mjs dev/scripts/relay-staging-capacity-identity.test.mjs dev/scripts/staging-relay-apply-guard.test.mjs dev/scripts/validate-relay-capacity-plan.test.mjs dev/scripts/verify-relay-capacity-transition.test.mjs dev/scripts/verify-relay-legacy-bootstrap.test.mjs dev/scripts/workload-identity-attribute-conditions.test.mjs", "typecheck": "pnpm -r typecheck" }, "devDependencies": { diff --git a/cloud/packages/relay-contract/src/host-close-reason.ts b/cloud/packages/relay-contract/src/host-close-reason.ts new file mode 100644 index 00000000000..3a5abde3f00 --- /dev/null +++ b/cloud/packages/relay-contract/src/host-close-reason.ts @@ -0,0 +1,18 @@ +// Mirror of src/shared/relay-host-close-reason.ts in the Orca app repo half. +// A host control socket may close with one of these as its WebSocket close +// reason; the cell records it so a later phone rejection can name the cause. +// Anything else (including the empty reason of an abrupt 1006) means "unknown", +// which is what every peer that predates this file sends. +export const RELAY_HOST_CLOSE_REASON = { + SIGNED_OUT: 'signed-out' +} as const + +export type RelayHostCloseReason = + (typeof RELAY_HOST_CLOSE_REASON)[keyof typeof RELAY_HOST_CLOSE_REASON] + +const REASONS: readonly string[] = Object.values(RELAY_HOST_CLOSE_REASON) + +export function relayHostCloseReasonFrom(value: unknown): RelayHostCloseReason | null { + const text = typeof value === 'string' ? value : (value?.toString() ?? '') + return REASONS.includes(text) ? (text as RelayHostCloseReason) : null +} diff --git a/cloud/packages/relay-contract/src/index.ts b/cloud/packages/relay-contract/src/index.ts index 2a7d7d0feda..aab3b53b5f3 100644 --- a/cloud/packages/relay-contract/src/index.ts +++ b/cloud/packages/relay-contract/src/index.ts @@ -5,6 +5,7 @@ export * from './control-messages.js' export * from './control-continuity.js' export * from './credential-messages.js' export * from './director-messages.js' +export * from './host-close-reason.js' export * from './host-proof-transcript.js' export * from './persistence-invariants.js' export * from './protocol-limits.js' diff --git a/config/docker/cli-launch-contract/Dockerfile b/config/docker/cli-launch-contract/Dockerfile index f6a618a8ece..c90cbcd979c 100644 --- a/config/docker/cli-launch-contract/Dockerfile +++ b/config/docker/cli-launch-contract/Dockerfile @@ -6,8 +6,13 @@ ARG LIBASOUND_PACKAGE=libasound2t64 ENV DEBIAN_FRONTEND=noninteractive # Install Electron's link-time libraries without adding a display server or FUSE. -RUN apt-get update \ - && apt-get install -y --no-install-recommends \ +# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts. +RUN for attempt in 1 2 3 4 5; do \ + apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \ + if [ "$attempt" = 5 ]; then exit 100; fi; \ + rm -rf /var/lib/apt/lists/*; sleep 20; \ + done \ + && apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \ bash \ ca-certificates \ coreutils \ diff --git a/config/docker/headless-pairing/Dockerfile b/config/docker/headless-pairing/Dockerfile index 03664f68b0d..e4b4cafeefc 100644 --- a/config/docker/headless-pairing/Dockerfile +++ b/config/docker/headless-pairing/Dockerfile @@ -5,8 +5,13 @@ ARG LIBASOUND_PACKAGE=libasound2t64 ENV DEBIAN_FRONTEND=noninteractive -RUN apt-get update \ - && apt-get install -y --no-install-recommends \ +# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts. +RUN for attempt in 1 2 3 4 5; do \ + apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \ + if [ "$attempt" = 5 ]; then exit 100; fi; \ + rm -rf /var/lib/apt/lists/*; sleep 20; \ + done \ + && apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \ bash \ ca-certificates \ dbus-x11 \ diff --git a/config/docker/headless-serve-shutdown/Dockerfile b/config/docker/headless-serve-shutdown/Dockerfile index 13b1ed2b69f..8ee7b942499 100644 --- a/config/docker/headless-serve-shutdown/Dockerfile +++ b/config/docker/headless-serve-shutdown/Dockerfile @@ -2,8 +2,13 @@ FROM ubuntu@sha256:678c6550cc43645e08669028bc177f50be4e7c5b8cca677067b1914d4afc7 ENV DEBIAN_FRONTEND=noninteractive -RUN apt-get update \ - && apt-get install -y --no-install-recommends \ +# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts. +RUN for attempt in 1 2 3 4 5; do \ + apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \ + if [ "$attempt" = 5 ]; then exit 100; fi; \ + rm -rf /var/lib/apt/lists/*; sleep 20; \ + done \ + && apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \ bash \ ca-certificates \ dbus-x11 \ diff --git a/config/electron-builder.config.cjs b/config/electron-builder.config.cjs index ebf4d275678..7e0009b3a24 100644 --- a/config/electron-builder.config.cjs +++ b/config/electron-builder.config.cjs @@ -19,6 +19,7 @@ const { } = require('./scripts/verify-packaged-node-pty-job-ownership.cjs') const { verifySkillsCliRuntime } = require('./scripts/verify-skills-cli-runtime.cjs') const { verifyStaticAppImagePackage } = require('./scripts/static-appimage-package-contract.cjs') +const { signWindowsUninstallerViaSignPath } = require('./scripts/windows-uninstaller-signing.cjs') // Why: dev-channel builds must carry the *release* identity — same bundle id, // Developer ID signature, and notarization ticket — or Squirrel.Mac refuses to @@ -401,9 +402,17 @@ module.exports = { // name is absent. An unsigned build that still claimed 'SignPath Foundation' // would therefore reject its own channel's next build — and its way back to // stable with it. Dropping it is what makes dev→dev and dev→stable work. - ...(isWinDevChannel - ? { verifyUpdateCodeSignature: false } - : { signtoolOptions: { publisherName: 'SignPath Foundation' } }), + // Why a sign hook on a build that does not sign: it is the only moment + // electron-builder exposes the NSIS uninstaller (built in its own makensis + // pass, embedded, then deleted). The hook signs nothing — it relays the file + // to and from the CI SignPath request, and is inert when the relay env vars + // are unset, so local and dev builds are unaffected. publisherName stays on + // its existing channel split above. + signtoolOptions: { + sign: signWindowsUninstallerViaSignPath, + ...(isWinDevChannel ? {} : { publisherName: 'SignPath Foundation' }) + }, + ...(isWinDevChannel ? { verifyUpdateCodeSignature: false } : {}), extraResources: [ ...commonExtraResources, ...createPackagedRuntimeNodeModuleResources('win32'), diff --git a/config/nsis/orca-installer-hooks.nsh b/config/nsis/orca-installer-hooks.nsh index ca80c99fc6d..d89439073ab 100644 --- a/config/nsis/orca-installer-hooks.nsh +++ b/config/nsis/orca-installer-hooks.nsh @@ -49,22 +49,48 @@ ; --------------------------------------------------------------------------- ; Clean up the relocated terminal daemon on a REAL uninstall. ; -; Why: the daemon host is deliberately copied to a distinct image name -; (orca-terminal-daemon.exe) under %LOCALAPPDATA%\Orca\daemon-host so that app -; UPDATES cannot kill it — that relocation is what keeps terminals alive across -; updates. The same design means a normal uninstall's process sweep and file -; removal both miss it, leaving an orphaned daemon plus its runtime copy behind. +; Why: the daemon host is deliberately copied OUT of the install dir into +; %LOCALAPPDATA%\Orca\daemon-host so that app UPDATES cannot kill it — +; electron-builder's kill sweep selects processes whose image path is under +; $INSTDIR, and that relocation is what keeps terminals alive across updates. +; The same design means a normal uninstall's process sweep and file removal both +; miss it, leaving an orphaned daemon plus its runtime copy behind. ; ; The ${isUpdated} guard is essential: electron-builder runs this uninstaller as ; part of uninstallOldVersion on EVERY update, and killing the daemon there would ; defeat the whole feature. Only clean up on a genuine uninstall. ; -; The image name and the LOCALAPPDATA folder name must stay in sync with -; DAEMON_HOST_EXE_NAME and LOCAL_HOST_ROOT_NAME in -; src/main/daemon/daemon-host-relocation.ts. +; The LOCALAPPDATA folder name must stay in sync with LOCAL_HOST_ROOT_NAME in +; src/main/daemon/daemon-host-relocation.ts. See +; docs/reference/windows-daemon-host-relocation.md. !macro customUnInstall ${ifNot} ${isUpdated} - nsExec::Exec 'taskkill /F /IM orca-terminal-daemon.exe' + Push $0 + Push $1 + Push $2 + ; The host exe is a verbatim copy of the app exe, so the app's own image name + ; reaches it; the second name covers hosts left by builds that renamed the copy. + ; Filtered to the current user like upstream's per-user KILL_PROCESS, so an + ; elevated machine-wide uninstall cannot reach another logged-on user's session. + ; NSIS expands USERNAME itself: routing through cmd.exe only to get %USERNAME% + ; would add two interpreter spawns to the uninstall path for nothing. + ReadEnvStr $1 USERNAME + ${if} $1 == "" + ; Measured: taskkill rejects an empty filter value outright ("The search filter + ; cannot be recognized") and kills nothing, so with no USERNAME to scope by, + ; kill unfiltered rather than not at all. USERNAME is set in every session an + ; uninstaller runs in, so this is a backstop, not the expected path. + StrCpy $2 "" + ${else} + StrCpy $2 '/FI "USERNAME eq $1"' + ${endIf} + nsExec::Exec 'taskkill /F /IM "${APP_EXECUTABLE_FILENAME}" $2' + Pop $0 + nsExec::Exec 'taskkill /F /IM "orca-terminal-daemon.exe" $2' + Pop $0 + Pop $2 + Pop $1 + Pop $0 ; Give the OS a moment to release the image lock before removing the tree. Sleep 500 RMDir /r "$LOCALAPPDATA\Orca\daemon-host" diff --git a/config/oxlint-performance-audit.json b/config/oxlint-performance-audit.json new file mode 100644 index 00000000000..2912c6b8e03 --- /dev/null +++ b/config/oxlint-performance-audit.json @@ -0,0 +1,35 @@ +{ + "$schema": "../node_modules/oxlint/configuration_schema.json", + "plugins": [], + "categories": { + "correctness": "off", + "suspicious": "off", + "pedantic": "off", + "perf": "off", + "style": "off", + "restriction": "off", + "nursery": "off" + }, + "jsPlugins": [ + { + "name": "app-store-performance", + "specifier": "../config/oxlint-plugins/app-store-performance.mjs" + }, + { + "name": "quadratic-buffer-concat", + "specifier": "../config/oxlint-plugins/quadratic-buffer-concat.mjs" + }, + { + "name": "sort-comparator-performance", + "specifier": "../config/oxlint-plugins/sort-comparator-performance.mjs" + } + ], + "rules": { + "app-store-performance/require-selector": "warn", + "app-store-performance/no-identity-selector": "warn", + "app-store-performance/no-fresh-selector-result": "warn", + "quadratic-buffer-concat/no-loop-carried-concat": "warn", + "sort-comparator-performance/no-repeated-collator": "warn" + }, + "ignorePatterns": ["**/node_modules", "**/dist", "**/out", "**/*.test.*", "**/*.spec.*"] +} diff --git a/config/oxlint-plugins/sort-comparator-performance.mjs b/config/oxlint-plugins/sort-comparator-performance.mjs new file mode 100644 index 00000000000..cd3444cf65f --- /dev/null +++ b/config/oxlint-plugins/sort-comparator-performance.mjs @@ -0,0 +1,60 @@ +const FUNCTION_TYPES = new Set([ + 'ArrowFunctionExpression', + 'FunctionExpression', + 'FunctionDeclaration' +]) + +function propertyName(node) { + if (node?.type !== 'MemberExpression') { + return null + } + if (!node.computed && node.property.type === 'Identifier') { + return node.property.name + } + return node.property.type === 'Literal' ? node.property.value : null +} + +function isInlineSortComparator(node) { + for (let parent = node.parent; parent; parent = parent.parent) { + if (!FUNCTION_TYPES.has(parent.type)) { + continue + } + const call = parent.parent + return ( + call?.type === 'CallExpression' && + call.arguments[0] === parent && + ['sort', 'toSorted'].includes(propertyName(call.callee)) + ) + } + return false +} + +function isCollatorConstruction(node) { + return ( + node.callee?.object?.type === 'Identifier' && + node.callee.object.name === 'Intl' && + propertyName(node.callee) === 'Collator' + ) +} + +function createRule(context) { + function inspect(node) { + const optionedComparison = + node.type === 'CallExpression' && + propertyName(node.callee) === 'localeCompare' && + node.arguments.length >= 3 + if ((optionedComparison || isCollatorConstruction(node)) && isInlineSortComparator(node)) { + context.report({ + node, + message: + 'Create one Intl.Collator before sorting and reuse its compare method; resolving collation options inside the comparator repeats setup for every comparison. Preserve the locale, options, and tie-breaker.' + }) + } + } + return { CallExpression: inspect, NewExpression: inspect } +} + +export default { + meta: { name: 'sort-comparator-performance' }, + rules: { 'no-repeated-collator': { create: createRule } } +} diff --git a/config/patches/@vscode__windows-process-tree@0.8.0.patch b/config/patches/@vscode__windows-process-tree@0.8.0.patch index 10780f5288a..fe5e4be44b1 100644 --- a/config/patches/@vscode__windows-process-tree@0.8.0.patch +++ b/config/patches/@vscode__windows-process-tree@0.8.0.patch @@ -27,15 +27,424 @@ index 855bd4b86f0a3c18c7594212c0e42b6e35bc4001..33774e7ae296f0de39dd94156673c9e7 "/guard:cf", "/sdl", diff --git a/src/process.cc b/src/process.cc -index 3eea92077c4d1d433119361d5c432881859131e9..1998f4addd4d7e9aba946ea6f7f7a4a5d13291bc 100644 +index 3eea92077c4d1d433119361d5c432881859131e9..738775f6fcdfb676054386fe34c0380327ed1863 100644 --- a/src/process.cc +++ b/src/process.cc -@@ -37,7 +37,7 @@ uint32_t GetRawProcessList(std::vector& process_info, - process_info.push_back(std::move(pinfo)); - process_count++; - } +@@ -1,108 +1,112 @@ +-/*--------------------------------------------------------------------------------------------- +- * Copyright (c) Microsoft Corporation. All rights reserved. +- * Licensed under the MIT License. See License.txt in the project root for license information. +- *--------------------------------------------------------------------------------------------*/ +- +-#include "process.h" +-#include "process_commandline.h" +- +-#include +-#include +-#include +- +-uint32_t GetRawProcessList(std::vector& process_info, +- DWORD process_data_flags) { +- // Fetch the PID and PPIDs +- PROCESSENTRY32 process_entry = { 0 }; +- DWORD parent_pid = 0; +- uint32_t process_count = 0; +- HANDLE snapshot_handle = CreateToolhelp32Snapshot(TH32CS_SNAPPROCESS, 0); +- process_entry.dwSize = sizeof(PROCESSENTRY32); +- if (Process32First(snapshot_handle, &process_entry)) { +- do { +- if (process_entry.th32ProcessID != 0) { +- ProcessInfo pinfo; +- pinfo.pid = process_entry.th32ProcessID; +- pinfo.ppid = process_entry.th32ParentProcessID; +- +- if (MEMORY & process_data_flags) { +- GetProcessMemoryUsage(pinfo); +- } +- +- if (COMMANDLINE & process_data_flags) { +- GetProcessCommandLine(pinfo); +- } +- +- strcpy(pinfo.name, process_entry.szExeFile); +- process_info.push_back(std::move(pinfo)); +- process_count++; +- } - } while (process_count < 1024 && Process32Next(snapshot_handle, &process_entry)); +- } +- +- CloseHandle(snapshot_handle); +- return process_count; +-} +- +-void GetProcessMemoryUsage(ProcessInfo& process_info) { +- DWORD pid = process_info.pid; +- HANDLE hProcess; +- PROCESS_MEMORY_COUNTERS pmc; +- +- hProcess = OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, false, pid); +- +- if (hProcess == NULL) { +- return; +- } +- +- if (GetProcessMemoryInfo(hProcess, &pmc, sizeof(pmc))) { +- process_info.memory = (DWORD)pmc.WorkingSetSize; +- } +- +- CloseHandle(hProcess); +-} +- +-// Per documentation, it is not recommended to add or subtract values from the FILETIME +-// structure, or to cast it to ULARGE_INTEGER as this can cause alignment faults on 64-bit Windows. +-// Copy the high and low part to a ULARGE_INTEGER and peform arithmetic on that instead. +-// See https://msdn.microsoft.com/en-us/library/windows/desktop/ms724284(v=vs.85).aspx +-ULONGLONG GetTotalTime(const FILETIME* kernelTime, const FILETIME* userTime) { +- ULARGE_INTEGER kt, ut; +- kt.LowPart = (*kernelTime).dwLowDateTime; +- kt.HighPart = (*kernelTime).dwHighDateTime; +- +- ut.LowPart = (*userTime).dwLowDateTime; +- ut.HighPart = (*userTime).dwHighDateTime; +- +- return kt.QuadPart + ut.QuadPart; +-} +- +-void GetCpuUsage(Cpu& cpu_info, bool first_pass) { +- DWORD pid = cpu_info.pid; +- HANDLE hProcess; +- +- hProcess = OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, false, pid); +- +- if (hProcess == NULL) { +- return; +- } +- +- FILETIME creationTime, exitTime, kernelTime, userTime; +- FILETIME sysIdleTime, sysKernelTime, sysUserTime; +- if (GetProcessTimes(hProcess, &creationTime, &exitTime, &kernelTime, &userTime) +- && GetSystemTimes(&sysIdleTime, &sysKernelTime, &sysUserTime)) { +- if (first_pass) { +- cpu_info.initialProcRunTime = GetTotalTime(&kernelTime, &userTime); +- cpu_info.initialSystemTime = GetTotalTime(&sysKernelTime, &sysUserTime); +- } else { +- ULONGLONG endProcTime = GetTotalTime(&kernelTime, &userTime); +- ULONGLONG endSysTime = GetTotalTime(&sysKernelTime, &sysUserTime); +- +- cpu_info.cpu = 100.0 * (endProcTime - cpu_info.initialProcRunTime) / (endSysTime - cpu_info.initialSystemTime); +- } +- } else { +- cpu_info.cpu = std::numeric_limits::quiet_NaN(); +- } +- +- CloseHandle(hProcess); ++/*--------------------------------------------------------------------------------------------- ++ * Copyright (c) Microsoft Corporation. All rights reserved. ++ * Licensed under the MIT License. See License.txt in the project root for license information. ++ *--------------------------------------------------------------------------------------------*/ ++ ++#include "process.h" ++#include "process_commandline.h" ++ ++#include ++#include ++#include ++ ++uint32_t GetRawProcessList(std::vector& process_info, ++ DWORD process_data_flags) { ++ // Fetch the PID and PPIDs ++ PROCESSENTRY32 process_entry = { 0 }; ++ DWORD parent_pid = 0; ++ uint32_t process_count = 0; ++ HANDLE snapshot_handle = CreateToolhelp32Snapshot(TH32CS_SNAPPROCESS, 0); ++ process_entry.dwSize = sizeof(PROCESSENTRY32); ++ if (Process32First(snapshot_handle, &process_entry)) { ++ do { ++ if (process_entry.th32ProcessID != 0) { ++ // Value-initialize: `memory` is otherwise stack garbage when the flag is unset. ++ ProcessInfo pinfo{}; ++ pinfo.pid = process_entry.th32ProcessID; ++ pinfo.ppid = process_entry.th32ParentProcessID; ++ ++ if (MEMORY & process_data_flags) { ++ GetProcessMemoryUsage(pinfo); ++ } ++ ++ if (COMMANDLINE & process_data_flags) { ++ GetProcessCommandLine(pinfo); ++ } ++ ++ strcpy(pinfo.name, process_entry.szExeFile); ++ process_info.push_back(std::move(pinfo)); ++ process_count++; ++ } + } while (Process32Next(snapshot_handle, &process_entry)); - } - - CloseHandle(snapshot_handle); ++ } ++ ++ CloseHandle(snapshot_handle); ++ return process_count; ++} ++ ++void GetProcessMemoryUsage(ProcessInfo& process_info) { ++ DWORD pid = process_info.pid; ++ HANDLE hProcess; ++ PROCESS_MEMORY_COUNTERS pmc; ++ ++ // PROCESS_VM_READ is never used here -- GetProcessMemoryInfo reads counters the ++ // kernel keeps, not the address space -- and acquiring it is what EDR scores. ++ hProcess = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, false, pid); ++ ++ if (hProcess == NULL) { ++ return; ++ } ++ ++ if (GetProcessMemoryInfo(hProcess, &pmc, sizeof(pmc))) { ++ process_info.memory = (DWORD)pmc.WorkingSetSize; ++ } ++ ++ CloseHandle(hProcess); ++} ++ ++// Per documentation, it is not recommended to add or subtract values from the FILETIME ++// structure, or to cast it to ULARGE_INTEGER as this can cause alignment faults on 64-bit Windows. ++// Copy the high and low part to a ULARGE_INTEGER and peform arithmetic on that instead. ++// See https://msdn.microsoft.com/en-us/library/windows/desktop/ms724284(v=vs.85).aspx ++ULONGLONG GetTotalTime(const FILETIME* kernelTime, const FILETIME* userTime) { ++ ULARGE_INTEGER kt, ut; ++ kt.LowPart = (*kernelTime).dwLowDateTime; ++ kt.HighPart = (*kernelTime).dwHighDateTime; ++ ++ ut.LowPart = (*userTime).dwLowDateTime; ++ ut.HighPart = (*userTime).dwHighDateTime; ++ ++ return kt.QuadPart + ut.QuadPart; ++} ++ ++void GetCpuUsage(Cpu& cpu_info, bool first_pass) { ++ DWORD pid = cpu_info.pid; ++ HANDLE hProcess; ++ ++ // GetProcessTimes needs no more than PROCESS_QUERY_LIMITED_INFORMATION. ++ hProcess = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, false, pid); ++ ++ if (hProcess == NULL) { ++ return; ++ } ++ ++ FILETIME creationTime, exitTime, kernelTime, userTime; ++ FILETIME sysIdleTime, sysKernelTime, sysUserTime; ++ if (GetProcessTimes(hProcess, &creationTime, &exitTime, &kernelTime, &userTime) ++ && GetSystemTimes(&sysIdleTime, &sysKernelTime, &sysUserTime)) { ++ if (first_pass) { ++ cpu_info.initialProcRunTime = GetTotalTime(&kernelTime, &userTime); ++ cpu_info.initialSystemTime = GetTotalTime(&sysKernelTime, &sysUserTime); ++ } else { ++ ULONGLONG endProcTime = GetTotalTime(&kernelTime, &userTime); ++ ULONGLONG endSysTime = GetTotalTime(&sysKernelTime, &sysUserTime); ++ ++ cpu_info.cpu = 100.0 * (endProcTime - cpu_info.initialProcRunTime) / (endSysTime - cpu_info.initialSystemTime); ++ } ++ } else { ++ cpu_info.cpu = std::numeric_limits::quiet_NaN(); ++ } ++ ++ CloseHandle(hProcess); + } +\ No newline at end of file +diff --git a/src/process_commandline.cc b/src/process_commandline.cc +index ea822b120e8038a4803e34647042f08f4aaf5ca1..25907c0bf542bed6c72b1b462b19bcf3210c3cfd 100644 +--- a/src/process_commandline.cc ++++ b/src/process_commandline.cc +@@ -1,67 +1,125 @@ +-/*--------------------------------------------------------------------------------------------- +- * Copyright (c) Microsoft Corporation. All rights reserved. +- * Licensed under the MIT License. See License.txt in the project root for license information. +- *--------------------------------------------------------------------------------------------*/ +- +-#include "process.h" +-#include "process_commandline.h" +-#include +-#include +-#include +- +-bool GetProcessCommandLine(ProcessInfo& process_info) { +- HINSTANCE ntdll = GetModuleHandleW(L"ntdll.dll"); +- if (!ntdll) { +- return false; +- } +- +- decltype(NtQueryInformationProcess)* nt_query_information_process = +- reinterpret_cast( +- GetProcAddress(ntdll, "NtQueryInformationProcess")); +- +- if (!nt_query_information_process) { +- return false; +- } +- +- PROCESS_BASIC_INFORMATION pbi{}; +- PEB peb = {NULL}; +- RTL_USER_PROCESS_PARAMETERS process_parameters = {NULL}; +- +- // Get process handle +- DWORD pid = process_info.pid; +- HANDLE hProcess = OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, FALSE, pid); +- if (hProcess == INVALID_HANDLE_VALUE) { +- return false; +- } +- +- // Get Process Environment Block (PEB) +- NTSTATUS status = nt_query_information_process(hProcess, ProcessBasicInformation, &pbi, sizeof(pbi), nullptr); +- if (NT_SUCCESS(status) && pbi.PebBaseAddress) { +- // Read PEB +- if (ReadProcessMemory(hProcess, pbi.PebBaseAddress, &peb, sizeof(peb), nullptr)) { +- // Read the processs parameters +- if (ReadProcessMemory(hProcess, peb.ProcessParameters, &process_parameters, sizeof(RTL_USER_PROCESS_PARAMETERS), nullptr)) { +- if (process_parameters.CommandLine.Length > 0) { +- std::wstring buffer; +- buffer.resize(process_parameters.CommandLine.Length / sizeof(wchar_t)); +- if (ReadProcessMemory(hProcess, process_parameters.CommandLine.Buffer, &buffer[0], process_parameters.CommandLine.Length, nullptr)) { +- int wide_length = static_cast(buffer.length()); +- int charcount = WideCharToMultiByte(CP_UTF8, 0, buffer.data(), wide_length, +- NULL, 0, NULL, NULL); +- if (charcount) { +- process_info.commandLine.resize(static_cast(charcount)); +- WideCharToMultiByte(CP_UTF8, 0, buffer.data(), wide_length, +- &process_info.commandLine[0], charcount, +- NULL, NULL); +- } +- CloseHandle(hProcess); +- return true; +- } +- } +- } +- } +- } +- +- CloseHandle(hProcess); +- return false; +-} ++/*--------------------------------------------------------------------------------------------- ++ * Copyright (c) Microsoft Corporation. All rights reserved. ++ * Licensed under the MIT License. See License.txt in the project root for license information. ++ *--------------------------------------------------------------------------------------------*/ ++ ++#include "process.h" ++#include "process_commandline.h" ++#include ++#include ++#include ++ ++namespace { ++ ++// Windows 8.1 and later hand back a process's command line as a UNICODE_STRING ++// the kernel builds, needing only PROCESS_QUERY_LIMITED_INFORMATION. ++// ++// There is deliberately no PEB fallback. Reading the command line out of the ++// target's address space -- opening it for VM reads and then chaining ++// memory reads across every pid on a timer -- is the credential-dumping ++// primitive this reader exists to not perform, so it is absent from the binary ++// rather than one anomalous NTSTATUS away. Electron's floor is Windows 10, so ++// every OS Orca supports has this class; if a hooked ntdll refuses it anyway, ++// the command line comes back empty, which callers already handle, instead of ++// silently reinstating the primitive on exactly the instrumented machines this ++// reader was written for. ++const ULONG kProcessCommandLineInformation = 60; ++ ++const NTSTATUS kStatusInfoLengthMismatch = static_cast(0xC0000004L); ++const NTSTATUS kStatusBufferTooSmall = static_cast(0xC0000023L); ++ ++// A command line is a UNICODE_STRING, whose Length is a USHORT, so the kernel ++// can never need more than the header plus 64 KiB. Refusing anything larger ++// keeps a bogus size from throwing bad_alloc out of a scan that has already ++// walked most of the table. ++const ULONG kMaxCommandLineBytes = sizeof(UNICODE_STRING) + 0xFFFF + sizeof(wchar_t); ++ ++// winternl.h's PROCESSINFOCLASS does not name class 60 and its enumerator range ++// stops far short of it, so the class travels as a ULONG rather than a cast enum. ++typedef NTSTATUS(NTAPI* NtQueryInformationProcessFn)(HANDLE, ULONG, PVOID, ULONG, PULONG); ++ ++// ntdll ships no import library for this entry point; it has to be resolved. ++NtQueryInformationProcessFn ResolveNtQueryInformationProcess() { ++ HMODULE ntdll = GetModuleHandleW(L"ntdll.dll"); ++ if (!ntdll) { ++ return nullptr; ++ } ++ return reinterpret_cast( ++ GetProcAddress(ntdll, "NtQueryInformationProcess")); ++} ++ ++NtQueryInformationProcessFn NtQueryInformationProcessEntry() { ++ static NtQueryInformationProcessFn entry = ResolveNtQueryInformationProcess(); ++ return entry; ++} ++ ++bool StoreCommandLineUtf8(ProcessInfo& process_info, const wchar_t* data, size_t wide_length) { ++ if (wide_length == 0) { ++ return false; ++ } ++ int length = static_cast(wide_length); ++ int charcount = WideCharToMultiByte(CP_UTF8, 0, data, length, NULL, 0, NULL, NULL); ++ if (!charcount) { ++ return false; ++ } ++ process_info.commandLine.resize(static_cast(charcount)); ++ WideCharToMultiByte(CP_UTF8, 0, data, length, &process_info.commandLine[0], charcount, NULL, ++ NULL); ++ return true; ++} ++ ++} // namespace ++ ++bool GetProcessCommandLine(ProcessInfo& process_info) { ++ NtQueryInformationProcessFn query = NtQueryInformationProcessEntry(); ++ if (!query) { ++ return false; ++ } ++ ++ HANDLE process = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, FALSE, process_info.pid); ++ if (process == NULL) { ++ return false; ++ } ++ ++ ULONG size = 0; ++ NTSTATUS status = query(process, kProcessCommandLineInformation, nullptr, 0, &size); ++ if (NT_SUCCESS(status)) { ++ // Nothing was written, so there is no command line to read. ++ CloseHandle(process); ++ return false; ++ } ++ if (status != kStatusInfoLengthMismatch && status != kStatusBufferTooSmall) { ++ CloseHandle(process); ++ return false; ++ } ++ if (size < sizeof(UNICODE_STRING) || size > kMaxCommandLineBytes) { ++ CloseHandle(process); ++ return false; ++ } ++ ++ std::vector buffer(size); ++ status = query(process, kProcessCommandLineInformation, &buffer[0], size, &size); ++ CloseHandle(process); ++ if (!NT_SUCCESS(status)) { ++ return false; ++ } ++ ++ // Header and characters arrive in one allocation, but treat the header as ++ // untrusted: a hooked ntdll is the case this reader is written for, and an ++ // unchecked Buffer/Length here would be an over-read encoded straight into JS. ++ // Bound against buffer.size(), never `size` -- the second query overwrote it. ++ const UNICODE_STRING* command_line = reinterpret_cast(&buffer[0]); ++ const unsigned char* begin = &buffer[0]; ++ const unsigned char* end = begin + buffer.size(); ++ const unsigned char* chars = reinterpret_cast(command_line->Buffer); ++ if (chars == nullptr || chars < begin + sizeof(UNICODE_STRING) || chars > end || ++ command_line->Length > static_cast(end - chars)) { ++ return false; ++ } ++ ++ // True only when a command line was actually stored, so "empty" and "not ++ // recovered" stay the same answer they were before this reader replaced the ++ // PEB read. `src/process.cc` discards the result either way. ++ return StoreCommandLineUtf8(process_info, command_line->Buffer, ++ command_line->Length / sizeof(wchar_t)); ++} diff --git a/config/patches/node-pty@1.1.0.patch b/config/patches/node-pty@1.1.0.patch index ff474f7d95e..8f5045b932a 100644 --- a/config/patches/node-pty@1.1.0.patch +++ b/config/patches/node-pty@1.1.0.patch @@ -603,7 +603,7 @@ index 7b4b9e1f990fbf95b51528bb56dc9717f5b87532..2ae787c5bd4f3eba470584dc658a01a5 } #endif diff --git a/src/win/conpty.cc b/src/win/conpty.cc -index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a6a4082ce 100644 +index 7b286d3d644c26141df516929703aa6e129df4b2..4aed260dd68e6a171dcfd349e9a7c5c97209248e 100644 --- a/src/win/conpty.cc +++ b/src/win/conpty.cc @@ -18,6 +18,7 @@ @@ -614,7 +614,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a #include #include #include -@@ -44,12 +45,29 @@ struct pty_baton { +@@ -44,12 +45,39 @@ struct pty_baton { HANDLE hOut; HPCON hpc; @@ -630,22 +630,32 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a + // refused to create or assign one (an outer job without breakaway rights), + // in which case callers fall back to their pre-job behaviour. + HANDLE hJob = nullptr; ++ ++ // Orca: teardown needs BOTH the shell's death and an explicit kill() before ++ // the baton can be freed, so each side records that it has run. Whichever ++ // arrives second frees it. Freeing on the shell's death alone -- what this ++ // file did before -- destroyed the only record of `hpc` while ++ // ClosePseudoConsole was still owed, which is why a self-exiting shell ++ // leaked its pseudoconsole and the console host it reaps (#18601 / F24). ++ bool shellExited = false; ++ bool consoleClosed = false; pty_baton(int _id, HANDLE _hIn, HANDLE _hOut, HPCON _hpc) : id(_id), hIn(_hIn), hOut(_hOut), hpc(_hpc) {}; }; static std::vector> ptyHandles; -+// Orca: guards the job accessors below against the exit watcher thread. It does -+// NOT make the whole table safe -- PtyResize/PtyClear/PtyKill read it unlocked, -+// as they always have -- but it closes the window this patch opened, where the -+// watcher can close hShell/hJob and free the baton between a lookup and its use. ++// Orca: guards the job accessors below, and PtyKill, against the exit watcher ++// thread. It does NOT make the whole table safe -- PtyResize and PtyClear still ++// read it unlocked, as they always have -- but it closes the window this patch ++// opened, where the watcher can close hShell/hJob and free the baton between a ++// lookup and its use. +// Handle VALUES are recycled aggressively, so an unguarded read could pass the +// shell-pid check against an unrelated process and terminate the wrong job. +static std::mutex ptyJobMutex; static volatile LONG ptyCounter; static pty_baton* get_pty_baton(int id) { -@@ -102,8 +120,27 @@ void SetupExitCallback(Napi::Env env, Napi::Function cb, pty_baton* baton) { +@@ -102,8 +130,31 @@ void SetupExitCallback(Napi::Env env, Napi::Function cb, pty_baton* baton) { // Get process exit code. GetExitCodeProcess(baton->hShell, (LPDWORD)(&exit_event->exit_code)); // Clean up handles @@ -665,9 +675,13 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a + // Why inside the lock: erasing frees the baton the job accessors hold a + // pointer to. Note remove_pty_baton must not be an assert() argument -- + // NDEBUG would compile the call away and leak every baton. -+ const bool removed = remove_pty_baton(baton->id); -+ assert(removed); -+ (void)removed; ++ baton->shellExited = true; ++ if (baton->consoleClosed) { ++ const bool removed = remove_pty_baton(baton->id); ++ assert(removed); ++ (void)removed; ++ } ++ // Else PtyKill has not run yet and still owns hpc. It frees the baton. + } + // Why the lock ends here: BlockingCall below waits on the JS thread, and the + // JS thread can be waiting on ptyJobMutex inside PtyTerminateJob. Holding @@ -675,7 +689,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a auto status = tsfn.BlockingCall(exit_event, callback); // In main thread switch (status) { -@@ -409,6 +446,15 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { +@@ -409,6 +460,15 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { throw errorWithCode(info, "UpdateProcThreadAttribute failed"); } @@ -691,7 +705,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a PROCESS_INFORMATION piClient{}; fSuccess = !!CreateProcessW( nullptr, -@@ -416,7 +462,10 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { +@@ -416,7 +476,10 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { nullptr, // lpProcessAttributes nullptr, // lpThreadAttributes false, // bInheritHandles VERY IMPORTANT that this is false @@ -703,7 +717,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a envArg, // lpEnvironment mutableCwd.get(), // lpCurrentDirectory &siEx.StartupInfo, // lpStartupInfo -@@ -426,8 +475,47 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { +@@ -426,8 +489,47 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { throw errorWithCode(info, "Cannot create process"); } @@ -753,7 +767,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a if (useConptyDll && fLoadedDll) { PFNRELEASEPSEUDOCONSOLE const pfnReleasePseudoConsole = (PFNRELEASEPSEUDOCONSOLE)GetProcAddress( -@@ -440,6 +528,8 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { +@@ -440,6 +542,8 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { // Update handle handle->hShell = piClient.hProcess; @@ -762,7 +776,91 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a // Close the thread handle to avoid resource leak CloseHandle(piClient.hThread); -@@ -567,6 +657,143 @@ static Napi::Value PtyKill(const Napi::CallbackInfo& info) { +@@ -544,29 +648,215 @@ static Napi::Value PtyKill(const Napi::CallbackInfo& info) { + int id = info[0].As().Int32Value(); + const bool useConptyDll = info[1].As().Value(); + +- const pty_baton* handle = get_pty_baton(id); ++ // Orca: resolve the DLL BEFORE touching any baton state, for the same reason ++ // PtyConnect does it before creating anything. LoadConptyDll throws when ++ // conpty.dll is missing, and a throw after consoleClosed was set would strand ++ // the pseudoconsole permanently: the retry would find the work already ++ // claimed and do nothing. Only the useConptyDll path can throw here; the ++ // other returns kernel32. ++ HANDLE hLibrary = LoadConptyDll(info, useConptyDll); ++ PFNCLOSEPSEUDOCONSOLE pfnClosePseudoConsole = nullptr; ++ if (hLibrary != nullptr) { ++ pfnClosePseudoConsole = (PFNCLOSEPSEUDOCONSOLE)GetProcAddress( ++ (HMODULE)hLibrary, ++ useConptyDll ? "ConptyClosePseudoConsole" : "ClosePseudoConsole"); ++ } + +- if (handle != nullptr) { +- HANDLE hLibrary = LoadConptyDll(info, useConptyDll); +- bool fLoadedDll = hLibrary != nullptr; +- if (fLoadedDll) +- { +- PFNCLOSEPSEUDOCONSOLE const pfnClosePseudoConsole = (PFNCLOSEPSEUDOCONSOLE)GetProcAddress( +- (HMODULE)hLibrary, +- useConptyDll ? "ConptyClosePseudoConsole" : "ClosePseudoConsole"); +- if (pfnClosePseudoConsole) +- { +- pfnClosePseudoConsole(handle->hpc); ++ // Orca: the baton now outlives the shell, so this runs on a self-exited pty ++ // too -- that is the whole point. Take what we need under the lock: the ++ // watcher thread nulls hShell the moment the shell dies, and TerminateProcess ++ // on a handle it just closed is an invalid-handle operation. Duplicating ++ // rather than reordering keeps upstream's close-then-terminate sequence. ++ HPCON hpc = nullptr; ++ HANDLE hShellDup = nullptr; ++ bool owed = false; ++ { ++ std::lock_guard guard(ptyJobMutex); ++ pty_baton* handle = get_pty_baton(id); ++ // Why the consoleClosed check: a second kill() would otherwise close the ++ // same pseudoconsole twice. Upstream relied on the baton being gone. ++ if (handle != nullptr && !handle->consoleClosed) { ++ hpc = handle->hpc; ++ owed = true; ++ handle->consoleClosed = true; ++ // Null hShell means a self-exited pty, where there is nothing to kill. ++ if (useConptyDll && handle->hShell != nullptr) { ++ if (!DuplicateHandle(GetCurrentProcess(), handle->hShell, GetCurrentProcess(), ++ &hShellDup, 0, FALSE, DUPLICATE_SAME_ACCESS)) { ++ // Why terminate here instead of skipping: a failed duplication leaves ++ // hShellDup null, which is indistinguishable from the self-exit case, ++ // and skipping would leave the shell RUNNING after its pane closed -- ++ // a worse outcome than the leak this all exists to fix. hShell is ++ // valid under this lock and TerminateProcess does not block, so the ++ // only cost is that this rare path kills before the console closes. ++ hShellDup = nullptr; ++ TerminateProcess(handle->hShell, 1); ++ } ++ } ++ if (handle->shellExited) { ++ const bool removed = remove_pty_baton(id); ++ assert(removed); ++ (void)removed; + } ++ // Else the shell is still running and the watcher frees the baton. + } +- if (useConptyDll) { +- TerminateProcess(handle->hShell, 1); ++ } ++ ++ // Why outside the lock: ClosePseudoConsole blocks until the conout side has ++ // drained, and the watcher must be able to take the lock while it does. ++ if (owed) { ++ if (pfnClosePseudoConsole) ++ { ++ pfnClosePseudoConsole(hpc); ++ } ++ if (hShellDup != nullptr) { ++ TerminateProcess(hShellDup, 1); ++ CloseHandle(hShellDup); + } + } + return env.Undefined(); } @@ -808,9 +906,11 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a + * Orca: the pids still alive in this pty's tree, straight from the kernel. + * + * Descendant liveness for a tree that is still tracked, including children that -+ * detached from the console. Once the shell exits the baton is gone, so this -+ * returns null rather than an empty list -- null means "no answer", never -+ * "they died". Also returns null when no job was assigned. ++ * detached from the console. Once the shell exits the watcher nulls hJob, which ++ * ownsShell rejects, so this returns null rather than an empty list -- null ++ * means "no answer", never "they died". (The baton itself now outlives the ++ * shell, until kill() runs; hJob is what makes the answer null.) Also returns ++ * null when no job was assigned. + * + * Does not include the ConPTY console host: CreatePseudoConsole spawns it + * before this job exists, so it is not a member and ClosePseudoConsole is what @@ -906,7 +1006,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a /** * Init */ -@@ -577,6 +804,9 @@ Napi::Object init(Napi::Env env, Napi::Object exports) { +@@ -577,6 +867,9 @@ Napi::Object init(Napi::Env env, Napi::Object exports) { exports.Set("resize", Napi::Function::New(env, PtyResize)); exports.Set("clear", Napi::Function::New(env, PtyClear)); exports.Set("kill", Napi::Function::New(env, PtyKill)); @@ -917,7 +1017,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a }; diff --git a/lib/windowsPtyAgent.js b/lib/windowsPtyAgent.js -index a358ffb..fb3a96f 100644 +index a358ffb177357e177661033c1b092f9c9d0e5f5a..26c2a4c58799ce649f5113131e4c52f7ed2d87ad 100644 --- a/lib/windowsPtyAgent.js +++ b/lib/windowsPtyAgent.js @@ -136,6 +136,9 @@ var WindowsPtyAgent = /** @class */ (function () { @@ -930,6 +1030,20 @@ index a358ffb..fb3a96f 100644 this._outSocket.readable = false; this._getConsoleProcessList().then(function (consoleProcessList) { consoleProcessList.forEach(function (pid) { +@@ -154,9 +157,10 @@ var WindowsPtyAgent = /** @class */ (function () { + // Close the input write handle to signal the end of session. + this._inSocket.destroy(); + this._ptyNative.kill(this._pty, this._useConptyDll); +- this._outSocket.on('data', function () { +- _this._conoutSocketWorker.dispose(); +- }); ++ // Orca: dispose unconditionally, as the non-DLL branch above does. ++ // Waiting for another 'data' event leaks the conout worker on every ++ // self-exiting shell, because no more data ever arrives (F24). ++ this._conoutSocketWorker.dispose(); + } + } + else { diff --git a/lib/windowsTerminal.js b/lib/windowsTerminal.js index 3c38f89..e20b3e6 100644 --- a/lib/windowsTerminal.js @@ -1015,7 +1129,7 @@ index 3c38f89..e20b3e6 100644 \ No newline at end of file +//# sourceMappingURL=windowsTerminal.js.map diff --git a/src/windowsPtyAgent.ts b/src/windowsPtyAgent.ts -index d705444..ce611b8 100644 +index d7054449516f0c9a62af351c2caa17331206d530..0c28a32e2e1db2b3f208ddde8443cd4e67bb1ad6 100644 --- a/src/windowsPtyAgent.ts +++ b/src/windowsPtyAgent.ts @@ -143,6 +143,9 @@ export class WindowsPtyAgent { @@ -1028,6 +1142,20 @@ index d705444..ce611b8 100644 this._outSocket.readable = false; this._getConsoleProcessList().then(consoleProcessList => { consoleProcessList.forEach((pid: number) => { +@@ -159,9 +162,10 @@ export class WindowsPtyAgent { + // Close the input write handle to signal the end of session. + this._inSocket.destroy(); + (this._ptyNative as IConptyNative).kill(this._pty, this._useConptyDll); +- this._outSocket.on('data', () => { +- this._conoutSocketWorker.dispose(); +- }); ++ // Orca: dispose unconditionally, as the non-DLL branch above does. ++ // Waiting for another 'data' event leaks the conout worker on every ++ // self-exiting shell, because no more data ever arrives (F24). ++ this._conoutSocketWorker.dispose(); + } + } else { + // Because pty.kill closes the handle, it will kill most processes by itself. diff --git a/src/windowsTerminal.ts b/src/windowsTerminal.ts index 13f6c6d..eda63c8 100644 --- a/src/windowsTerminal.ts diff --git a/config/performance-audit.md b/config/performance-audit.md new file mode 100644 index 00000000000..f105bfbe787 --- /dev/null +++ b/config/performance-audit.md @@ -0,0 +1,38 @@ +# Performance regression checks + +`pnpm --silent audit:perf > performance-audit.json` scans production `src/` with +the existing app-store and buffer-concatenation rules plus the sort-comparator +rule. Warnings are advisory in this full inventory; tool/parser failures fail. +New warning findings on changed lines fail `pnpm check:code-quality:changed`. +Tests, generated files, `mobile/` and `cloud/` are outside this source audit. + +The sort rule detects optioned `localeCompare` and `Intl.Collator` construction +inside inline `sort`/`toSorted` callbacks. Construct one collator outside the +callback, preserving locale, options and tie-breakers. If the locale changes at +runtime, reconstruct at the next sort or key the cache by locale. Bare comparisons +and standalone equality checks are allowed. There is no autofix or interprocedural +analysis: named comparators, aliases, custom methods and deferred callbacks need +manual review. A warning identifies repeated setup, not proof of visible lag. + +`pnpm test:perf:contracts` runs the explicit selection in +`vitest.performance.config.ts`: SQLite statement reuse and schema parity, relay +filesystem concurrency, tokenizer rejection, highlighting cache, queued +cancellation, terminal backing-memory retention and detector fixtures. Missing +listed files fail configuration loading. Tests run serially, without retries, +and inherit the full suite's setup and forced-GC support. This makes existing +regression coverage easy to run and attribute; it does not create new workload +coverage by itself. + +`.github/workflows/performance-contracts.yml` runs daily and manually on Linux, +macOS and Windows, and on PRs changing this tooling or any listed contract file. +It uploads per-OS JSON test results, plus the source inventory once from Linux +because that scan is OS-independent. Its schedule starts after merge. Run the existing +`test:e2e:terminal-perf:scale:report` for rendered typing/frame budgets and +`test:e2e:ssh-docker-perf` for real transport behavior. Relay unit tests do not +measure SSH RTT, WSL scheduling or a packaged Electron renderer. + +To extend coverage, select a production-path regression with an operation-count, +identity, queue-admission or retained-memory oracle. Confirm it fails with the +old behavior. Use controlled, counterbalanced benchmark samples for timings; +avoid new machine-dependent millisecond gates in the normal unit suite. A green +source scan and these contracts cannot establish that the whole app is fast. diff --git a/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs b/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs new file mode 100644 index 00000000000..1e908754dc6 --- /dev/null +++ b/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs @@ -0,0 +1,205 @@ +const { createHash } = require('node:crypto') +const { readFileSync, renameSync, rmSync, writeFileSync } = require('node:fs') +const { join, resolve } = require('node:path') + +/** + * Release the ConPTY teardown handles a relay's npm-installed node-pty never releases. + * + * Two files, and the ORDER of one of the edits is the whole fix. + * + * `windowsPtyAgent.js` -- `kill()` flips `readable` on both sockets and destroys neither. + * `_cleanUpProcess` destroys `_outSocket`, so the conout handle comes back; nothing ever destroys + * `_inSocket`, and it wraps a real Windows named-pipe handle from `fs.openSync(term.conin, 'w')`. + * Every terminal leaks one File handle for the life of the host process. + * + * The obvious fix -- and the placement `config/patches/node-pty@1.1.0.patch` uses -- releases it at + * the TOP of the branch, before `_getConsoleProcessList()` forks and before the native kill. That is + * measurably worse than leaving the leak alone: teardown aborts partway, the forked console-list + * agent is never reaped, and both pipe handles stay alive instead of one. This asset releases it at + * the END of the branch instead, after the fork and the kill have already happened. + * + * Measured on a Windows SSH host, 20 spawn/kill cycles, handles bucketed by NT object type + * (identical numbers standalone and through a real relay). Every row is the NON-DLL branch, which + * is the branch a relay runs -- see the divergence note below for why that matters: + * + * published node-pty File +1/terminal, Process flat + * desktop patch placement File +2/terminal, Process +1/terminal <-- 3x WORSE + * released last (here) File flat, Process flat + * + * `windowsTerminal.js` carries the desktop's error-listener hunks verbatim. The conin listener is + * what keeps a pipe error retiring one terminal instead of the host -- its own comment names the + * failure mode: "Without a listener, Node promotes errors such as write EAGAIN to uncaughtException". + * It is not what fixes the leak (adding it changed nothing on its own), but it is the guard that + * makes destroying conin safe at all. + * + * Why this ships as a relay asset rather than only in config/patches/node-pty@1.1.0.patch: pnpm + * patches do not cross the SSH boundary -- a relay host runs the tree `npm install` put there. + * + * DELIBERATE DIVERGENCE FROM THE DESKTOP, AND WHY IT IS NOT A DESKTOP-TERMINAL BUG: the two hosts + * do not run the same branch of `kill()`. node-pty defaults `_useConptyDll` to false + * (`windowsPtyAgent.js`). Every desktop site that opens a terminal pane sets it true -- + * `local-pty-utils.ts` (two) and `native-pty-spawn.ts` -- as does the `windows-conpty-warmup.ts` + * warm-up, so all of those take the `else` branch, where UPSTREAM ALREADY destroys the input + * socket. The relay passes no such option (`src/relay/pty-handler.ts`), so it takes the + * `!useConptyDll` branch -- the one this asset and the desktop patch both edit. + * + * THE DESKTOP IS NOT ENTIRELY OFF THAT BRANCH. Two desktop sites omit the option and so run it + * too: the hidden rate-limit probes in `src/main/rate-limits/claude-pty.ts` and + * `codex-pty-rate-limit-probe.ts`. Both recur -- their fetchers poll -- and both tear down through + * `kill()`, so this hunk is live on the desktop, just never for a pane a user can see. Do not + * restate this as "the desktop never executes that branch": that sentence stood here for two + * revisions and is false. + * + * What the numbers above therefore do NOT cover: they were measured on relay-style spawn/kill + * cycles. Whether the early placement costs the same +2 File / +1 Process across a probe's + * lifecycle is UNMEASURED -- plausible, not established, and worth measuring before anyone quotes + * a desktop figure. What IS settled is the claim this comment replaced: that the desktop patch made + * every Windows user worse off ON EVERY TERMINAL. Terminals take the DLL branch, and the harness + * that produced that claim defaulted into the branch it was not trying to measure. + * + * The divergence is therefore about which branch each host runs for the workload that matters, not + * about a regression in the terminals users open. The test still pins it, because a future "sync + * the patches" would put the early placement onto the relay's branch, where it does cost +2 File + * and +1 Process per terminal. + * + * If you extend this enumeration, grep for `node-pty` rather than for a static import: those two + * probes were missed three times because they use `await import('node-pty')`. + * + * THE SELF-EXIT LEAK: FIXED FOR THE DESKTOP BY #18635, STILL LIVE ON A RELAY. A terminal that exits + * on its own is also torn down through `kill()` -- both hosts call `destroy()` on natural exit and + * `WindowsTerminal.destroy()` is `kill()` -- but the shell is already gone by then, so the ordering + * this asset relies on does not hold. Measured over 20 self-exit cycles on the NON-DLL branch: + * published +3 File/+1 Process per terminal, desktop patch placement +2/+1, this tree +2/+1. This + * asset does not close it. + * + * #18635 does, in `config/patches/node-pty@1.1.0.patch`: the baton outlives the shell so `PtyKill` + * still reaches `ClosePseudoConsole`, plus an unconditional conout dispose on the DLL branch. That + * fix does not reach a Windows relay, and no hunk in THIS file can carry it, because it is mostly + * NATIVE (`src/win/conpty.cc`) and this asset only rewrites `lib/*.js`. Three delivery paths exist + * and none currently covers Windows: + * + * - the pnpm patch does not cross the SSH boundary -- the remote `npm install` yields upstream's + * unpatched node-pty; + * - the orcad prebuild matrix has no win32 entry (`MATRIX_SLOTS`, + * `config/scripts/build-orcad-prebuilds.mjs`), so no Windows binary is ever compiled from + * patched source to ship; + * - a relay asset CAN patch native source and rebuild on the host -- that is exactly what + * `node-pty-1.1.0-master-cloexec-patch.cjs` does -- but it returns + * `skipped:unsupported-platform` for anything but linux/darwin. Extending it to win32 means + * requiring an MSVC toolchain on the relay host, a far heavier precondition than on Linux, + * where node-gyp already runs at install time. + * + * So a Windows SSH relay still leaks a pseudoconsole per self-exiting terminal, and closing it is a + * DELIVERY problem, not another hunk here. Do not read #18635's flat self-exit relay numbers as + * covering deployed relays: they were measured against a locally rebuilt binary, so they describe + * the relay CODE PATH on a patched tree, not the tree a relay host actually installs. + */ + +const EXPECTED_NODE_PTY_VERSION = '1.1.0' + +/** Each entry is one published file, its patched form, and the edits between them. */ +const PATCH_TARGETS = [ + { + relativePath: ['lib', 'windowsPtyAgent.js'], + originalSha256: '8636d16b38266112204061a22b135734177c242837982fd3a4055be726efa64a', + patchedSha256: '1e23ef480569e73706e3ab4f5482c7e553c76f51414ae8e7b0bdcc2fd75f7280', + replacements: [ + [ + ' this._ptyNative.kill(this._pty, this._useConptyDll);\n this._conoutSocketWorker.dispose();\n', + ' this._ptyNative.kill(this._pty, this._useConptyDll);\n this._conoutSocketWorker.dispose();\n // Orca: released AFTER the console-list fork and the native kill, not before them.\n // Destroying conin first aborts teardown partway -- measured on a Windows SSH relay\n // as +2 File and +1 Process handles per terminal, against +1 File unpatched.\n this._inSocket.destroy();\n' + ] + ] + }, + { + relativePath: ['lib', 'windowsTerminal.js'], + originalSha256: 'c3a65716f53fed0135a8a633373d5f9c2ab092544d651f27ef0a67096dd3bcd9', + patchedSha256: '8247ecd69be8b18257050fb026b290024612c5ffc6d492ff1d46f81e613be2cf', + replacements: [ + [ + ' _this._agent = new windowsPtyAgent_1.WindowsPtyAgent(file, args, parsedEnv, cwd, _this._cols, _this._rows, false, opt.useConpty, opt.useConptyDll, opt.conptyInheritCursor);\n _this._socket = _this._agent.outSocket;\n // Not available until `ready` event emitted.\n _this._pid = _this._agent.innerPid;', + " _this._agent = new windowsPtyAgent_1.WindowsPtyAgent(file, args, parsedEnv, cwd, _this._cols, _this._rows, false, opt.useConpty, opt.useConptyDll, opt.conptyInheritCursor);\n _this._socket = _this._agent.outSocket;\n // Attach before readiness so a broken ConPTY output pipe cannot be unhandled.\n _this._socket.on('error', function (err) {\n var code = err && err.code;\n // PTY output can report EPIPE before `_close()` wins the race.\n _this._close();\n if (code === 'EPIPE' || code === 'ERR_STREAM_PUSH_AFTER_EOF' || code === 'ERR_STREAM_DESTROYED') {\n return;\n }\n // EIO, happens when someone closes our child process: the only process\n // in the terminal.\n // node < 0.6.14: errno 5\n // node >= 0.6.14: read EIO\n if (typeof code === 'string') {\n if (~code.indexOf('errno 5') || ~code.indexOf('EIO'))\n return;\n }\n // Throw anything else.\n if (_this.listeners('error').length < 2) {\n throw err;\n }\n });\n // Not available until `ready` event emitted.\n _this._pid = _this._agent.innerPid;" + ], + [ + " }\n });\n // Shutdown if `error` event is emitted.\n _this._socket.on('error', function (err) {\n // Close terminal session.\n _this._close();\n // EIO, happens when someone closes our child process: the only process\n // in the terminal.\n // node < 0.6.14: errno 5\n // node >= 0.6.14: read EIO\n if (err.code) {\n if (~err.code.indexOf('errno 5') || ~err.code.indexOf('EIO'))\n return;\n }\n // Throw anything else.\n if (_this.listeners('error').length < 2) {\n throw err;\n }\n });\n // Cleanup after the socket is closed.\n _this._socket.on('close', function () {", + " }\n });\n // Cleanup after the socket is closed.\n _this._socket.on('close', function () {" + ], + [ + ' _this._readable = true;\n _this._writable = true;\n _this._forwardEvents();\n return _this;', + " _this._readable = true;\n _this._writable = true;\n // A ConPTY input-pipe error must retire only this terminal. Without a listener, Node promotes\n // errors such as write EAGAIN to uncaughtException and kills every PTY in the daemon.\n _this._agent.inSocket.on('error', function () {\n if (!_this._writable) {\n return;\n }\n _this._close();\n try {\n _this._agent.kill();\n }\n catch (_a) {\n // The failing pipe may have raced process exit; the terminal is already unwritable.\n }\n });\n _this._forwardEvents();\n return _this;" + ], + [ + 'exports.WindowsTerminal = WindowsTerminal;\n//# sourceMappingURL=windowsTerminal.js.map', + 'exports.WindowsTerminal = WindowsTerminal;\n//# sourceMappingURL=windowsTerminal.js.map\n' + ] + ] + } +] + +function inspectTarget(relayDir, target) { + const nodePtyDir = resolve(relayDir, 'node_modules', 'node-pty') + const packageJson = JSON.parse(readFileSync(join(nodePtyDir, 'package.json'), 'utf8')) + if (packageJson.version !== EXPECTED_NODE_PTY_VERSION) { + throw new Error( + `Refusing to patch node-pty ${packageJson.version}; expected ${EXPECTED_NODE_PTY_VERSION}` + ) + } + const filePath = join(nodePtyDir, ...target.relativePath) + return { filePath, source: readFileSync(filePath, 'utf8') } +} + +function assertPatchedNodePtyWindowsTeardown(relayDir = process.cwd()) { + for (const target of PATCH_TARGETS) { + const inspected = inspectTarget(relayDir, target) + if (sourceSha256(inspected.source) !== target.patchedSha256) { + throw new Error( + `node-pty ConPTY teardown release is not installed in ${target.relativePath.join('/')}` + ) + } + } +} + +function patchNodePtyWindowsTeardown(relayDir = process.cwd()) { + for (const target of PATCH_TARGETS) { + const inspected = inspectTarget(relayDir, target) + const sourceHash = sourceSha256(inspected.source) + if (sourceHash === target.patchedSha256) { + continue + } + if (sourceHash !== target.originalSha256) { + throw new Error( + `Refusing to patch unexpected node-pty source in ${target.relativePath.join('/')}` + ) + } + let patchedSource = inspected.source + for (const [from, to] of target.replacements) { + // Why the count check: an anchor that matched twice would patch the wrong site silently, and + // the hash below would then reject a tree this script had already rewritten. + if (patchedSource.split(from).length - 1 !== 1) { + throw new Error(`Refusing to patch ${target.relativePath.join('/')}; anchor is not unique`) + } + patchedSource = patchedSource.replace(from, to) + } + const temporaryPath = `${inspected.filePath}.orca-patch-${process.pid}` + // Why: a terminated remote install must leave either known source version recoverable on reconnect. + try { + writeFileSync(temporaryPath, patchedSource) + renameSync(temporaryPath, inspected.filePath) + } finally { + rmSync(temporaryPath, { force: true }) + } + } + assertPatchedNodePtyWindowsTeardown(relayDir) +} + +function sourceSha256(source) { + return createHash('sha256').update(source).digest('hex') +} + +if (require.main === module) { + patchNodePtyWindowsTeardown() +} + +module.exports = { + assertPatchedNodePtyWindowsTeardown, + patchNodePtyWindowsTeardown +} diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc index b7905fa0419..d48a239354f 100644 --- a/config/reliability-gates.jsonc +++ b/config/reliability-gates.jsonc @@ -10,6 +10,102 @@ } }, "gates": [ + { + "id": "terminal-output.prestarted-shell-snapshot-adoption", + "title": "Prestarted shell adoption paints covered output once", + "maturity": "experimental", + "protection": "partial", + "owner": "terminal-runtime", + "layer": "renderer-transport-and-live-electron", + "surfaces": [ + "backend-created first terminal", + "daemon snapshot adoption", + "deferred live output" + ], + "platforms": ["macos", "linux", "windows"], + "providers": ["local", "daemon", "wsl", "ssh", "remote-runtime"], + "coveredPlatforms": ["macos", "linux", "windows"], + "coveredProviders": ["local", "daemon", "wsl"], + "coverageNotes": "macOS daemon-backed Electron journey verifies same PID and terminal identity plus rendered output. Focused renderer contracts pass on Linux, Windows and WSL. Neighboring SSH model and replay contracts pass locally; no new live SSH or paired-runtime journey.", + "motivatingLinks": [ + "https://github.com/user-attachments/assets/e8c6d1dc-6150-4c3d-b55a-3d12efefdd04", + "https://github.com/user-attachments/assets/b0328f88-34ac-4d51-8119-9efe17072435" + ], + "invariant": "Adopting a prestarted terminal preserves its existing process and paints snapshot-covered startup output once while retaining subsequent live output. Missing sequence proof or blank snapshots must not authorize dropping output.", + "oracle": "Pass snapshot sequence and proven zero keyboard flags through real IPC transport projection. Deliver snapshot-covered and newer output before reattach resolves; drain replay parse callbacks and require one startup marker and the newer output. Repeat with no sequence and blank snapshot to retain unproven bytes. In Electron select a prestarted workspace, type a generated marker and compare PID and stable terminal identities before and after.", + "commands": [ + "pnpm test src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts", + "pnpm test src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts src/renderer/src/components/terminal-pane/pty-connection-hidden-snapshot-live-overlap.test.ts src/renderer/src/components/terminal-pane/pty-connection-replay-payload-handling.test.ts src/renderer/src/components/terminal-pane/pty-connection/reattach-payload-ssh-reconnect-model-paint.test.ts", + "pnpm test src/renderer/src/components/terminal-pane/pty-connection src/renderer/src/components/terminal-pane/pty-transport" + ], + "testFiles": [ + "src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts", + "src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts" + ], + "assertionRefs": [ + { + "file": "src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts", + "assertions": [ + "zero and nonzero snapshot sequence and proven zero keyboard flags survive IPC projection" + ] + }, + { + "file": "src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts", + "assertions": [ + "startup output covered by the snapshot is painted once", + "new output remains visible", + "legacy unsequenced and blank snapshots retain bytes" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-04", + "runner": "local", + "platform": "macos", + "command": "pnpm test src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts src/renderer/src/components/terminal-pane/pty-connection-hidden-snapshot-live-overlap.test.ts src/renderer/src/components/terminal-pane/pty-connection-replay-payload-handling.test.ts src/renderer/src/components/terminal-pane/pty-connection/reattach-payload-ssh-reconnect-model-paint.test.ts", + "result": "passed", + "durationSeconds": 2.34, + "summary": "5 suites / 48 tests pass. Focused 2-suite runs independently pass 27 tests on Linux, Windows and WSL." + }, + { + "date": "2026-09-04", + "runner": "local", + "platform": "macos", + "command": "pnpm test src/renderer/src/components/terminal-pane/pty-connection src/renderer/src/components/terminal-pane/pty-transport", + "result": "passed", + "durationSeconds": 5.67, + "summary": "Broader connection/transport gate: 77 files and 813 tests passed, including neighboring restore, reconnect, input and replay behavior. Log: artifacts/worktree-create/orca-draft-replay-broader-gate.log." + } + ], + "runtimeBudget": { + "p95Seconds": 15, + "scope": "focused renderer transport and deferred-adoption contracts" + }, + "flakeHistory": { + "status": "unknown", + "evidence": "Focused local and remote runs pass; no CI soak history." + }, + "redGreenEvidence": { + "status": "partial", + "evidence": "Metadata tests fail before forwarding. Corrected parse-draining regression observes two startup markers when the baseline installation is removed, and one after restoration. Initial missing-live-output failure was a harness parse-drain omission and is not red proof. Before/fixed Electron screenshots show duplicate/single startup output." + }, + "performanceBudget": { + "required": true, + "evidence": "Reuses existing snapshot baseline reconciliation with no new scan, timer or subprocess. Corrected daemon-backed rendered trial reaches replay at 116.6 ms and generated keyboard output at 177 ms after selecting the prestarted workspace. This measures selection/adoption, not ordinary composer creation." + }, + "promotionCriteria": [ + "Meet manifest CI and soak policy.", + "Retain intentional-break and rendered identity/output proof.", + "Exercise live SSH and paired-runtime snapshot adoption before claiming full provider coverage." + ], + "knownGaps": [ + "Composer draft creation and cancellation are not implemented by this gate.", + "No new live SSH, Windows or WSL UI run; remote evidence is focused contract tests.", + "Mixed-version snapshots without sequence proof intentionally retain legacy behavior." + ], + "demotionRule": "Keep experimental or demote if adoption duplicates covered output, drops newer or unproven output, changes terminal ownership, or flakes without explanation." + }, { "id": "cmd-j-tabs.host-qualified-candidate-ownership", "title": "Cmd-J tab candidates retain execution-host ownership", @@ -2855,7 +2951,7 @@ "https://github.com/stablyai/orca/pull/13876" ], "invariant": "Opening one HTML preview from a paired client renders the workspace document in exactly one client-local browser tab, located by that document and served over the orca-preview scheme. The client gains exactly that one browser workspace and it is the document one — blank where a URL page carries a URL, named by the document, with the chip naming the file — while the host gains no browser page at all, neither in its own page registry nor in the tab snapshot its clients publish into. The preview occupies its own split without taking focus from the source editor; an explicit click activates it, and closing it removes only the preview. Following a document from a file link is the other half of that switch and does move the reader to it, tab group included, whether the preview is new or already open, because opening a file is a request to look at it. A document tab quit with the client comes back as the same row on a grant the relaunched client mints afresh. A preview is named by the browser page it is open in, not by a namespace of its own, and the page registry has two halves: a workspace-document guest is registered in its own map and is absent from the browsing one entirely. That absence is the fence. Page, session and profile management, agent tab enumeration and command targeting, download routing and certificate attribution all read the browsing map directly, in more places than a per-channel guard could be remembered in, so none of them can name a document page and none of them carries a guard. Browser tools the reader drives (element grab, hover describe, selection capture, the annotation viewport bridge) are the one operation that legitimately spans the halves, and they go through the single authority that reads both, keyed by the page and its hosting renderer. The halves are disjoint in both directions: browsing registration refuses a page the document half already holds, and minting a grant refuses a page the browsing half already holds, so one id can never name a surface in both. The headless backend acts on that refusal by destroying the window it had already opened rather than leaving a policy-less page behind an id nothing can drive, keeping nothing under that id for its own shutdown to hand back. Registration refuses on the same terms when the guest it was asked about is already gone. The exit door is guarded in both its halves: a preview withdraws by revoking its grant and never through the unregister channel, so a page the document half holds arriving there is refused before either the registration teardown or the grab-state disposal beside it, which would otherwise drop the intent an in-flight preview grab compares by identity and leave that grab answering ok without ever arming its guest. A bridge request whose guest does not resolve is refused without tearing down the page it named, so a misaddressed request cannot cancel a healthy page's in-flight downloads and grabs. The annotation viewport bridge resolves its guest when its serialized op actually runs rather than when the request arrived, so a cross-process navigation while it waited cannot leave the bridge installed in a retired guest while the reader looks at a new one. State main keys by a preview's page is disposed when that page's grant is revoked, which is the only signal a preview's surface is gone. A tool asking for a page whose guest has not attached yet waits for that registration and arms when it arrives, rather than answering not-ready at the reader; that wait resolves only the request already naming this page, never the worktree-wide or any-tab waits the CLI and agents use to ask for a browser tab to drive. Handing the previewed document to the reader's own machine routes on the owners its grant was minted against — the file's own connection owner and the worktree's own runtime owner, neither read from the tab's stored fields. Only a document proven to live on this machine reaches the client OS; one with a resolved remote owner is downloaded first; and one whose owner cannot be resolved at all, workspace root included, is refused with a message naming that, because the download route would otherwise read the same absolute path on the client and hand back a same-named local file under the remote document's name. A runtime-owned path that falls outside its worktree root is refused by that route itself and surfaces as a failure toast rather than a download. Nothing the document does writes a file to this machine either: the preview partition denies downloads outright instead of routing them through the browser download flow, which has no page to attribute a preview's bytes to and would otherwise reserve a name in this desktop's Downloads folder and write them there unprompted. That refusal is visible to the reader and invisible to the document: the preview's shell carries a fixed sentence saying downloads are off, published at most once per preview per interval so a document asking in a loop cannot fill Orca's chrome, while the page itself gets back exactly what it got before, which is nothing. The sentence names no file, because the document chooses the name it offers; and a refusal never takes the document away the way an entry document's own failure does, whatever it names. A preview is a browser tab, not an editor tab in a preview mode: it is named the way a browser tab is named — by the document it shows when that document declares a title, and by the file it shows when it does not — while the chip goes on naming the file and the host whatever the document calls itself. A title is refused on the same terms the url is: a document that declares none has Chromium report the grant URL as its title, and that title is stored, mirrored onto the tab and written to disk, so anything carrying the scheme falls back to the file instead. It is created by the preview action as a page located by its document, it carries the workspace-relative path copy the editor's path header owned, and closing it revokes the grant that made the document readable while a URL tab closing beside it revokes nothing. Chrome persisted by builds that made previews editor tabs is dropped on restore rather than coming back naming a surface no restore can produce, and the ordinary editor tab for the same document is left alone. A document tab is held back at the mobile publish boundary — no client holds its grant, and the wire has no tab kind for it — while an ordinary browser tab beside it still publishes. It is held back from the group projection that publishes tab order, recency and group activity as well as from the tab list itself, so no published group names a tab the phone is never sent. A browser page can be located by a workspace document instead of a URL, and the document is the whole of its stored identity. The grant and the orca-preview URL that document is served over are minted when the page mounts and replaced by a hard reload, so neither is ever written to the page's url, mirrored onto its tab, persisted or published: such a page's url is the blank URL from creation through restore, including when a session written elsewhere carries a grant URL in, and what the session carries is the worktree and path a restored page mints afresh against today's owners. Every door onto a page's url holds that line — creation, the title update, and the navigation commit alike — so a report about a document page cannot give it a URL it never had, and the title fallback and the loading affordance follow the url each door actually wrote. The mirror carries the document too, so a tab entry cannot go on naming a document its active page has left. Every guest in the app is policy-attached through one door: a workspace document takes a restricted profile there rather than a separate installer beside it, so the attachment bookkeeping that door owns — what registration refuses, and what teardown frees — covers a preview on the same terms as a browsing page, and a preview takes none of the browsing machinery that door installs. That authority answers from the moment the embedder hands the guest over rather than only after a later navigation: the guest binds to the grant it is already showing, so the tools reach the document the reader opened and not just one they navigated to. A read the host reports as truncated or over-cap is refused rather than served partially, and a document outside the paired worktree is refused with a message naming that boundary instead of a bare read failure. The rendered document reaches nothing off-machine on its own: every served response carries a self-only content security policy, the preview session cancels any request that is not in-document, subframes cannot navigate outside the grant, a guest no document has yet bound to a grant may not navigate at all, the guest gathers no ICE candidates, and an SSH path that canonicalizes outside the grant root is refused before it is read. The one route out is a link the reader presses: a trusted click on an anchor, reported by the preview's own preload from a guest still bound to a live grant, leaves as an Orca browser tab rather than a native window or a dead click — and only after the reader confirms the exact destination URL, so a document cannot spend a single stray press exfiltrating what it can read into a link it authored. The preview hands its guest that focus itself whenever it is the surface the reader is in — a browsing page gets it from the chrome around it, and a preview has no chrome to get it from — and it does so only then, so a preview mounted behind a terminal or an editor never takes the keyboard from what the reader is actually in. It offers again when the window itself takes focus back and nothing in the embedder has claimed that focus, because another app coming to the front lands focus on the embedder rather than the guest and the route out would otherwise stay shut until something remounted the pane — while the same window focus also arrives when the reader presses a tab, that being the guest's own blur returning, and taking focus back from there would fight the reader for their own click. Nothing else does. A navigation or popup the document starts by itself is swallowed whatever else is happening, including immediately after a genuine press elsewhere in the document, so a page that can read its grant cannot hand it to a browser tab; a middle click opens nothing; and a fragment link is answered inside the document. A preview attach carries the preview preload and no renderer-supplied one, and no other attach path can acquire it. A subresource the workspace will not send degrades the document to a notice naming that file, never to a failure panel over a page that rendered. A grant outlives neither the tab that owns it nor the renderer document that minted it, and only the trusted renderer can mint or revoke one. For the browser creations this gate still owns, owner-pinned creation returns the canonical host page identity before navigation readiness; delayed navigation cannot turn a created page into an unidentifiable failure or a duplicate retry. Capability rejection before host mutation must preserve the original error, issue no RPC, surface a failure toast, and remove only a caller-declared newly-created empty split. Post-create reconciliation failure requires exact rollback; ambiguous rollback rejects without local fallback.", - "oracle": "In paired Electron, write an HTML fixture that declares its own title on the host, invoke the Explorer preview action on the client, and require the document text to be readable out of the orca-preview guest before judging any absence. With that presence established, require the client to hold exactly one browser workspace more than its baseline and that workspace to be the document one: page and tab url blank, the document path mirrored onto both, the tab named by the document's title, the chip naming the file, no editor row of the retired preview species anywhere, and the guest URL carrying the orca-preview scheme. Ask the host through its own page registry as well as through the tab snapshot, and in the same run open an ordinary URL browser tab from the same client and require that one to arrive in both — the presence precondition without which “the host gained nothing” is satisfied just as well by an oracle that cannot see browser pages at all. Require the preview to sit in a group other than the source editor's while the active group and tab remain the source editor's. Then click away to the terminal, click the preview tab, require it to reactivate and still render, close it with its own X, and require the document tab to be gone while the host still holds only the URL tab and the source group, source editor and terminal survive. Quit the client with a document tab open and relaunch it on the same profile: require the same workspace row to come back, blank and named by the document, rendering the document again over a grant URL that differs from the one that was quit, with no preview-scheme or document-named page anywhere in what the host holds. Prove the halves are live by flipping one product property at a time and requiring the run to fail: publish document workspaces to the host like ordinary ones, and stop mirroring the document onto the workspace row. Have the fixture document attempt its own egress on every load — an unattended window.open and location.href to an off-machine URL, plus an inline ICE gathering probe — and require the same baseline counts and a candidate count of zero, so the document's own attempts are measured rather than assumed. Then, as a separate phase after the close oracle has already run, bring the client window to the front, press the document's heading with a real mouse event, and require the document to report that the same press drove it to attempt a second window.open and location.href while both browser counts stay at that phase's baseline and nothing routes — the case a recent-input gate cannot distinguish from the press's own effect. Only then press the target=_blank link with a real mouse event and require both a recorded routing call that returned success and a browser count above that baseline, with the preview tab still open. Drive the preload's click policy as a unit oracle over a real document: a dispatched click, a trusted press on an external anchor, an anchor reached through what it wraps, an SVG animated href, a sibling preview link, fragment and percent-encoded fragment targets, a bare hash, and a middle click. Create a browser page located by a workspace document, handing creation a live grant URL, and require its stored url, its mirrored tab url and the written session payload all to be blank with no orca-preview string anywhere in what was written, while an ordinary page created the same way keeps the URL it was given and asks for the address bar the document page never does. Parse the written page and tab through the session schema and require the document to survive both halves. Hydrate them back and require the document page to return blank and still named — including when its page row was salvaged away and only the tab's own copy remains, and when a foreign session carried a grant URL into both rows. Drive the mirror across a page switch out of the document and back, and across a repair in which the document is the only mirrored field that differs. Name a document page from its document and require the tab to take that name, name it with an empty title and require the file, name it with a live grant URL and require the file again with no preview scheme anywhere in the written session, and require an ordinary blank browser tab beside it to still be called New Tab. Dispatch a title update out of a rendered preview's own guest and require it to reach the page state while the identity chip still reads the document's workspace-relative path. Attach a browsing guest and a workspace-document guest through the same method in one run and require the browsing one to take clicked-link routing, popup handling and anti-detection while the document guest takes none of them, stays inside the grant it is showing, denies every window it asks for, and is dropped from the page-keyed document registry by the same teardown that frees its id for a later attach. Register a browsing guest and attach a workspace-document guest in one run against the real manager, require the one door to answer each page with the guest of its own half, and require the document page to be absent from the browsing map and from its enumeration. Drive both browsing registration entry points with a page the document half already holds and require them to register nothing, and drive the mint channel with a page the browsing half already holds and require it to refuse; and drive the offscreen one with a guest that is missing and with one already destroyed, requiring the same refusal. Arm a grab on a live preview target, drive the unregister channel at that same target in the window before the queued operation runs, and require the grab to reach the guest anyway — then drive the same sequence for an ordinary browser page and require its grab state to be disposed after all. Hold one viewport-bridge op open, queue a second behind it, swap the page's guest while that second op waits, and require the injection to land in the guest the page has then. Revoke a grant after a tool has run against its page and require that page's grab state to be cancelled and disposed. Ask a tool for a document page whose guest has not attached, require the request to park in the registration wait, attach the guest, and require the same request to arm on it; require a page nothing ever renders to answer not-ready once that wait elapses. With a document open, ask a tool for a browsing page id and require it to be answered by the browsing half or not at all, with the same channel reaching the document guest under the page it really renders. Drive open-externally for a document whose per-file owner is remote while the workspace-scoped owner is unresolved, for a runtime-owned worktree whose preview tab carries no runtime id of its own, and for a worktree that resolves no runtime owner while the tab still carries one, requiring the download route in each; and for an owner that cannot be resolved at all, and for an unknown workspace root while nothing names another host, requiring a refusal that neither opens nor downloads. Drive the headless backend with a page the document half already holds and require it to reject, destroy the window, and unregister nothing — then shut the backend down and require it still to have unregistered nothing. Navigate a bound preview guest at a second grant through both latch events and require it to stay on the grant it bound to. Mount a preview while a renderer drag is already in flight and require its guest to be click-through at the moment it is appended, not a turn later. Render the editor panel shell in each remaining tab mode and require the path header exactly where the surface does not already name itself. Drive the preview action and require a browser tab located by the document rather than an editor tab, require a second open of the same document to activate the tab it is already in, and require closing that tab to revoke its grant while a URL tab closed beside it revokes none. Hydrate a session carrying preview chrome from a build that made previews editor tabs and require it dropped while the ordinary editor tab for the same document survives. Publish a worktree holding a document tab and a URL tab and require only the URL tab to reach the mobile snapshot. Install the shared partition policies for a preview partition and for an ordinary browsing partition in the same run, fire each one's own will-download listener, and require the preview's to cancel while the browsing one still reaches the download router — then require the preview protocol installer to be what asks for that deny. In the same run, require the cancelled download to raise a reader-facing notice and the routed one to raise none. Drive that notice directly for a guest bound to a live grant, for repeated attempts inside and outside its interval, for two previews at once, and for a contents no preview is bound to; require the guest registry to name the bound grant for a live preview guest and nothing for a contents that is not one, has committed no document, or is gone. Drive the shell with a refusal and require one fixed sentence, still one row after three more refusals, standing beside an asset failure rather than being counted with it, gone behind the failure panel, and ignored when it names another preview's grant. Drive the main-side report gate directly for a sender that is no preview guest, a guest with no bound or a revoked grant, a genuine press Electron's webview focus flag misreports as unfocused, and non-web URLs; drive the reader-facing confirmation for accept, cancel, and a confirmed tab the browser refuses; and drive will-attach-webview in both preload directions. Run the per-owner reader, grant-containment, scheme-admission, guest-policy, and plan-routing contracts as unit oracles, including a host-reported truncation, an over-cap binary, and an out-of-worktree paired path. Drive the reader-facing component with the payloads the reader can actually produce — the entry document fails only as truncated or unreadable, a subresource additionally as a refused format — and require the asset case to leave the guest mounted. Drive the closed-tab cleanup hook, the window installer, and the grant IPC handlers directly, requiring the grant to be released when the preview tab closes, cleared at window creation and on a cross-document main-frame navigation, and refused to any sender that is not the trusted renderer. For the browser creations this gate still owns, run the unchanged contract oracle for direct create and side-preview callers with absent status, unknown capabilities, and a mixed-version host, requiring the original unsupported error or visible toast, zero RPCs, and no retained new split; hold a real navigation response beyond the 15-second client deadline after host creation and require the first RPC to return the exact host inventory page ID, one host page, and no retry; repeat reconciliation faults against headless serve and retain the separate exact-page reconciliation rollback oracle.", + "oracle": "In paired Electron, write an HTML fixture that declares its own title on the host, invoke the Explorer preview action on the client, and require the document text to be readable out of the orca-preview guest before judging any absence. With that presence established, require the client to hold exactly one browser workspace more than its baseline and that workspace to be the document one: page and tab url blank, the document path mirrored onto both, the tab named by the document's title, the chip naming the file, no editor row of the retired preview species anywhere, and the guest URL carrying the orca-preview scheme. Ask the host through its own page registry as well as through the tab snapshot, and in the same run open an ordinary URL browser tab from the same client and require that one to arrive in both — the presence precondition without which “the host gained nothing” is satisfied just as well by an oracle that cannot see browser pages at all. Require the preview to sit in a group other than the source editor's while the active group and tab remain the source editor's. Then click away to the terminal, click the preview tab, require it to reactivate and still render, close it with its own X, and require the document tab to be gone while the host still holds only the URL tab and the source group, source editor and terminal survive. Quit the client with a document tab open and relaunch it on the same profile: require the same workspace row to come back, blank and named by the document, rendering the document again over a grant URL that differs from the one that was quit, with no preview-scheme or document-named page anywhere in what the host holds. Prove the halves are live by flipping one product property at a time and requiring the run to fail: publish document workspaces to the host like ordinary ones, and stop mirroring the document onto the workspace row. Have the fixture document attempt its own egress on every load — an unattended window.open and location.href to an off-machine URL, plus an inline ICE gathering probe — and require the same baseline counts and a candidate count of zero, so the document's own attempts are measured rather than assumed. Then, as a separate phase after the close oracle has already run, bring the client window to the front, press the document's heading with a real mouse event, and require the document to report that the same press drove it to attempt a second window.open and location.href while both browser counts stay at that phase's baseline and nothing routes — the case a recent-input gate cannot distinguish from the press's own effect. Only then press the target=_blank link with a real mouse event and require both a recorded routing call that returned success and a browser count above that baseline, with the preview tab still open. Drive the preload's click policy as a unit oracle over a real document: a dispatched click, a trusted press on an external anchor, an anchor reached through what it wraps, an SVG animated href, a sibling preview link, fragment and percent-encoded fragment targets, a bare hash, and a middle click. Create a browser page located by a workspace document, handing creation a live grant URL, and require its stored url, its mirrored tab url and the written session payload all to be blank with no orca-preview string anywhere in what was written, while an ordinary page created the same way keeps the URL it was given and asks for the address bar the document page never does. Parse the written page and tab through the session schema and require the document to survive both halves. Hydrate them back and require the document page to return blank and still named — including when its page row was salvaged away and only the tab's own copy remains, and when a foreign session carried a grant URL into both rows. Drive the mirror across a page switch out of the document and back, and across a repair in which the document is the only mirrored field that differs. Name a document page from its document and require the tab to take that name, name it with an empty title and require the file, name it with a live grant URL and require the file again with no preview scheme anywhere in the written session, and require an ordinary blank browser tab beside it to still be called New Tab. Dispatch a title update out of a rendered preview's own guest and require it to reach the page state while the identity chip still reads the document's workspace-relative path. Attach a browsing guest and a workspace-document guest through the same method in one run and require the browsing one to take clicked-link routing, popup handling and auth-identity detach tracking while the document guest takes none of them, stays inside the grant it is showing, denies every window it asks for, and is dropped from the page-keyed document registry by the same teardown that frees its id for a later attach. Register a browsing guest and attach a workspace-document guest in one run against the real manager, require the one door to answer each page with the guest of its own half, and require the document page to be absent from the browsing map and from its enumeration. Drive both browsing registration entry points with a page the document half already holds and require them to register nothing, and drive the mint channel with a page the browsing half already holds and require it to refuse; and drive the offscreen one with a guest that is missing and with one already destroyed, requiring the same refusal. Arm a grab on a live preview target, drive the unregister channel at that same target in the window before the queued operation runs, and require the grab to reach the guest anyway — then drive the same sequence for an ordinary browser page and require its grab state to be disposed after all. Hold one viewport-bridge op open, queue a second behind it, swap the page's guest while that second op waits, and require the injection to land in the guest the page has then. Revoke a grant after a tool has run against its page and require that page's grab state to be cancelled and disposed. Ask a tool for a document page whose guest has not attached, require the request to park in the registration wait, attach the guest, and require the same request to arm on it; require a page nothing ever renders to answer not-ready once that wait elapses. With a document open, ask a tool for a browsing page id and require it to be answered by the browsing half or not at all, with the same channel reaching the document guest under the page it really renders. Drive open-externally for a document whose per-file owner is remote while the workspace-scoped owner is unresolved, for a runtime-owned worktree whose preview tab carries no runtime id of its own, and for a worktree that resolves no runtime owner while the tab still carries one, requiring the download route in each; and for an owner that cannot be resolved at all, and for an unknown workspace root while nothing names another host, requiring a refusal that neither opens nor downloads. Drive the headless backend with a page the document half already holds and require it to reject, destroy the window, and unregister nothing — then shut the backend down and require it still to have unregistered nothing. Navigate a bound preview guest at a second grant through both latch events and require it to stay on the grant it bound to. Mount a preview while a renderer drag is already in flight and require its guest to be click-through at the moment it is appended, not a turn later. Render the editor panel shell in each remaining tab mode and require the path header exactly where the surface does not already name itself. Drive the preview action and require a browser tab located by the document rather than an editor tab, require a second open of the same document to activate the tab it is already in, and require closing that tab to revoke its grant while a URL tab closed beside it revokes none. Hydrate a session carrying preview chrome from a build that made previews editor tabs and require it dropped while the ordinary editor tab for the same document survives. Publish a worktree holding a document tab and a URL tab and require only the URL tab to reach the mobile snapshot. Install the shared partition policies for a preview partition and for an ordinary browsing partition in the same run, fire each one's own will-download listener, and require the preview's to cancel while the browsing one still reaches the download router — then require the preview protocol installer to be what asks for that deny. In the same run, require the cancelled download to raise a reader-facing notice and the routed one to raise none. Drive that notice directly for a guest bound to a live grant, for repeated attempts inside and outside its interval, for two previews at once, and for a contents no preview is bound to; require the guest registry to name the bound grant for a live preview guest and nothing for a contents that is not one, has committed no document, or is gone. Drive the shell with a refusal and require one fixed sentence, still one row after three more refusals, standing beside an asset failure rather than being counted with it, gone behind the failure panel, and ignored when it names another preview's grant. Drive the main-side report gate directly for a sender that is no preview guest, a guest with no bound or a revoked grant, a genuine press Electron's webview focus flag misreports as unfocused, and non-web URLs; drive the reader-facing confirmation for accept, cancel, and a confirmed tab the browser refuses; and drive will-attach-webview in both preload directions. Run the per-owner reader, grant-containment, scheme-admission, guest-policy, and plan-routing contracts as unit oracles, including a host-reported truncation, an over-cap binary, and an out-of-worktree paired path. Drive the reader-facing component with the payloads the reader can actually produce — the entry document fails only as truncated or unreadable, a subresource additionally as a refused format — and require the asset case to leave the guest mounted. Drive the closed-tab cleanup hook, the window installer, and the grant IPC handlers directly, requiring the grant to be released when the preview tab closes, cleared at window creation and on a cross-document main-frame navigation, and refused to any sender that is not the trusted renderer. For the browser creations this gate still owns, run the unchanged contract oracle for direct create and side-preview callers with absent status, unknown capabilities, and a mixed-version host, requiring the original unsupported error or visible toast, zero RPCs, and no retained new split; hold a real navigation response beyond the 15-second client deadline after host creation and require the first RPC to return the exact host inventory page ID, one host page, and no retry; repeat reconciliation faults against headless serve and retain the separate exact-page reconciliation rollback oracle.", "commands": [ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-browser.test.ts src/main/runtime/rpc/methods/browser.test.ts src/renderer/src/lib/file-preview.test.ts src/renderer/src/runtime/web-session-browser-placement.test.ts src/renderer/src/runtime/web-runtime-session.test.ts src/renderer/src/runtime/web-session-tabs-sync.test.ts src/renderer/src/runtime/remote-server-parity.test.ts", "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/runtime/web-runtime-browser-materialization.test.ts", @@ -17922,6 +18018,139 @@ "The sentinel changes a pane title within an existing layout; concurrent split and close conflicts remain separate coverage." ], "demotionRule": "Demote if a failed push suppresses an identical retry, a successful equal write resumes redundant churn, or the routed observer journey flakes without a diagnosed cause." + }, + { + "id": "ssh.docker-recovery-and-resource-lifecycle", + "title": "Docker SSH reconnect, host faults, listing and watcher lifecycle", + "maturity": "experimental", + "protection": "partial", + "owner": "terminal-runtime", + "layer": "electron-docker-ssh", + "surfaces": [ + "SSH terminal recovery", + "SSH remote resource ownership", + "remote file listing", + "remote explorer watcher recovery", + "Electron test process cleanup" + ], + "platforms": ["macos", "linux", "windows"], + "providers": ["ssh"], + "coveredPlatforms": ["macos"], + "coveredProviders": ["ssh"], + "coverageNotes": "A macOS Electron client drives a Linux Docker SSH execution host. The six-spec suite passed ten enabled cases with clean worker exit (5.2m). The formerly skipped frozen-host input case now waits for recovered authority before sending input and passed four separate executions (one initial and three repetitions). The existing flooded-shell fixme remains an explicitly reproduced application gap.", + "motivatingLinks": [ + "https://github.com/stablyai/orca/issues/18018", + "https://github.com/stablyai/orca/pull/18546", + "https://github.com/stablyai/orca/issues/12547" + ], + "invariant": "Transport loss and frozen-host silence must preserve the remote session; host relay loss may rebind a pane without accumulating reattachable leases. Reconnects must preserve usable terminal content, bounded PTYs/fds/processes, complete large listings, and independently recoverable watcher processes. Electron test shutdown must release inherited pipes after confirmed root exit without closing live-process pipes.", + "oracle": "Poll a changed connected SSH authority after injected faults, then require terminal output and appropriate PTY identity. Read remote process/fd state, listFiles replies, and rendered explorer rows. Resolve Playwright cleanup only after the root process exits and its inherited pipes close; live-process pipes remain untouched.", + "commands": [ + "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-transport-drop-recovery.spec.ts tests/e2e/ssh-docker-half-open-link.spec.ts tests/e2e/ssh-docker-quick-open-large-listing.spec.ts tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts tests/e2e/ssh-docker-resource-accumulation.spec.ts tests/e2e/ssh-docker-watcher-isolation.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1", + "pnpm exec vitest run --config config/vitest.config.ts tests/e2e/helpers/electron-process-shutdown.unit.test.ts" + ], + "testFiles": [ + "tests/e2e/ssh-docker-transport-drop-recovery.spec.ts", + "tests/e2e/ssh-docker-half-open-link.spec.ts", + "tests/e2e/ssh-docker-quick-open-large-listing.spec.ts", + "tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts", + "tests/e2e/ssh-docker-resource-accumulation.spec.ts", + "tests/e2e/ssh-docker-watcher-isolation.spec.ts", + "tests/e2e/helpers/electron-process-shutdown.unit.test.ts" + ], + "assertionRefs": [ + { + "file": "tests/e2e/ssh-docker-transport-drop-recovery.spec.ts", + "assertions": [ + "preserves transport-drop PTY and scrollback, replaces relay-loss binding, and keeps one reattachable lease per pane" + ] + }, + { + "file": "tests/e2e/ssh-docker-half-open-link.spec.ts", + "assertions": [ + "leaves connected after host freeze and renders process-produced output after recovery" + ] + }, + { + "file": "tests/e2e/ssh-docker-quick-open-large-listing.spec.ts", + "assertions": [ + "returns both a bounded client page and a complete legacy-client remote listing" + ] + }, + { + "file": "tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts", + "assertions": [ + "restores shell scrollback and full-screen output and opens a usable fresh tab" + ] + }, + { + "file": "tests/e2e/ssh-docker-resource-accumulation.spec.ts", + "assertions": [ + "keeps remote pts devices, relay fds, process counts and inherited master fds bounded" + ] + }, + { + "file": "tests/e2e/ssh-docker-watcher-isolation.spec.ts", + "assertions": [ + "keeps rendered explorer changes and terminal output live after watcher crash and repairs a deleted watcher artifact" + ] + }, + { + "file": "tests/e2e/helpers/electron-process-shutdown.unit.test.ts", + "assertions": [ + "releases inherited pipes after confirmed exit, including prior exit", + "retains live-process pipes on shutdown timeout" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-05", + "runner": "local", + "platform": "macos", + "result": "passed", + "command": "pnpm exec vitest run --config config/vitest.config.ts tests/e2e/helpers/electron-process-shutdown.unit.test.ts", + "durationSeconds": 0.168, + "summary": "All three shutdown regression tests passed; disabling pipe release fails the first two by timeout. Two half-open Electron repetitions separately passed in 1.7m without worker teardown timeout." + }, + { + "date": "2026-09-05", + "runner": "local", + "platform": "macos", + "result": "passed", + "command": "ORCA_E2E_SSH_DOCKER=1 SKIP_BUILD=1 pnpm exec playwright test tests/e2e/ssh-docker-transport-drop-recovery.spec.ts tests/e2e/ssh-docker-half-open-link.spec.ts tests/e2e/ssh-docker-quick-open-large-listing.spec.ts tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts tests/e2e/ssh-docker-resource-accumulation.spec.ts tests/e2e/ssh-docker-watcher-isolation.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1", + "durationSeconds": 312, + "summary": "Six specs: ten passed, two existing fixme skipped, clean worker shutdown. Baseline same enabled suite: ten passed but worker teardown timed out (7.3m)." + } + ], + "runtimeBudget": { + "p95Seconds": 420, + "scope": "per Electron Docker test; measured suite p95 and CI soak not yet established" + }, + "flakeHistory": { + "status": "flaky", + "evidence": "Baseline: ten enabled tests passed, two fixme skipped, worker teardown timed out (7.3m). After pipe cleanup: ten passed and worker exited cleanly (5.2m); two half-open repeats passed (1.7m). The formerly skipped thaw-input case failed before its recovered-authority wait and passed 1+3 executions afterward (56.9s + 2.6m). Flood failed both its original input oracle and a strengthened producer-completion oracle after recovery." + }, + "redGreenEvidence": { + "status": "partial", + "evidence": "Disabling exited-process pipe release causes two shutdown contract tests to time out; restoring it passes 3/3. Baseline Docker worker teardown failed; final six-spec enabled run and half-open repeats exit successfully. Frozen-host input fails without the post-thaw recovered-authority wait and passes four runs with it. Full product fault/recovery mutation coverage and CI history remain missing." + }, + "performanceBudget": { + "required": true, + "evidence": "Test-only bounded pipe destruction and authority polling; no production polling, subprocesses, or runtime work added. Remote resources are counted instead of using wall-clock leak thresholds." + }, + "promotionCriteria": [ + "Require complete six-spec repeat runs with clean worker shutdown.", + "Resolve the remaining #18018 flooded-shell reproduction and remove its fixme marker.", + "Collect CI runtime and flake history plus product red/green evidence before blocking." + ], + "knownGaps": [ + "The disconnected 48MB flood still loses its relay channel: original post-flood input marker failed in 60s, and waiting for the finite producer completion marker failed in 120s. It remains an explicit #18018 fixme reproduction; frozen-host input is re-enabled after four successful runs.", + "Linux and Windows desktop clients, WSL, folder workspaces, paired runtimes and live agent CLIs are not exercised by these Docker specs.", + "Some legacy assertions inspect terminal serialization or backing state rather than rendered DOM; no blanket visual coverage claim.", + "No p95 CI history or full product mutation proof." + ], + "demotionRule": "Keep experimental while any recovery reproduction fails or any teardown, identity, resource-count, or rendered oracle flakes; never promote by extending sleeps or retries." } ] } diff --git a/config/scripts/agent-inspection-cadence-batching-benchmark.mjs b/config/scripts/agent-inspection-cadence-batching-benchmark.mjs new file mode 100644 index 00000000000..128627676c4 --- /dev/null +++ b/config/scripts/agent-inspection-cadence-batching-benchmark.mjs @@ -0,0 +1,139 @@ +#!/usr/bin/env node +// Counts how many whole-host process-table captures the agent-completion cadence costs. +// +// Local panes all resolve out of one TTL-deduped snapshot, and the inspection queue collapses +// every shared-observation task enqueued in the same tick onto a single capture. So the capture +// count is the number of DISTINCT wake instants across panes, not the number of pane wakes. +// +// This drives the production interval picker (`nextCadenceInspectionDelayMs`) against a baseline +// that reproduces the pre-change ±10% jitter, over a simulated wall-clock window. +import { spawnSync } from 'node:child_process' +import fs from 'node:fs' +import nodeModule from 'node:module' +import path from 'node:path' +import process from 'node:process' +import { fileURLToPath } from 'node:url' + +if (!process.execArgv.includes('--experimental-transform-types')) { + const result = spawnSync( + process.execPath, + ['--experimental-transform-types', '--no-warnings', import.meta.filename], + { stdio: 'inherit' } + ) + process.exit(result.status ?? 1) +} + +nodeModule.registerHooks({ + resolve(specifier, context, nextResolve) { + if (specifier.startsWith('.') && !/\.[cm]?[jt]s$/.test(specifier) && context.parentURL) { + const candidate = new URL(`${specifier}.ts`, context.parentURL) + if (fs.existsSync(fileURLToPath(candidate))) { + return { url: candidate.href, shortCircuit: true } + } + } + return nextResolve(specifier, context) + } +}) + +const ROOT = path.resolve(import.meta.dirname, '../..') +const WINDOW_MS = Number(process.env.ORCA_INSPECTION_BENCH_WINDOW_MS ?? '60000') +const PANE_COUNTS = (process.env.ORCA_INSPECTION_BENCH_PANES ?? '1,2,4,8') + .split(',') + .map((value) => Number(value.trim())) + +if (!Number.isSafeInteger(WINDOW_MS) || WINDOW_MS <= 0) { + throw new Error(`ORCA_INSPECTION_BENCH_WINDOW_MS must be a positive integer, got ${WINDOW_MS}`) +} +for (const paneCount of PANE_COUNTS) { + if (!Number.isSafeInteger(paneCount) || paneCount <= 0) { + throw new Error(`ORCA_INSPECTION_BENCH_PANES entries must be positive, got ${paneCount}`) + } +} + +const { nextCadenceInspectionDelayMs } = await import( + path.join(ROOT, 'src/renderer/src/components/terminal-pane/agent-completion-poll-interval.ts') +) +const { POLL_TIER_INTERVAL_MS } = await import( + path.join(ROOT, 'src/renderer/src/components/terminal-pane/agent-completion-poll-cadence.ts') +) +const { PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS } = await import( + path.join(ROOT, 'src/shared/process-table-snapshot-reader.ts') +) + +// Pre-change: independent ±10% jitter per pane, re-rolled on every reschedule. +function baselineDelayMs(baseMs) { + return Math.round(baseMs * (1 + (Math.random() * 0.2 - 0.1))) +} + +function simulate(paneCount, baseMs, pickDelay) { + const startedAt = 1_700_000_000_000 + const wakes = [] + for (let pane = 0; pane < paneCount; pane += 1) { + // Panes mount at arbitrary moments, which is what spreads them apart in the first place. + let clock = startedAt + Math.floor(Math.random() * baseMs) + while ((clock += pickDelay(baseMs, clock)) < startedAt + WINDOW_MS) { + wakes.push(clock) + } + } + // A wake is served from the snapshot the previous capture produced until that snapshot's TTL + // lapses, so the TTL window starts at the capture, not on an epoch grid. + let captures = 0 + let snapshotExpiresAt = -Infinity + for (const wakeAt of wakes.sort((left, right) => left - right)) { + if (wakeAt >= snapshotExpiresAt) { + captures += 1 + snapshotExpiresAt = wakeAt + PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS + } + } + return captures +} + +function medianOf(rounds, run) { + const samples = Array.from({ length: rounds }, run).sort((left, right) => left - right) + return samples[Math.floor(samples.length / 2)] +} + +const baseMs = POLL_TIER_INTERVAL_MS.idle +console.log( + `Agent-completion cadence — whole-host \`ps\` captures over ${WINDOW_MS / 1000}s at the idle tier (${baseMs}ms)\n` +) +console.log('| visible panes | before | after | reduction |') +console.log('| --- | --- | --- | --- |') +for (const paneCount of PANE_COUNTS) { + const before = medianOf(21, () => simulate(paneCount, baseMs, baselineDelayMs)) + const after = medianOf(21, () => + simulate(paneCount, baseMs, (base, now) => + nextCadenceInspectionDelayMs({ + baseMs: base, + hasConsecutiveErrors: false, + alignToSharedGrid: true, + now + }) + ) + ) + // A window shorter than one cadence tier can leave the baseline at zero; reporting a + // percentage off that divides by zero and prints a meaningless reduction. + const reduction = before > 0 ? `${(((before - after) / before) * 100).toFixed(0)}%` : 'n/a' + console.log(`| ${paneCount} | ${before} | ${after} | ${reduction} |`) +} + +// Detection latency must not regress: the grid deadline is always within one interval. +let worstDelay = 0 +for (let sample = 0; sample < 100_000; sample += 1) { + const now = 1_700_000_000_000 + sample * 7 + worstDelay = Math.max( + worstDelay, + nextCadenceInspectionDelayMs({ + baseMs, + hasConsecutiveErrors: false, + alignToSharedGrid: true, + now + }) + ) +} +if (worstDelay > baseMs) { + throw new Error(`grid alignment delayed a poll to ${worstDelay}ms, above the ${baseMs}ms tier`) +} +console.log( + `\nWorst observed wait: ${worstDelay}ms (tier interval ${baseMs}ms) — no inspection is ever delayed.` +) diff --git a/config/scripts/benchmark-browser-tunnel-framing.mjs b/config/scripts/benchmark-browser-tunnel-framing.mjs new file mode 100644 index 00000000000..e91fd0887f6 --- /dev/null +++ b/config/scripts/benchmark-browser-tunnel-framing.mjs @@ -0,0 +1,110 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { readFileSync } from 'node:fs' +import { stripTypeScriptTypes } from 'node:module' +import { performance } from 'node:perf_hooks' + +// Run from the worktree root: node config/scripts/benchmark-browser-tunnel-framing.mjs [base-ref] +const path = 'src/shared/browser-network-tunnel-stream-framing.ts' +const baselineRef = process.argv[2] ?? 'HEAD' +const beforeSource = execFileSync('git', ['show', `${baselineRef}:${path}`], { + encoding: 'utf8' +}) +const afterSource = readFileSync(path, 'utf8') +const load = (source) => + import( + `data:text/javascript;base64,${Buffer.from( + stripTypeScriptTypes(source, { mode: 'transform' }) + ).toString('base64')}` + ) +const before = await load(beforeSource) +const after = await load(afterSource) + +function measure(module, chunks, payload, repetitions) { + let frameCount = 0 + let lastFrame + const onFrame = (frame) => { + frameCount++ + lastFrame = frame + } + const onError = (error) => { + throw error + } + const run = () => { + const decoder = new module.BrowserNetworkTunnelStreamFrameDecoder(onFrame, onError) + for (const chunk of chunks) { + decoder.feed(chunk) + } + } + run() + assert.deepEqual(lastFrame, payload) + const samples = [] + for (let sample = 0; sample < 5; sample++) { + const start = performance.now() + for (let iteration = 0; iteration < repetitions; iteration++) { + run() + } + samples.push((performance.now() - start) / repetitions) + } + assert.equal(frameCount, 1 + 5 * repetitions) + return samples.sort((a, b) => a - b)[2] +} + +function countCopies(module, chunks) { + const originalSet = Uint8Array.prototype.set + const originalSlice = Uint8Array.prototype.slice + let copied = 0 + Uint8Array.prototype.set = function (source, offset) { + copied += source.length + return originalSet.call(this, source, offset) + } + Uint8Array.prototype.slice = function (...args) { + const result = originalSlice.apply(this, args) + copied += result.length + return result + } + try { + const decoder = new module.BrowserNetworkTunnelStreamFrameDecoder( + () => {}, + (error) => { + throw error + } + ) + for (const chunk of chunks) { + decoder.feed(chunk) + } + } finally { + Uint8Array.prototype.set = originalSet + Uint8Array.prototype.slice = originalSlice + } + return copied +} + +const rows = [] +for (const [payloadBytes, chunkBytes, repetitions] of [ + [1, 5, 10000], + [64 * 1024, 65540, 1000], + [64 * 1024, 4096, 100], + [64 * 1024, 256, 25], + [64 * 1024, 16, 5], + [64 * 1024, 1, 1] +]) { + const payload = Uint8Array.from({ length: payloadBytes }, (_, index) => index % 251) + const encoded = before.encodeBrowserNetworkTunnelStreamFrame(payload) + const chunks = [] + for (let offset = 0; offset < encoded.length; offset += chunkBytes) { + chunks.push(encoded.subarray(offset, offset + chunkBytes)) + } + const beforeMs = measure(before, chunks, payload, repetitions) + const afterMs = measure(after, chunks, payload, repetitions) + rows.push({ + payloadBytes, + chunkBytes, + beforeMs: +beforeMs.toFixed(6), + afterMs: +afterMs.toFixed(6), + speedup: +(beforeMs / afterMs).toFixed(2), + beforeCopiedBytes: countCopies(before, chunks), + afterCopiedBytes: countCopies(after, chunks) + }) +} +console.log(JSON.stringify({ node: process.version, baselineRef, rows }, null, 2)) diff --git a/config/scripts/benchmark-cli-error-imports.mjs b/config/scripts/benchmark-cli-error-imports.mjs new file mode 100644 index 00000000000..a4648f84aec --- /dev/null +++ b/config/scripts/benchmark-cli-error-imports.mjs @@ -0,0 +1,121 @@ +import assert from 'node:assert/strict' +import { createRequire } from 'node:module' +import { existsSync, realpathSync } from 'node:fs' +import { delimiter, join, resolve } from 'node:path' + +// Emit each revision with tsc -p config/tsconfig.cli.json --outDir --composite false --incremental false. +// Run: node config/scripts/benchmark-cli-error-imports.mjs +const [beforeDir, afterDir] = process.argv.slice(2) +assert.ok(beforeDir && afterDir, 'Pass distinct before and after TypeScript output directories.') +assert.notEqual( + realpathSync(beforeDir), + realpathSync(afterDir), + 'Do not compare a build to itself.' +) +const entries = { + before: join(resolve(beforeDir), 'cli', 'index.js'), + after: join(resolve(afterDir), 'cli', 'index.js') +} +for (const entry of Object.values(entries)) { + assert.ok(existsSync(entry), `Missing emitted CLI: ${entry}`) +} + +const { runProcessSync } = createRequire(import.meta.url)( + join(resolve(afterDir), 'shared', 'child-process', 'run-process.js') +) + +const child = String.raw` + const { performance } = require('node:perf_hooks') + const { writeSync } = require('node:fs') + const { createHash } = require('node:crypto') + const { basename } = require('node:path') + let stdout = '', stderr = '' + process.stdout.write = (text) => { stdout += text; return true } + process.stderr.write = (text) => { stderr += text; return true } + const started = performance.now() + const cli = require(process.argv[1]) + const importMs = performance.now() - started + cli.main(JSON.parse(process.argv[2])).then(() => { + const totalMs = performance.now() - started + const modules = Object.keys(require.cache) + writeSync(1, JSON.stringify({ + importMs, totalMs, modules: modules.length, + featureFormatters: modules.filter((file) => ['browser', 'terminal', 'project', 'automation', 'workspace', 'computer'].some((name) => basename(file) === name + '-format.js')), + stdout: createHash('sha256').update(stdout).digest('hex'), + stderr: createHash('sha256').update(stderr).digest('hex'), + exitCode: process.exitCode || 0 + })) + process.exitCode = 0 + }).catch((error) => { writeSync(2, String(error)); process.exitCode = 1 }) +` +const cases = [ + ['--help'], + ['help', 'terminal', 'read'], + ['does-not-exist'], + ['computer', 'click', '--does-not-exist'], + ['does-not-exist', '--json'] +] +const median = (values) => [...values].sort((a, b) => a - b)[Math.floor(values.length / 2)] +const summarize = (samples) => ({ + importMs: median(samples.map((sample) => sample.importMs)), + totalMs: median(samples.map((sample) => sample.totalMs)), + modules: samples[0].modules +}) +const rows = [] +for (const args of cases) { + const samples = { before: [], after: [] } + let expected + for (let run = 0; run < 22; run++) { + for (const variant of run % 2 ? ['after', 'before'] : ['before', 'after']) { + const result = runProcessSync({ + program: process.execPath, + args: ['-e', child, entries[variant], JSON.stringify(args)], + timeoutMs: 30_000, + env: { + ...process.env, + NODE_PATH: [resolve('node_modules'), process.env.NODE_PATH] + .filter(Boolean) + .join(delimiter) + } + }) + assert.equal(result.timedOut, false, 'CLI child timed out.') + assert.equal(result.code, 0, result.stderr) + const sample = JSON.parse(result.stdout) + const output = { stdout: sample.stdout, stderr: sample.stderr, exitCode: sample.exitCode } + expected ??= output + assert.deepEqual(output, expected, `${variant} output changed for ${args.join(' ')}`) + if (variant === 'after') { + assert.deepEqual( + sample.featureFormatters, + [], + 'Help and syntax errors must skip feature formatters.' + ) + } + if (run >= 2) { + samples[variant].push(sample) + } + } + } + assert.ok(samples.after[0].modules < samples.before[0].modules, 'Expected fewer loaded modules.') + rows.push({ + args, + before: summarize(samples.before), + after: summarize(samples.after), + output: expected, + samples + }) +} +console.log( + JSON.stringify( + { + node: process.version, + platform: process.platform, + measurement: + 'Fresh-process import + main; excludes process creation; warmed filesystem; 2 warmups and 20 samples per variant, alternating order.', + entries, + rows + }, + null, + 2 + ) +) diff --git a/config/scripts/benchmark-cli-response-framing.mjs b/config/scripts/benchmark-cli-response-framing.mjs new file mode 100644 index 00000000000..40aab8d08f7 --- /dev/null +++ b/config/scripts/benchmark-cli-response-framing.mjs @@ -0,0 +1,128 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { EventEmitter } from 'node:events' +import { readFileSync } from 'node:fs' +import Module from 'node:module' +import { dirname, resolve } from 'node:path' +import { performance } from 'node:perf_hooks' +import { build } from 'esbuild' + +// Run from the worktree root: node config/scripts/benchmark-cli-response-framing.mjs +const sourcePath = 'src/cli/runtime/transport.ts' +const baselineRef = process.argv[2] +assert.ok(baselineRef, 'Pass the pre-change transport revision as base-ref.') +const beforeSource = execFileSync('git', ['show', `${baselineRef}:${sourcePath}`], { + encoding: 'utf8' +}) +let chunks = [] + +async function loadTransport(source) { + const built = await build({ + stdin: { contents: source, loader: 'ts', resolveDir: dirname(resolve(sourcePath)) }, + bundle: true, + platform: 'node', + format: 'cjs', + write: false, + logLevel: 'silent' + }) + const module = new Module(resolve(sourcePath)) + const originalRequire = module.require.bind(module) + module.require = (name) => { + if (name === 'node:crypto') { + return { randomUUID: () => 'benchmark-request' } + } + if (name !== 'node:net') { + return originalRequire(name) + } + return { + createConnection() { + const socket = new EventEmitter() + socket.setEncoding = () => {} + socket.end = () => {} + socket.destroy = () => {} + socket.write = () => { + for (const chunk of chunks) { + socket.emit('data', chunk) + } + } + queueMicrotask(() => socket.emit('connect')) + return socket + } + } + } + module._compile(built.outputFiles[0].text, resolve(sourcePath)) + return module.exports.sendRequest +} + +const before = await loadTransport(beforeSource) +const after = await loadTransport(readFileSync(sourcePath, 'utf8')) +const metadata = { + runtimeId: 'benchmark-runtime', + authToken: 'benchmark-token', + transports: [{ kind: 'unix', endpoint: 'injected-socket' }] +} +const run = (sendRequest) => sendRequest(metadata, 'terminal.read', {}, 30000) + +async function measure(sendRequest, payloadBytes, repetitions) { + const warmup = await run(sendRequest) + assert.equal(warmup.result.data.length, payloadBytes) + const samples = [] + for (let sample = 0; sample < 5; sample++) { + const start = performance.now() + for (let iteration = 0; iteration < repetitions; iteration++) { + await run(sendRequest) + } + samples.push((performance.now() - start) / repetitions) + } + return samples.sort((a, b) => a - b)[2] +} + +async function searchedCharacters(sendRequest) { + const original = String.prototype.indexOf + let searched = 0 + String.prototype.indexOf = function (needle, position) { + if (needle === '\n') { + searched += this.length - (position ?? 0) + } + return original.call(this, needle, position) + } + try { + await run(sendRequest) + } finally { + String.prototype.indexOf = original + } + return searched +} + +const rows = [] +for (const [payloadBytes, chunkChars, repetitions] of [ + [32, 65536, 1000], + [1024 * 1024, 2 * 1024 * 1024, 20], + [1024 * 1024, 65536, 10], + [1024 * 1024, 4096, 5], + [4 * 1024 * 1024, 4096, 2], + [4 * 1024 * 1024, 256, 1] +]) { + const line = `${JSON.stringify({ + id: 'benchmark-request', + ok: true, + result: { data: 'x'.repeat(payloadBytes) }, + _meta: { runtimeId: 'benchmark-runtime' } + })}\n` + chunks = [] + for (let offset = 0; offset < line.length; offset += chunkChars) { + chunks.push(line.slice(offset, offset + chunkChars)) + } + const beforeMs = await measure(before, payloadBytes, repetitions) + const afterMs = await measure(after, payloadBytes, repetitions) + rows.push({ + payloadBytes, + chunkChars, + beforeMs: +beforeMs.toFixed(6), + afterMs: +afterMs.toFixed(6), + speedup: +(beforeMs / afterMs).toFixed(2), + beforeSearchedCharacters: await searchedCharacters(before), + afterSearchedCharacters: await searchedCharacters(after) + }) +} +console.log(JSON.stringify({ node: process.version, baselineRef, rows }, null, 2)) diff --git a/config/scripts/benchmark-explorer-dotfile-filter.mjs b/config/scripts/benchmark-explorer-dotfile-filter.mjs new file mode 100644 index 00000000000..e66a0ceb5af --- /dev/null +++ b/config/scripts/benchmark-explorer-dotfile-filter.mjs @@ -0,0 +1,165 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import Module from 'node:module' +import { resolve } from 'node:path' +import { performance } from 'node:perf_hooks' +import { build } from 'esbuild' + +// Pass the pre-change file-explorer-entries.ts snapshot as the only argument. +const baselinePath = process.argv[2] +assert.ok(baselinePath, 'Pass a pre-change file-explorer-entries.ts snapshot.') +const entry = 'src/renderer/src/components/right-sidebar/file-explorer-entries.ts' +const baseline = readFileSync(baselinePath, 'utf8') +assert.notEqual(baseline, readFileSync(entry, 'utf8'), 'Do not compare the source to itself.') + +async function load(useBaseline) { + const result = await build({ + stdin: { + contents: `export { isDotfileRelativePath } from './${entry}'; +export { createNameFilteredFileExplorerProjection } from './src/renderer/src/components/right-sidebar/file-explorer-name-filter-projection.ts';`, + resolveDir: process.cwd(), + loader: 'ts' + }, + bundle: true, + platform: 'node', + format: 'cjs', + write: false, + logLevel: 'silent', + alias: { '@': resolve('src/renderer/src') }, + plugins: useBaseline + ? [ + { + name: 'baseline-dotfile-predicate', + setup(builder) { + builder.onLoad({ filter: /file-explorer-entries\.ts$/ }, () => ({ + contents: baseline, + loader: 'ts' + })) + } + } + ] + : [] + }) + const module = new Module(resolve('dotfile-benchmark.cjs')) + module.paths = Module._nodeModulePaths(process.cwd()) + module._compile(result.outputFiles[0].text, module.id) + return module.exports +} + +const versions = [await load(true), await load(false)] +let parityCases = 0 +function check(path, depth) { + assert.equal( + versions[0].isDotfileRelativePath(path), + versions[1].isDotfileRelativePath(path), + path + ) + parityCases++ + if (depth > 0) { + for (const character of ['.', '/', '\\', 'a', '\n']) { + check(path + character, depth - 1) + } + } +} +check('', 8) + +function measure(functions, iterations = 1) { + let sink = 0 + const run = (fn) => { + for (let i = 0; i < iterations; i++) { + sink += Number(fn()) + } + } + for (const fn of functions) { + for (let warmup = 0; warmup < 3; warmup++) { + run(fn) + } + } + const samples = [[], []] + for (let round = 0; round < 11; round++) { + for (const variant of round % 2 ? [1, 0] : [0, 1]) { + const start = performance.now() + run(functions[variant]) + samples[variant].push(performance.now() - start) + } + } + return { + beforeMs: samples[0].sort((a, b) => a - b)[5], + afterMs: samples[1].sort((a, b) => a - b)[5], + iterations, + sink + } +} + +const predicates = [] +for (const path of [ + 'a', + '.env', + 'packages/pkg/src/file.tsx', + `a${'.'.repeat(254)}`, + `${'/'.repeat(4096)}.`, + `${'../'.repeat(1000)}file.ts`, + '😀/.你好', + '\n/.\n' +]) { + check(path, 0) + predicates.push({ + pathLength: path.length, + prefix: path.slice(0, 40), + ...measure( + versions.map((version) => () => version.isDotfileRelativePath(path)), + 10_000 + ) + }) +} + +const projections = [] +for (const count of [1000, 10_000, 100_000]) { + for (const query of ['nonmatching-needle', 'file-42']) { + const args = { + ignoredSet: new Set(['unrelated']), + nameFilter: { + query, + relativePaths: Array.from( + { length: count }, + (_, i) => `packages/package-${i % 50}/src/components/section-${i % 10}/file-${i}.tsx` + ) + }, + showDotfiles: false, + showGitIgnoredFiles: false, + worktreePath: '/workspace' + } + const functions = versions.map( + (version) => () => version.createNameFilteredFileExplorerProjection(args) + ) + const rows = functions.map((fn) => { + const projection = fn() + return Array.from({ length: projection.getVisibleCount() }, (_, i) => + projection.getRowAtIndex(i) + ) + }) + assert.deepEqual(rows[0], rows[1]) + projections.push({ + count, + query, + visibleRows: rows[0].length, + ...measure(functions.map((fn) => () => fn().getVisibleCount())) + }) + } +} +console.log( + JSON.stringify( + { + node: process.version, + platform: process.platform, + baselinePath: resolve(baselinePath), + parityCases, + samples: 11, + warmups: 3, + predicates, + projections + }, + null, + 2 + ) +) diff --git a/config/scripts/benchmark-sentinel-retention.mjs b/config/scripts/benchmark-sentinel-retention.mjs new file mode 100644 index 00000000000..93564eeac01 --- /dev/null +++ b/config/scripts/benchmark-sentinel-retention.mjs @@ -0,0 +1,72 @@ +import { strict as assert } from 'node:assert' +import { EventEmitter } from 'node:events' +import { mkdtemp, rm } from 'node:fs/promises' +import { createRequire } from 'node:module' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import { build } from 'esbuild' + +if (!global.gc) { + throw new Error('Run with node --expose-gc') +} +const root = resolve(import.meta.dirname, '../..') +const directory = await mkdtemp(join(tmpdir(), 'orca-sentinel-retention-')) +const output = join(directory, 'sentinel.cjs') +try { + await build({ + stdin: { + contents: `export {waitForSentinel} from './src/main/ssh/ssh-relay-deploy-helpers'; +export {RELAY_SENTINEL} from './src/main/ssh/relay-protocol';`, + resolveDir: root, + loader: 'ts' + }, + bundle: true, + platform: 'node', + format: 'cjs', + packages: 'external', + banner: { + js: `var require = require('node:module').createRequire(${JSON.stringify(join(root, 'package.json'))});` + }, + outfile: output + }) + const { waitForSentinel, RELAY_SENTINEL } = createRequire(import.meta.url)(output) + const held = [] + const banners = [] + for (let i = 0; i < 100; i++) { + const channel = Object.assign(new EventEmitter(), { + stderr: new EventEmitter(), + stdin: { write: () => true }, + close: () => {} + }) + const pending = waitForSentinel(channel) + banners.push(feedBanner(channel)) + channel.emit('data', Buffer.from(RELAY_SENTINEL)) + const transport = await pending + const received = [] + transport.onData((bytes) => received.push(bytes.toString())) + channel.emit('data', Buffer.from('frame')) + assert.deepEqual(received, ['frame']) + held.push({ channel, transport }) + } + await new Promise((resolve) => setImmediate(resolve)) + for (let i = 0; i < 5; i++) { + global.gc() + } + const retained = banners.filter((reference) => reference.deref() !== undefined).length + console.log( + JSON.stringify({ + connections: held.length, + bannerBytes: 65536, + retainedBannerBuffers: retained, + retainedBannerBytes: retained * 65536 + }) + ) +} finally { + await rm(directory, { recursive: true, force: true }) +} + +function feedBanner(channel) { + const banner = Buffer.alloc(65536, 120) + channel.emit('data', banner) + return new WeakRef(banner.buffer) +} diff --git a/config/scripts/benchmark-skill-depth.mjs b/config/scripts/benchmark-skill-depth.mjs new file mode 100644 index 00000000000..1ebb606f93c --- /dev/null +++ b/config/scripts/benchmark-skill-depth.mjs @@ -0,0 +1,122 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import * as fs from 'node:fs/promises' +import Module from 'node:module' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import { performance } from 'node:perf_hooks' +import { build } from 'esbuild' + +// Pass a pre-change skill-root-file-walk.ts snapshot as the only argument. +const baselinePath = process.argv[2] +const brokenLinks = process.argv.includes('--broken') +assert.ok(baselinePath, 'Pass a pre-change skill-root-file-walk.ts snapshot.') +const entry = 'src/main/skills/skill-root-file-walk.ts' +const baseline = readFileSync(baselinePath, 'utf8') +assert.notEqual(baseline, readFileSync(entry, 'utf8'), 'Do not compare the source to itself.') +let statCalls = 0 + +async function load(useBaseline) { + const result = await build({ + entryPoints: [entry], + bundle: true, + platform: 'node', + format: 'cjs', + write: false, + logLevel: 'silent', + plugins: useBaseline + ? [ + { + name: 'baseline-skill-depth', + setup(builder) { + builder.onLoad({ filter: /skill-root-file-walk\.ts$/ }, () => ({ + contents: baseline, + loader: 'ts' + })) + } + } + ] + : [] + }) + const module = new Module(resolve('skill-depth-benchmark.cjs')) + module.paths = Module._nodeModulePaths(process.cwd()) + const originalRequire = module.require.bind(module) + module.require = (name) => + name === 'node:fs/promises' + ? { + ...fs, + stat: (...args) => { + statCalls++ + return fs.stat(...args) + } + } + : originalRequire(name) + module._compile(result.outputFiles[0].text, module.id) + return module.exports.findSkillFiles +} + +const before = await load(true) +const after = await load(false) +const median = (values) => values.sort((a, b) => a - b)[Math.floor(values.length / 2)] +const temporaryRoot = await fs.mkdtemp(join(tmpdir(), 'orca-skill-depth-benchmark-')) +try { + for (const links of [0, 8, 100, 1000]) { + const root = join(temporaryRoot, String(links)) + const edge = join(root, 'a', 'b', 'c', 'd') + const target = join(temporaryRoot, 'target') + await fs.mkdir(edge, { recursive: true }) + await fs.mkdir(target, { recursive: true }) + await fs.writeFile(join(target, 'SKILL.md'), 'skill') + await fs.writeFile(join(edge, 'SKILL.md'), 'edge') + for (let index = 0; index < links; index++) { + await fs.symlink( + brokenLinks ? join(target, 'missing') : target, + join(edge, `link${index}`), + process.platform === 'win32' ? 'junction' : 'dir' + ) + } + for (const depth of [4, 5]) { + const timings = { before: [], after: [] } + const counts = {} + let rows + for (let sample = 0; sample < 13; sample++) { + const versions = + sample % 2 + ? [ + ['after', after], + ['before', before] + ] + : [ + ['before', before], + ['after', after] + ] + for (const [name, walk] of versions) { + statCalls = 0 + const start = performance.now() + const result = await walk(root, depth) + const elapsed = performance.now() - start + if (rows) { + assert.deepEqual(result, rows) + } + rows = result + counts[name] = statCalls + if (sample >= 2) { + timings[name].push(elapsed) + } + } + } + console.log( + JSON.stringify({ + links, + brokenLinks, + depth, + statCalls: counts, + rows: rows.length, + medianMs: { before: median(timings.before), after: median(timings.after) } + }) + ) + } + } +} finally { + await fs.rm(temporaryRoot, { recursive: true, force: true }) +} diff --git a/config/scripts/benchmark-tab-group-repair.mjs b/config/scripts/benchmark-tab-group-repair.mjs new file mode 100644 index 00000000000..17a1161fc4c --- /dev/null +++ b/config/scripts/benchmark-tab-group-repair.mjs @@ -0,0 +1,80 @@ +import { strict as assert } from 'node:assert' +import { mkdtemp, readFile, rm } from 'node:fs/promises' +import { createRequire } from 'node:module' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import { performance } from 'node:perf_hooks' +import { build } from 'esbuild' + +const root = resolve(import.meta.dirname, '../..') +const source = join(root, 'src/renderer/src/store/slices/tab-group-reference-repair.ts') +const directory = await mkdtemp(join(tmpdir(), 'orca-tab-repair-')) +const current = await readFile(source, 'utf8') +const indexed = `const orderedTabIds = new Set(group.tabOrder) + const missingTabIds = ownedTabIds.filter((tabId) => !orderedTabIds.has(tabId))` +assert(current.includes(indexed), 'Expected indexed implementation') +try { + const implementations = [] + for (const baseline of [true, false]) { + const outfile = join(directory, baseline ? 'before.cjs' : 'after.cjs') + await build({ + stdin: { + contents: baseline + ? current.replace( + indexed, + 'const missingTabIds = ownedTabIds.filter((tabId) => !group.tabOrder.includes(tabId))' + ) + : current, + resolveDir: resolve(source, '..'), + loader: 'ts' + }, + bundle: true, + platform: 'node', + format: 'cjs', + outfile, + alias: { '@': join(root, 'src/renderer/src') } + }) + implementations.push(createRequire(import.meta.url)(outfile).appendOwnedTabIdsToGroups) + } + const rows = [] + for (const count of [1, 10, 100, 1_000, 10_000]) { + for (const missing of [false, true]) { + const ids = Array.from({ length: count }, (_, i) => `tab-${i}`) + const groups = [ + { id: 'group', worktreeId: 'workspace', activeTabId: null, tabOrder: ids, recentTabIds: [] } + ] + const owners = new Map(ids.map((id) => [missing ? `missing-${id}` : id, 'group'])) + assert.deepEqual(implementations[0](groups, owners), implementations[1](groups, owners)) + const iterations = Math.max(1, Math.floor(10_000 / count)) + const samples = [[], []] + for (let sample = -3; sample < 11; sample++) { + for (const index of sample % 2 === 0 ? [0, 1] : [1, 0]) { + const start = performance.now() + for (let i = 0; i < iterations; i++) { + implementations[index](groups, owners) + } + const elapsed = (performance.now() - start) / iterations + if (sample >= 0) { + samples[index].push(elapsed) + } + } + } + rows.push({ + count, + missing, + iterations, + beforeMs: samples[0].sort((a, b) => a - b)[5], + afterMs: samples[1].sort((a, b) => a - b)[5] + }) + } + } + console.log( + JSON.stringify( + { node: process.version, platform: process.platform, samples: 11, warmups: 3, rows }, + null, + 2 + ) + ) +} finally { + await rm(directory, { recursive: true, force: true }) +} diff --git a/config/scripts/benchmark-transcript-reverse-lines.mjs b/config/scripts/benchmark-transcript-reverse-lines.mjs new file mode 100644 index 00000000000..e9d370d1839 --- /dev/null +++ b/config/scripts/benchmark-transcript-reverse-lines.mjs @@ -0,0 +1,124 @@ +import assert from 'node:assert/strict' +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import Module from 'node:module' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import { performance } from 'node:perf_hooks' +import { build } from 'esbuild' + +const entry = 'src/shared/agent-hook-listener/transcript-reader.ts' +assert.ok(process.argv[2], 'Pass a pre-change transcript-reader.ts snapshot.') +const baseline = readFileSync(process.argv[2], 'utf8') +assert.notEqual(baseline, readFileSync(entry, 'utf8'), 'Do not compare the source to itself.') + +async function load(useBaseline) { + const result = await build({ + stdin: { + contents: `export * from './${entry}'; +export { extractAssistantTextFromLine } from './src/shared/agent-hook-listener/transcript-entry-text.ts';`, + resolveDir: process.cwd(), + loader: 'ts' + }, + bundle: true, + platform: 'node', + format: 'cjs', + write: false, + logLevel: 'silent', + plugins: useBaseline + ? [ + { + name: 'baseline-transcript-reader', + setup(builder) { + builder.onLoad({ filter: /transcript-reader\.ts$/ }, () => ({ + contents: baseline, + loader: 'ts' + })) + } + } + ] + : [] + }) + const module = new Module(resolve('transcript-benchmark.cjs')) + module.paths = Module._nodeModulePaths(process.cwd()) + module._compile(result.outputFiles[0].text, module.id) + return module.exports +} + +const versions = [await load(true), await load(false)] +function measure(functions, iterations) { + let sink = 0 + const run = (fn) => { + for (let i = 0; i < iterations; i++) { + sink += fn()?.length ?? 0 + } + } + for (const fn of functions) { + for (let i = 0; i < 3; i++) { + run(fn) + } + } + const samples = [[], []] + for (let round = 0; round < 11; round++) { + for (const index of round % 2 ? [1, 0] : [0, 1]) { + const start = performance.now() + run(functions[index]) + samples[index].push((performance.now() - start) / iterations) + } + } + return { + beforeMs: samples[0].sort((a, b) => a - b)[5], + afterMs: samples[1].sort((a, b) => a - b)[5], + iterations, + sink + } +} + +const cases = [ + ['tiny', `${JSON.stringify({ role: 'assistant', content: 'hello' })}\n`, 10000], + ['64KiB line', `${JSON.stringify({ role: 'assistant', content: 'x'.repeat(65500) })}\n`, 100], + [ + '4MiB line', + `${JSON.stringify({ role: 'assistant', content: 'x'.repeat(4 * 1024 * 1024 - 40) })}\n`, + 10 + ], + [ + '1000 short tool lines', + Array.from({ length: 1000 }, () => + JSON.stringify({ role: 'tool', content: 'x'.repeat(100) }) + ).join('\n'), + 50 + ], + [ + 'Unicode line', + `${JSON.stringify({ role: 'assistant', content: '😀漢字'.repeat(16000) })}\n`, + 100 + ], + [ + 'leading and trailing blank lines', + `\n\r\n${JSON.stringify({ role: 'assistant', content: 'hello' })}\n\n`, + 10000 + ] +] +const directory = mkdtempSync(join(tmpdir(), 'orca-transcript-benchmark-')) +try { + for (const [name, text, iterations] of cases) { + const file = join(directory, 'transcript.jsonl') + writeFileSync(file, text) + const scanners = versions.map( + (v) => () => v.findLastExtractedTranscriptLineText(text, v.extractAssistantTextFromLine) + ) + const readers = versions.map((v) => () => v.readLastAssistantFromTranscriptOnce(file)) + assert.equal(scanners[0](), scanners[1](), name) + assert.equal(readers[0](), readers[1](), name) + console.log( + JSON.stringify({ + name, + bytes: Buffer.byteLength(text), + scanner: measure(scanners, iterations), + warmFileReader: measure(readers, Math.min(iterations, 100)) + }) + ) + } +} finally { + rmSync(directory, { recursive: true, force: true }) +} diff --git a/config/scripts/build-relay.mjs b/config/scripts/build-relay.mjs index 289c7a957bd..4d408712f97 100644 --- a/config/scripts/build-relay.mjs +++ b/config/scripts/build-relay.mjs @@ -57,6 +57,13 @@ const NODE_PTY_CONSOLE_LIST_PATCH_SOURCE = join( 'relay-assets', NODE_PTY_CONSOLE_LIST_PATCH_FILENAME ) +const NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME = 'node-pty-1.1.0-windows-pty-teardown-patch.cjs' +const NODE_PTY_WINDOWS_TEARDOWN_PATCH_SOURCE = join( + ROOT, + 'config', + 'relay-assets', + NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME +) const NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME = 'node-pty-1.1.0-master-cloexec-patch.cjs' const NODE_PTY_MASTER_CLOEXEC_PATCH_SOURCE = join( ROOT, @@ -132,6 +139,10 @@ for (const platform of RELAY_BUILD_PLATFORMS) { NODE_PTY_CONSOLE_LIST_PATCH_SOURCE, join(outDir, NODE_PTY_CONSOLE_LIST_PATCH_FILENAME) ) + copyFileSync( + NODE_PTY_WINDOWS_TEARDOWN_PATCH_SOURCE, + join(outDir, NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME) + ) } copyFileSync( NODE_PTY_MASTER_CLOEXEC_PATCH_SOURCE, diff --git a/config/scripts/build-windows-process-tree-relay-addon.mjs b/config/scripts/build-windows-process-tree-relay-addon.mjs index d3b9db939cd..9243f5a5b78 100644 --- a/config/scripts/build-windows-process-tree-relay-addon.mjs +++ b/config/scripts/build-windows-process-tree-relay-addon.mjs @@ -32,6 +32,8 @@ import { import { join, resolve } from 'node:path' import { RELAY_WINDOWS_PROCESS_TREE_FILENAME } from '../../src/shared/relay-artifacts.ts' import { + ensureWindowsProcessTreeCommandLinePatch, + inspectWindowsProcessTreeAddon, nodeGypRebuildInvocation, stageWindowsProcessTreeNodeAddonApiHeaders, WINDOWS_PROCESS_TREE_PACKAGE_DIR as PACKAGE_DIR @@ -89,6 +91,13 @@ function assertPatchApplied() { 'config/patches/@vscode__windows-process-tree@0.8.0.patch; run pnpm install.' ) } + if (processCc.includes('OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ')) { + throw new Error( + 'src/process.cc still takes PROCESS_VM_READ for memory or CPU counters it never reads ' + + 'from the address space. pnpm did not apply ' + + 'config/patches/@vscode__windows-process-tree@0.8.0.patch; run pnpm install.' + ) + } } // pnpm can materialize this CRLF package without applying its patch. Repair the @@ -123,6 +132,13 @@ function applyWindowsProcessTreeBuildFixes() { '' ) processCc = processCc.replace(/process_count < 1024 && /, '') + // The memory and CPU readers only ever call GetProcessMemoryInfo/GetProcessTimes, + // which need no more than PROCESS_QUERY_LIMITED_INFORMATION; taking VM_READ is + // what EDR scores. + processCc = processCc.replaceAll( + 'OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, false, pid)', + 'OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, false, pid)' + ) if (bindingGyp !== originalBinding) { writeFileSync(bindingPath, bindingGyp) @@ -131,7 +147,8 @@ function applyWindowsProcessTreeBuildFixes() { writeFileSync(processPath, processCc) } stageWindowsProcessTreeNodeAddonApiHeaders(PACKAGE_DIR) - if (bindingGyp !== originalBinding || processCc !== originalProcess) { + const repairedCommandLine = ensureWindowsProcessTreeCommandLinePatch(PACKAGE_DIR) + if (bindingGyp !== originalBinding || processCc !== originalProcess || repairedCommandLine) { console.warn('[windows-process-tree] Repaired un-applied pnpm patch hunks before build.') } } @@ -173,6 +190,14 @@ function main() { if (!existsSync(built)) { throw new Error(`node-gyp reported success but ${built} is missing.`) } + // Why check the artifact and not only the source: the source checks above run + // before node-gyp, and a stale build directory can outlive them. + if (inspectWindowsProcessTreeAddon(built) === 'unpatched') { + throw new Error( + 'The built addon still calls ReadProcessMemory, so it did not come from the patched ' + + 'command-line reader. A relay would get the primitive MDE scores as credential dumping.' + ) + } const machine = readPeMachine(built) if (machine !== PE_MACHINE[arch]) { throw new Error( diff --git a/config/scripts/ci-native-toolchain.test.mjs b/config/scripts/ci-native-toolchain.test.mjs new file mode 100644 index 00000000000..e35437da77c --- /dev/null +++ b/config/scripts/ci-native-toolchain.test.mjs @@ -0,0 +1,69 @@ +import { execFileSync } from 'node:child_process' +import { chmodSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { parse } from 'yaml' +import { describe, expect, it } from 'vitest' + +const steps = parse(readFileSync('.github/actions/install-node-dependencies/action.yml', 'utf8')) + .runs.steps +const toolchain = steps.find((step) => step.name === 'Use external node-gyp') + +describe('CI native toolchain preparation', () => { + it('probes only after both cache restore variants and before native rebuilding', () => { + const index = steps.indexOf(toolchain) + for (const id of ['native-cache-restore', 'native-cache-restore-only']) { + expect(index).toBeGreaterThan(steps.findIndex((step) => step.id === id)) + expect(toolchain.env.NATIVE_CACHE_HIT).toContain(`steps.${id}.outputs.cache-hit`) + } + expect(index).toBeLessThan(steps.findIndex((step) => step.name === 'Prepare native runtime')) + expect(toolchain.if).toBe("runner.os == 'Linux' && inputs.native-runtime != 'none'") + }) + + // The action's toolchain workaround only runs in Linux Bash. + it.skipIf(process.platform === 'win32').each([ + ['node', 'true', '0', false], + ['node', 'true', '1', true], + ['node', 'false', '0', true], + ['node', '', '0', true], + ['electron', 'true', '0', true], + ['electron', 'false', '0', true] + ])('runtime=%s cache=%s probe=%s installs=%s', (runtime, hit, probeStatus, installs) => { + const directory = mkdtempSync(join(tmpdir(), 'orca-ci-native-toolchain-')) + const log = join(directory, 'commands') + const environment = join(directory, 'github-env') + try { + writeFileSync(log, '') + writeFileSync(environment, '') + for (const [name, source] of [ + ['node', 'echo "node $*" >> "$COMMAND_LOG"\nexit "$PROBE_STATUS"'], + ['npm', 'echo "npm $*" >> "$COMMAND_LOG"\nif [ "$1" = root ]; then echo /global; fi'] + ]) { + const path = join(directory, name) + writeFileSync(path, `#!/bin/sh\n${source}\n`) + chmodSync(path, 0o755) + } + execFileSync('bash', ['-e', '-o', 'pipefail', '-c', toolchain.run], { + env: { + ...process.env, + PATH: `${directory}:${process.env.PATH}`, + NATIVE_RUNTIME: runtime, + NATIVE_CACHE_HIT: hit, + PROBE_STATUS: probeStatus, + COMMAND_LOG: log, + GITHUB_ENV: environment + } + }) + const commands = readFileSync(log, 'utf8') + expect(commands.includes('npm install -g node-gyp@11.5.0')).toBe(installs) + expect(commands.includes('node config/scripts/ensure-native-runtime.mjs --check-only')).toBe( + runtime === 'node' && hit === 'true' + ) + expect(readFileSync(environment, 'utf8')).toBe( + installs ? 'npm_config_node_gyp=/global/node-gyp/bin/node-gyp.js\n' : '' + ) + } finally { + rmSync(directory, { recursive: true, force: true }) + } + }) +}) diff --git a/config/scripts/cli-runtime-client-deferral-equivalence.mjs b/config/scripts/cli-runtime-client-deferral-equivalence.mjs index f443bf3b9f9..a231b5ba75a 100644 --- a/config/scripts/cli-runtime-client-deferral-equivalence.mjs +++ b/config/scripts/cli-runtime-client-deferral-equivalence.mjs @@ -2,7 +2,7 @@ // Equivalence check for deferring the RuntimeClient module graph in the CLI. // // Builds the CLI twice with the REAL tsc emit — once from the working tree and -// once with the seven touched files restored from git HEAD~ (the pre-deferral +// once with the touched files restored from git HEAD~ (the pre-deferral // implementation) — then compares stdout, stderr and exit code BYTE FOR BYTE // across a matrix of invocations. // @@ -13,7 +13,7 @@ // // Usage: node config/scripts/cli-runtime-client-deferral-equivalence.mjs [--baseline ] import { execFileSync, spawnSync } from 'node:child_process' -import { mkdirSync, mkdtempSync, rmSync, writeFileSync, readFileSync } from 'node:fs' +import { existsSync, mkdirSync, mkdtempSync, rmSync, writeFileSync, readFileSync } from 'node:fs' import { join, resolve } from 'node:path' import { fileURLToPath } from 'node:url' @@ -21,8 +21,11 @@ const REPO = fileURLToPath(new URL('../..', import.meta.url)) // The files this change touches. Restoring exactly these from the baseline rev // reconstructs the old implementation without disturbing anything else. +// Files absent at the baseline (e.g. cli-error.ts, split out of format.ts +// later) are removed for the baseline build and put back afterwards. const TOUCHED = [ 'src/cli/args.ts', + 'src/cli/cli-error.ts', 'src/cli/dispatch.ts', 'src/cli/flags.ts', 'src/cli/format.ts', @@ -72,12 +75,16 @@ function buildTree(label, baselineRev) { if (baselineRev) { for (const file of TOUCHED) { const path = join(REPO, file) - restored.push([path, readFileSync(path)]) - const old = execFileSync('git', ['show', `${baselineRev}:${file}`], { + restored.push([path, existsSync(path) ? readFileSync(path) : null]) + const old = spawnSync('git', ['show', `${baselineRev}:${file}`], { cwd: REPO, maxBuffer: 64 * 1024 * 1024 }) - writeFileSync(path, old) + if (old.status === 0) { + writeFileSync(path, old.stdout) + } else { + rmSync(path, { force: true }) + } } } execFileSync( @@ -97,7 +104,11 @@ function buildTree(label, baselineRev) { ) } finally { for (const [path, contents] of restored) { - writeFileSync(path, contents) + if (contents === null) { + rmSync(path, { force: true }) + } else { + writeFileSync(path, contents) + } } } return join(outDir, 'cli/index.js') diff --git a/config/scripts/create-draft-release.mjs b/config/scripts/create-draft-release.mjs index 3412a118491..1732e9a1e8a 100644 --- a/config/scripts/create-draft-release.mjs +++ b/config/scripts/create-draft-release.mjs @@ -128,10 +128,14 @@ export async function createDraftRelease({ throw new Error('token is required') } - const previousTag = latestPreviousPublishedDesktopReleaseTag( - await fetchRepoReleases(repo, token, fetchImpl), - tag - ) + const releases = await fetchRepoReleases(repo, token, fetchImpl) + const existingRelease = releases.find((release) => release?.tag_name === tag) + if (existingRelease && existingRelease.draft !== true) { + log(`Release ${tag} already exists and is published.`) + return + } + + const previousTag = latestPreviousPublishedDesktopReleaseTag(releases, tag) const generateNotesBody = { tag_name: tag, target_commitish: tag, @@ -156,24 +160,90 @@ export async function createDraftRelease({ typeof releaseNotes.name === 'string' && releaseNotes.name.length > 0 ? releaseNotes.name : tag const prerelease = tag.includes('-rc.') - // Why: GitHub's generated release notes can exceed the release body API - // limit, so create with a bounded body. Omit target_commitish because the - // release-cut tag already exists and GitHub rejects the tag name there. - await githubJson(fetchImpl, `https://api.github.com/repos/${repo}/releases`, token, { - method: 'POST', - body: JSON.stringify({ - tag_name: tag, - name, - body, - draft: true, - prerelease + if (existingRelease) { + if (!Number.isInteger(existingRelease.id)) { + throw new Error(`Draft release ${tag} is missing a GitHub release id`) + } + // Why: the listing is a snapshot; the draft can be published while notes + // generate, and patching then overwrites a live release body. + const currentRelease = await githubJson( + fetchImpl, + `https://api.github.com/repos/${repo}/releases/${existingRelease.id}`, + token + ) + if (currentRelease?.draft !== true) { + log(`Release ${tag} was published while notes were generated; leaving it unchanged.`) + return + } + // Why: the PATCH endpoint supports no conditional/versioned update, so the + // GET above cannot close the window. The PATCH response reports the state we + // actually wrote to; if publication won, put the published body back. + const patchedRelease = await githubJson( + fetchImpl, + `https://api.github.com/repos/${repo}/releases/${existingRelease.id}`, + token, + { + method: 'PATCH', + body: JSON.stringify({ body }) + } + ) + if (patchedRelease?.draft !== true) { + const publishedBody = typeof currentRelease.body === 'string' ? currentRelease.body : '' + if (publishedBody === body) { + log(`Release ${tag} was published while notes were patched; its body is unchanged.`) + return + } + // Why: the rollback must not clobber a body written after our PATCH, so + // restore only while the release still carries exactly what we wrote. + const releaseBeforeRollback = await githubJson( + fetchImpl, + `https://api.github.com/repos/${repo}/releases/${existingRelease.id}`, + token + ) + if (releaseBeforeRollback?.body !== body) { + log( + `Release ${tag} was published and its body changed again while notes were patched; leaving the newer body in place.` + ) + return + } + await githubJson( + fetchImpl, + `https://api.github.com/repos/${repo}/releases/${existingRelease.id}`, + token, + { + method: 'PATCH', + body: JSON.stringify({ body: publishedBody }) + } + ) + log( + `Release ${tag} was published while notes were patched; restored its published body and left the generated notes unapplied.` + ) + return + } + } else { + // Why: GitHub's generated release notes can exceed the release body API + // limit, so create with a bounded body. Omit target_commitish because the + // release-cut tag already exists and GitHub rejects the tag name there. + await githubJson(fetchImpl, `https://api.github.com/repos/${repo}/releases`, token, { + method: 'POST', + body: JSON.stringify({ + tag_name: tag, + name, + body, + draft: true, + prerelease + }) }) - }) + } if (generatedBody.length !== body.length) { - log(`Created draft release ${tag} with truncated generated notes (${body.length} chars).`) + log( + `${existingRelease ? 'Updated' : 'Created'} draft release ${tag} with truncated generated notes (${body.length} chars).` + ) } else { - log(`Created draft release ${tag} with generated notes (${body.length} chars).`) + log( + `${existingRelease ? 'Updated' : 'Created'} draft release ${tag} with generated notes (${body.length} chars).` + ) } } diff --git a/config/scripts/create-draft-release.test.mjs b/config/scripts/create-draft-release.test.mjs index b330ccb9423..911ac00be63 100644 --- a/config/scripts/create-draft-release.test.mjs +++ b/config/scripts/create-draft-release.test.mjs @@ -132,7 +132,7 @@ describe('createDraftRelease', () => { it('creates a draft release with bounded generated notes', async () => { const fetchImpl = vi .fn() - .mockResolvedValueOnce(jsonResponse([release('v1.4.35'), release('v1.4.36')])) + .mockResolvedValueOnce(jsonResponse([release('v1.4.35')])) .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36', body: 'a'.repeat(130_000) })) .mockResolvedValueOnce(jsonResponse({ tag_name: 'v1.4.36', draft: true })) @@ -184,7 +184,7 @@ describe('createDraftRelease', () => { it('marks rc tags as prereleases', async () => { const fetchImpl = vi .fn() - .mockResolvedValueOnce(jsonResponse([release('v1.4.36'), release('v1.4.36-rc.1')])) + .mockResolvedValueOnce(jsonResponse([release('v1.4.36')])) .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36-rc.1', body: 'notes' })) .mockResolvedValueOnce(jsonResponse({ tag_name: 'v1.4.36-rc.1', draft: true })) @@ -200,10 +200,136 @@ describe('createDraftRelease', () => { expect(createBody.prerelease).toBe(true) }) + it('regenerates notes for an existing draft release', async () => { + const fetchImpl = vi + .fn() + .mockResolvedValueOnce( + jsonResponse([release('v1.4.35'), release('v1.4.36', { draft: true, id: 42 })]) + ) + .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36', body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: true, body: 'stale' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: true, body: 'notes' })) + + await createDraftRelease({ + repo: 'stablyai/orca', + tag: 'v1.4.36', + token: 'token', + fetchImpl, + log: vi.fn() + }) + + expect(fetchImpl).toHaveBeenNthCalledWith( + 3, + 'https://api.github.com/repos/stablyai/orca/releases/42', + expect.not.objectContaining({ method: expect.anything() }) + ) + expect(fetchImpl).toHaveBeenNthCalledWith( + 4, + 'https://api.github.com/repos/stablyai/orca/releases/42', + expect.objectContaining({ method: 'PATCH', body: JSON.stringify({ body: 'notes' }) }) + ) + }) + + it('skips the update when the draft was published while notes were generated', async () => { + const fetchImpl = vi + .fn() + .mockResolvedValueOnce( + jsonResponse([release('v1.4.35'), release('v1.4.36', { draft: true, id: 42 })]) + ) + .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36', body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: false })) + + await createDraftRelease({ + repo: 'stablyai/orca', + tag: 'v1.4.36', + token: 'token', + fetchImpl, + log: vi.fn() + }) + + expect(fetchImpl).toHaveBeenCalledTimes(3) + expect(fetchImpl).toHaveBeenNthCalledWith( + 3, + 'https://api.github.com/repos/stablyai/orca/releases/42', + expect.not.objectContaining({ method: expect.anything() }) + ) + }) + + it('restores the published body when publication lands between the check and the patch', async () => { + const log = vi.fn() + const fetchImpl = vi + .fn() + .mockResolvedValueOnce( + jsonResponse([release('v1.4.35'), release('v1.4.36', { draft: true, id: 42 })]) + ) + .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36', body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: true, body: 'hand-written notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: false, body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: false, body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: false, body: 'hand-written notes' })) + + await createDraftRelease({ + repo: 'stablyai/orca', + tag: 'v1.4.36', + token: 'token', + fetchImpl, + log + }) + + expect(fetchImpl).toHaveBeenCalledTimes(6) + expect(fetchImpl).toHaveBeenNthCalledWith( + 6, + 'https://api.github.com/repos/stablyai/orca/releases/42', + expect.objectContaining({ + method: 'PATCH', + body: JSON.stringify({ body: 'hand-written notes' }) + }) + ) + expect(log).toHaveBeenCalledWith(expect.stringContaining('restored its published body')) + }) + + it('leaves a body written after the patch in place instead of rolling it back', async () => { + const log = vi.fn() + const fetchImpl = vi + .fn() + .mockResolvedValueOnce( + jsonResponse([release('v1.4.35'), release('v1.4.36', { draft: true, id: 42 })]) + ) + .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36', body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: true, body: 'hand-written notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: false, body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: false, body: 'newer published body' })) + + await createDraftRelease({ + repo: 'stablyai/orca', + tag: 'v1.4.36', + token: 'token', + fetchImpl, + log + }) + + expect(fetchImpl).toHaveBeenCalledTimes(5) + expect(log).toHaveBeenCalledWith(expect.stringContaining('leaving the newer body in place')) + }) + + it('preserves notes on an existing published release', async () => { + const fetchImpl = vi.fn().mockResolvedValueOnce(jsonResponse([release('v1.4.36', { id: 42 })])) + + await createDraftRelease({ + repo: 'stablyai/orca', + tag: 'v1.4.36', + token: 'token', + fetchImpl, + log: vi.fn() + }) + + expect(fetchImpl).toHaveBeenCalledTimes(1) + }) + it('omits previous_tag_name for the first desktop release so notes fall back to the GitHub default', async () => { const fetchImpl = vi .fn() - .mockResolvedValueOnce(jsonResponse([release('v1.4.36'), release('mobile-v0.0.12')])) + .mockResolvedValueOnce(jsonResponse([release('mobile-v0.0.12')])) .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36', body: 'notes' })) .mockResolvedValueOnce(jsonResponse({ tag_name: 'v1.4.36', draft: true })) diff --git a/config/scripts/electron-builder-markdown-associations.test.mjs b/config/scripts/electron-builder-markdown-associations.test.mjs index 7ae3b1c9428..58f6f8d8865 100644 --- a/config/scripts/electron-builder-markdown-associations.test.mjs +++ b/config/scripts/electron-builder-markdown-associations.test.mjs @@ -103,14 +103,24 @@ describe('electron-builder markdown file associations', () => { // Why: this include was renamed from daemon-host-uninstall.nsh to carry the markdown // hooks too. electron-builder allows only one include, so a merge that drops the daemon - // sweep would silently orphan a running orca-terminal-daemon.exe on every uninstall. + // sweep would silently orphan a running daemon host on every uninstall. + // + // Asserted against comment-stripped script, and on the app exe name first: the relocated + // host is a verbatim copy of the app exe (daemonHostExeName, daemon-host-relocation.ts), + // so a macro that kills only orca-terminal-daemon.exe matches no running process. The + // prose above the macro names both, so a toContain over the raw file proves nothing. it('keeps the daemon-host uninstall sweep across the include rename', async () => { - const hooks = await readInstallerHooks() + const script = stripNsisCommentLines(await readInstallerHooks()) - expect(hooks).toContain('orca-terminal-daemon.exe') - expect(hooks).toContain('$LOCALAPPDATA\\Orca\\daemon-host') + expect(script).toMatch(/taskkill[^\n]*\/IM\s+"?\$\{APP_EXECUTABLE_FILENAME\}"?/) + // Legacy name, so hosts left by builds that renamed the copy still get reaped. + expect(script).toMatch(/taskkill[^\n]*\/IM\s+"?orca-terminal-daemon\.exe"?/) + // Scopes both kills to the uninstalling user: an elevated machine-wide uninstall must + // not reach another logged-on user's session. + expect(script).toMatch(/\/FI\s+"USERNAME eq /) + expect(script).toContain('$LOCALAPPDATA\\Orca\\daemon-host') // Without this guard, uninstallOldVersion would kill the daemon on every update — // defeating the relocation that keeps terminals alive across updates. - expect(hooks).toMatch(/\$\{ifNot\}\s+\$\{isUpdated\}/) + expect(script).toMatch(/\$\{ifNot\}\s+\$\{isUpdated\}/) }) }) diff --git a/config/scripts/ensure-native-runtime.mjs b/config/scripts/ensure-native-runtime.mjs index a4cc6db8843..b2a47b99d5b 100644 --- a/config/scripts/ensure-native-runtime.mjs +++ b/config/scripts/ensure-native-runtime.mjs @@ -5,6 +5,12 @@ import { createRequire } from 'node:module' import { existsSync, readFileSync } from 'node:fs' import { release } from 'node:os' import { basename, dirname, resolve } from 'node:path' +import { + ensureWindowsProcessTreeCommandLinePatch, + inspectWindowsProcessTreeAddon, + stageWindowsProcessTreeNodeAddonApiHeaders, + windowsProcessTreeAddonPath +} from './windows-process-tree-gyp-rebuild.mjs' const require = createRequire(import.meta.url) const { assertNodePtyJobOwnership } = require('./node-pty-job-ownership.cjs') @@ -253,11 +259,18 @@ function collectNativeModuleFailures() { function loadNativeModule(moduleName) { if (moduleName === '@vscode/windows-process-tree') { - // A bare require already loads the .node addon on win32, so it catches an - // ABI mismatch on its own. What it cannot catch is a snapshot that comes - // back empty -- the shape a blocked CreateToolhelp32Snapshot produces -- - // so check the addon actually enumerates before calling the runtime healthy. + // A bare require loads the .node addon on win32, so it catches an ABI + // mismatch on its own. What it cannot catch is *which* addon loaded: the + // published tarball ships a prebuilt built from unpatched source that is + // node-addon-api, so it requires cleanly and then reads every process's + // command line out of its address space. Check the binary, not the load. require(moduleName) + if (inspectWindowsProcessTreeAddon(windowsProcessTreeAddonPath()) === 'unpatched') { + throw new Error( + 'the loaded addon still calls ReadProcessMemory, so it was not built from the patched ' + + 'source. Rebuild it (pnpm run rebuild:electron) rather than using the published prebuild.' + ) + } return } if (moduleName === 'windows-native-registry') { @@ -368,6 +381,14 @@ function getWindowsBuildNumber() { function rebuildNodeRuntimeModules(moduleNames) { for (const moduleName of moduleNames) { const moduleDir = dirname(require.resolve(`${moduleName}/package.json`)) + if (moduleName === '@vscode/windows-process-tree') { + // Why before node-gyp: this module is rebuilt precisely because the + // binary was the unpatched one, and pnpm materializes it unpatched often + // enough that compiling the source as-is would just rebuild the same + // reader and fail the verify pass. + ensureWindowsProcessTreeCommandLinePatch(moduleDir) + stageWindowsProcessTreeNodeAddonApiHeaders(moduleDir) + } console.warn(`[native-runtime] Rebuilding ${moduleName} with node-gyp.`) runPnpm(['exec', 'node-gyp', 'rebuild'], { cwd: moduleDir }) if (moduleName === 'node-pty' && process.platform === 'win32') { diff --git a/config/scripts/ensure-native-runtime.test.mjs b/config/scripts/ensure-native-runtime.test.mjs index ea6e876e619..973e2f6852d 100644 --- a/config/scripts/ensure-native-runtime.test.mjs +++ b/config/scripts/ensure-native-runtime.test.mjs @@ -12,6 +12,7 @@ import { tmpdir } from 'node:os' import { delimiter, join } from 'node:path' import { fileURLToPath } from 'node:url' import { describe, expect, it } from 'vitest' +import { copyScriptWithLocalModules } from './script-module-dependencies.mjs' const sourceScriptPath = fileURLToPath(new URL('./ensure-native-runtime.mjs', import.meta.url)) const sourceNodePtyJobOwnershipPath = fileURLToPath( @@ -27,7 +28,6 @@ describe('ensure-native-runtime', () => { const logPath = join(projectDir, 'native-runtime.log') const markerPath = join(projectDir, 'rebuilt.marker') const binDir = join(projectDir, 'bin') - copyFileSync(sourceScriptPath, scriptPath) writeFakeNativeModules(projectDir) writeNodePtyPatchFile(projectDir) writeFakePnpm(binDir) @@ -67,7 +67,6 @@ describe('ensure-native-runtime', () => { const logPath = join(projectDir, 'native-runtime.log') const markerPath = join(projectDir, 'rebuilt.marker') const binDir = join(projectDir, 'bin') - copyFileSync(sourceScriptPath, scriptPath) writeFakeNativeModules(projectDir, { windowsRegistryRequiresMarker: true }) writeNodePtyPatchFile(projectDir) writeFakePnpm(binDir) @@ -102,7 +101,6 @@ describe('ensure-native-runtime', () => { const logPath = join(projectDir, 'native-runtime.log') const markerPath = join(projectDir, 'rebuilt.marker') const binDir = join(projectDir, 'bin') - copyFileSync(sourceScriptPath, scriptPath) writeLoadableNativeModules(projectDir) writeNodePtyPatchFile(projectDir) writeFakePnpm(binDir) @@ -137,7 +135,6 @@ describe('ensure-native-runtime', () => { const logPath = join(projectDir, 'native-runtime.log') const markerPath = join(projectDir, 'rebuilt.marker') const binDir = join(projectDir, 'bin') - copyFileSync(sourceScriptPath, scriptPath) writeLoadableNativeModules(projectDir) writeNodePtyPatchFile(projectDir) writePatchedNodePtyBuildArtifacts(projectDir) @@ -171,7 +168,6 @@ describe('ensure-native-runtime', () => { const logPath = join(projectDir, 'native-runtime.log') const markerPath = join(projectDir, 'rebuilt.marker') const binDir = join(projectDir, 'bin') - copyFileSync(sourceScriptPath, scriptPath) writeLoadableNativeModules(projectDir, { nativeDir: '../build/Release/' }) writeNodePtyPatchFile(projectDir) writePatchedNodePtyBuildArtifacts(projectDir) @@ -198,7 +194,9 @@ describe('ensure-native-runtime', () => { function mkTempProject() { const projectDir = mkdtempSync(join(tmpdir(), 'orca-native-runtime-')) - mkdirSync(join(projectDir, 'config', 'scripts'), { recursive: true }) + // Walked, not listed: the script imports windows-process-tree-gyp-rebuild.mjs, and a fixture + // missing it fails every case with a module-resolution error instead of the defect under test. + copyScriptWithLocalModules(sourceScriptPath, join(projectDir, 'config', 'scripts')) copyFileSync( sourceNodePtyJobOwnershipPath, join(projectDir, 'config', 'scripts', 'node-pty-job-ownership.cjs') diff --git a/config/scripts/file-explorer-deletion-roots-benchmark.mjs b/config/scripts/file-explorer-deletion-roots-benchmark.mjs new file mode 100644 index 00000000000..d276964861b --- /dev/null +++ b/config/scripts/file-explorer-deletion-roots-benchmark.mjs @@ -0,0 +1,75 @@ +import assert from 'node:assert/strict' +import { join } from 'node:path' +import { performance } from 'node:perf_hooks' +import { fileURLToPath } from 'node:url' +import { build } from 'esbuild' + +const root = fileURLToPath(new URL('../..', import.meta.url)) +const bundled = await build({ + stdin: { + contents: `export { selectDeletionRoots } from './file-explorer-batch-deletion'; + export { isPathEqualOrDescendant } from './file-explorer-paths';`, + resolveDir: join(root, 'src/renderer/src/components/right-sidebar'), + loader: 'ts' + }, + alias: { '@': join(root, 'src/renderer/src') }, + bundle: true, + platform: 'node', + format: 'esm', + write: false, + logLevel: 'silent' +}) +const { selectDeletionRoots, isPathEqualOrDescendant } = await import( + `data:text/javascript;base64,${Buffer.from(bundled.outputFiles[0].text).toString('base64')}` +) + +// Original production selector; both paths use the same path-comparison implementation. +function original(nodes) { + return nodes.filter( + (n) => + !nodes.some( + (other) => other !== n && other.isDirectory && isPathEqualOrDescendant(n.path, other.path) + ) + ) +} + +function measure(run, nodes) { + for (let index = 0; index < 3; index++) { + run(nodes) + } + const samples = [] + for (let index = 0; index < 11; index++) { + const start = performance.now() + run(nodes) + samples.push(performance.now() - start) + } + return samples.sort((a, b) => a - b)[5] +} + +const results = [] +for (const [fileCount, directoryCount] of [ + [100, 0], + [1000, 0], + [5000, 0], + [5000, 5], + [0, 100] +]) { + const nodes = Array.from({ length: fileCount + directoryCount }, (_, index) => ({ + name: `item-${index}`, + path: `/repo/item-${index}`, + relativePath: `item-${index}`, + isDirectory: index >= fileCount, + depth: 0 + })) + const expected = original(nodes) + const actual = selectDeletionRoots(nodes) + assert.equal(actual.length, expected.length) + actual.forEach((node, index) => assert.equal(node, expected[index])) + results.push({ + fileCount, + directoryCount, + beforeMs: measure(original, nodes), + afterMs: measure(selectDeletionRoots, nodes) + }) +} +console.log(JSON.stringify({ node: process.version, platform: process.platform, results }, null, 2)) diff --git a/config/scripts/hourly-preflight-workflow.test.mjs b/config/scripts/hourly-preflight-workflow.test.mjs new file mode 100644 index 00000000000..2bec40b329b --- /dev/null +++ b/config/scripts/hourly-preflight-workflow.test.mjs @@ -0,0 +1,91 @@ +import { mkdtempSync, readFileSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { parse } from 'yaml' +import { runProcess } from '../../src/shared/child-process/run-process' + +const workflow = parse( + readFileSync(new URL('../../.github/workflows/hourly-mac-build.yml', import.meta.url), 'utf8') +) +const preflight = workflow.jobs.preflight +const freshness = preflight.steps.find((step) => step.id === 'freshness') +const head = 'abcdef0123'.repeat(4) + +async function checkFreshness(overrides = {}) { + const directory = mkdtempSync(join(tmpdir(), 'hourly-preflight-')) + const output = join(directory, 'output') + try { + const result = await runProcess({ + program: 'bash', + args: [ + '-c', + `gh() { + case "$1 $2" in + "api "*) printf '%s\\n' "$HEAD_SHA" ;; + "release list") printf '%s\\n' "$LAST_TAG" ;; + "release view") printf '%s\\n' "$LAST_SHA" ;; + *) return 1 ;; + esac + } + ${freshness.run}` + ], + env: { + ...process.env, + GITHUB_OUTPUT: output, + GITHUB_REPOSITORY: 'stablyai/orca', + MAIN_REPO_TOKEN: 'main-token', + HOURLY_REPO: 'stablyai/orca-hourly', + HEAD_SHA: head, + LAST_TAG: 'previous-hourly', + LAST_SHA: head.slice(0, 12), + FORCED: 'false', + ...overrides + } + }) + return { + exitCode: result.code, + stderr: result.stderr, + stdout: result.stdout, + output: result.code === 0 ? readFileSync(output, 'utf8') : '' + } + } finally { + rmSync(directory, { recursive: true, force: true }) + } +} + +describe('hourly build preflight', () => { + it('gates Mac allocation and pins the checkout and downstream identity', () => { + const build = workflow.jobs['build-hourly-mac'] + expect(preflight['runs-on']).toBe('ubuntu-latest') + expect(preflight.steps.some((step) => step.uses?.startsWith('actions/checkout'))).toBe(false) + expect( + preflight.steps.find((step) => step.id === 'app_token').with['permission-contents'] + ).toBe('read') + expect(build.needs).toBe('preflight') + expect(build.if).toBe("needs.preflight.outputs.should_build == 'true'") + expect(build.steps.find((step) => step.name === 'Checkout').with.ref).toBe( + build.outputs.head_sha + ) + expect(build.outputs.head_sha).toBe('${{ needs.preflight.outputs.head_sha }}') + expect(build.steps.find((step) => step.id === 'release').env.SHA).toBe(build.outputs.head_sha) + expect(workflow.concurrency).toEqual({ group: 'hourly-mac-build', 'cancel-in-progress': false }) + }) + + it.each([ + ['unchanged', {}, false], + ['changed', { LAST_SHA: '123456789012' }, true], + ['forced', { FORCED: 'true' }, true], + ['first build', { LAST_TAG: '' }, true], + ['missing prior identity', { LAST_SHA: '' }, true] + ])('%s main selects the expected build decision', async (_name, env, shouldBuild) => { + const result = await checkFreshness(env) + expect(result.exitCode, `${result.stdout} ${result.stderr}`).toBe(0) + expect(result.output).toBe(`head_sha=${head}\nshould_build=${shouldBuild}\n`) + }) + + it('fails closed when main cannot be resolved, even when forced', async () => { + const result = await checkFreshness({ HEAD_SHA: '', FORCED: 'true' }) + expect(result.exitCode).not.toBe(0) + }) +}) diff --git a/config/scripts/locale-collator-sort-benchmark.mjs b/config/scripts/locale-collator-sort-benchmark.mjs index 8a68331bd44..3d34466532a 100644 --- a/config/scripts/locale-collator-sort-benchmark.mjs +++ b/config/scripts/locale-collator-sort-benchmark.mjs @@ -3,10 +3,12 @@ import { performance } from 'node:perf_hooks' import { fileURLToPath } from 'node:url' import { createJiti } from 'jiti' +import { buildCounterbalancedSchedule } from './counterbalanced-benchmark-schedule.mjs' -const ROUND_COUNT = 5 +const ROUND_COUNT = 6 const MIN_ROUND_MS = 120 const jiti = createJiti(import.meta.url, { + jsx: true, alias: { '@': fileURLToPath(new URL('../../src/renderer/src', import.meta.url)) } }) const { compareBaseSensitivityLocaleText } = await jiti.import( @@ -15,6 +17,10 @@ const { compareBaseSensitivityLocaleText } = await jiti.import( const { sortJiraIssues } = await jiti.import( '../../src/renderer/src/components/jira-issue-sorter.ts' ) +const { sortAutomationListViewItems } = await jiti.import( + '../../src/renderer/src/components/automations/automation-list-view.ts' +) +const { getIntlLocale } = await jiti.import('../../src/renderer/src/i18n/i18n.ts') let randomState = 0x9e3779b9 function random() { @@ -83,20 +89,21 @@ function measurePair(before, after) { const afterIterations = calibrate(after) const beforeSamples = [] const afterSamples = [] - for (let round = 0; round < ROUND_COUNT; round += 1) { - if (round % 2 === 0) { - beforeSamples.push(measureRound(before, beforeIterations)) - afterSamples.push(measureRound(after, afterIterations)) - } else { - afterSamples.push(measureRound(after, afterIterations)) - beforeSamples.push(measureRound(before, beforeIterations)) + for (const pair of buildCounterbalancedSchedule(ROUND_COUNT, 'before', 'after')) { + for (const arm of pair) { + if (arm === 'before') { + beforeSamples.push(measureRound(before, beforeIterations)) + } else { + afterSamples.push(measureRound(after, afterIterations)) + } } } - const middle = Math.floor(ROUND_COUNT / 2) - return { - beforeMs: beforeSamples.sort((a, b) => a - b)[middle], - afterMs: afterSamples.sort((a, b) => a - b)[middle] + const median = (samples) => { + samples.sort((a, b) => a - b) + const middle = samples.length / 2 + return (samples[middle - 1] + samples[middle]) / 2 } + return { beforeMs: median(beforeSamples), afterMs: median(afterSamples) } } function assertSameOrder(before, after, label) { @@ -111,7 +118,9 @@ function assertSameOrder(before, after, label) { } const pad = (value, width) => String(value).padStart(width) -console.log('Renderer locale sort, ms per sort (median of 5 rounds). Lower is better.') +console.log( + 'Renderer locale sort, ms per sort (median of 6 counterbalanced rounds). Lower is better.' +) console.log( `${pad('mode', 9)} ${pad('items', 7)} ${pad('per-call', 11)} ${pad('reused', 11)} ${pad('speedup', 9)}` ) @@ -145,3 +154,31 @@ for (const count of [10, 50, 250]) { console.log( '\n36 rows matches the Linear page size, 50 matches the picker/Jira scale, and\n250 is a stress case. Both arms assert identical output before timing.' ) + +for (const count of [10, 100, 1000]) { + const items = makeBaseSensitivityValues(count).map((name, index) => ({ + id: `automation-${index}`, + name, + lastRunAt: null + })) + const before = () => { + const locale = getIntlLocale() + function compare(left, right) { + return ( + left.name.localeCompare(right.name, locale, { sensitivity: 'base' }) || + left.id.localeCompare(right.id) + ) + } + return [...items].sort(compare).map((item) => item.id) + } + const after = () => + sortAutomationListViewItems(items, { field: 'name', direction: 'asc' }).map((item) => item.id) + assertSameOrder(before, after, `automation ${count}`) + const { beforeMs, afterMs } = measurePair(before, after) + console.log( + `${pad('automation', 10)} ${pad(count, 7)} ${pad(`${beforeMs.toFixed(3)} ms`, 11)} ${pad(`${afterMs.toFixed(3)} ms`, 11)} ${pad(`${(beforeMs / afterMs).toFixed(1)}x`, 9)}` + ) +} +console.log( + 'Automation arm calls the production sorter; 1000 rows is a scaling fixture, not a measured user inventory. No timing gate.' +) diff --git a/config/scripts/locale-ko-key-overrides.json b/config/scripts/locale-ko-key-overrides.json index f368ecc3cbc..bf5f62d1fa5 100644 --- a/config/scripts/locale-ko-key-overrides.json +++ b/config/scripts/locale-ko-key-overrides.json @@ -492,7 +492,7 @@ "ko": "agent CLI를 찾지 못했습니다. 하나를 설치하거나 설정에서 기본 agent를 선택하세요." }, "auto.components.Terminal.7958465754": { - "ko": "실행 중인 프로세스가 있는 로컬 terminals이 있습니다. 그래도 창을 닫으시겠습니까?" + "ko": "실행 중인 프로세스가 있는 terminals이 있습니다. 그래도 창을 닫으시겠습니까?" }, "auto.components.Terminal.cdc9ac4b2d": { "ko": "편집기" diff --git a/config/scripts/mobile-file-ranking-benchmark.mjs b/config/scripts/mobile-file-ranking-benchmark.mjs new file mode 100644 index 00000000000..68ac5b9d977 --- /dev/null +++ b/config/scripts/mobile-file-ranking-benchmark.mjs @@ -0,0 +1,53 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { readFileSync } from 'node:fs' +import { stripTypeScriptTypes } from 'node:module' +import { performance } from 'node:perf_hooks' + +const baseline = process.argv[2] +if (!baseline) { + throw new Error('Usage: node config/scripts/mobile-file-ranking-benchmark.mjs ') +} +async function load(source) { + const js = stripTypeScriptTypes(source, { mode: 'transform' }) + return await import(`data:text/javascript;base64,${Buffer.from(js).toString('base64')}`) +} +function measure(fn, paths, query) { + for (let warmup = 0; warmup < 10; warmup++) { + fn(paths, query, 16) + } + const samples = [] + for (let i = 0; i < 9; i++) { + const start = performance.now() + fn(paths, query, 16) + samples.push(performance.now() - start) + } + return samples.sort((a, b) => a - b)[4] +} +const results = [] +for (const [file, name] of [ + ['src/main/runtime/runtime-mobile-file-path-search.ts', 'rankRuntimeMobileFilePaths'], + ['mobile/src/session/mobile-native-chat-autocomplete.ts', 'rankSuggestions'] +]) { + const before = ( + await load(execFileSync('git', ['show', `${baseline}:${file}`], { encoding: 'utf8' })) + )[name] + const after = (await load(readFileSync(file, 'utf8')))[name] + for (const count of [100, 100000]) { + const paths = Array.from( + { length: count }, + (_, i) => `src/components/workspace/group-${i % 100}/file-${i}.tsx` + ) + for (const query of ['file-9', 'missing', 'workspace']) { + assert.deepEqual(after(paths, query, 16), before(paths, query, 16)) + results.push({ + function: name, + paths: count, + query, + beforeMs: measure(before, paths, query), + afterMs: measure(after, paths, query) + }) + } + } +} +console.log(JSON.stringify({ node: process.version, platform: process.platform, results }, null, 2)) diff --git a/config/scripts/mobile-markdown-placeholder-benchmark.mjs b/config/scripts/mobile-markdown-placeholder-benchmark.mjs new file mode 100644 index 00000000000..20280e5a8d2 --- /dev/null +++ b/config/scripts/mobile-markdown-placeholder-benchmark.mjs @@ -0,0 +1,58 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { readFileSync } from 'node:fs' +import { dirname, resolve } from 'node:path' +import { performance } from 'node:perf_hooks' +import { build } from 'esbuild' + +const sourcePath = 'mobile/src/components/mobile-markdown-preview-html.ts' +const baselineRef = process.argv[2] +if (!baselineRef) { + throw new Error( + 'Usage: node config/scripts/mobile-markdown-placeholder-benchmark.mjs ' + ) +} +async function load(source) { + const result = await build({ + stdin: { contents: source, resolveDir: dirname(resolve(sourcePath)), loader: 'ts' }, + bundle: true, + write: false, + platform: 'node', + format: 'esm' + }) + return ( + await import( + `data:text/javascript;base64,${Buffer.from(result.outputFiles[0].text).toString('base64')}` + ) + ).normalizeMobileMarkdownPreviewHtml +} +const before = await load( + execFileSync('git', ['show', `${baselineRef}:${sourcePath}`], { encoding: 'utf8' }) +) +const after = await load(readFileSync(sourcePath, 'utf8')) +function measure(fn, input, repeats) { + const samples = [] + for (let run = 0; run < repeats; run++) { + const start = performance.now() + fn(input) + samples.push(performance.now() - start) + } + return samples.sort((a, b) => a - b)[Math.floor(samples.length / 2)] +} +const results = [] +for (const [shape, input] of [ + ['ordinary Markdown', '# Hello\n\n

Use `Array` and bold.

'], + ...[2048, 8192, 16384].map((length) => [ + `${length} underscore collision`, + `\uE000ORCA_MD_CODE_${'_'.repeat(length)}0\uE000 and \`Array\`` + ]) +]) { + assert.equal(after(input), before(input)) + results.push({ + shape, + bytes: Buffer.byteLength(input), + beforeMs: measure(before, input, 5), + afterMs: measure(after, input, 15) + }) +} +console.log(JSON.stringify({ node: process.version, platform: process.platform, results }, null, 2)) diff --git a/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs b/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs new file mode 100644 index 00000000000..ba3e64bfa2f --- /dev/null +++ b/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs @@ -0,0 +1,223 @@ +// The relay's copy of the ConPTY teardown release, and the guard that keeps it in lockstep with +// `config/patches/node-pty@1.1.0.patch`. pnpm patches do not cross the SSH boundary, so a relay runs +// the tree `npm install` put there, and every terminal on a Windows SSH host leaked one File handle +// for the life of the relay process. +// +// The ORDER of the conin release is the fix. Releasing it at the top of the branch -- the placement +// the desktop patch uses -- was measured at 3x WORSE than shipping nothing (File +2/terminal and a +// new Process +1/terminal); releasing it after the console-list fork and the native kill is flat. +// +// Those numbers are the `!useConptyDll` branch, which is the branch a RELAY runs. Every desktop +// site that opens a terminal pane sets `useConptyDll: true` and takes the other branch, where +// upstream already destroys the input socket. Two hidden rate-limit probes +// (`src/main/rate-limits/claude-pty.ts`, `codex-pty-rate-limit-probe.ts`) do omit the option and so +// do run this hunk, but no user-visible pane does. The divergence pinned below is about which +// branch each host runs for terminals -- not about a regression in the panes users open. +import { createRequire } from 'node:module' +import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { join, resolve } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' + +const require = createRequire(import.meta.url) +const { + assertPatchedNodePtyWindowsTeardown, + patchNodePtyWindowsTeardown +} = require('../relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs') +const projectDir = resolve(import.meta.dirname, '..', '..') +const cleanupDirs = [] + +const PATCHED_FILES = ['windowsPtyAgent.js', 'windowsTerminal.js'] + +/** The hunks config/patches/node-pty@1.1.0.patch adds to the installed desktop tree. */ +const DESKTOP_HUNKS = { + 'windowsPtyAgent.js': [ + [ + [ + ' this._inSocket.readable = false;', + ' // The non-DLL path previously only flipped `readable`, leaving the', + ' // conin PipeWrap alive until the host exited (#947).', + ' this._inSocket.destroy();', + ' this._outSocket.readable = false;', + '' + ].join('\n'), + [ + ' this._inSocket.readable = false;', + ' this._outSocket.readable = false;', + '' + ].join('\n') + ], + // The useConptyDll branch, which only the DESKTOP runs -- the relay takes the + // non-DLL branch above, where the dispose is already unconditional. Listed here + // so un-applying still yields published; the relay asset needs no counterpart. + [ + [ + ' // Orca: dispose unconditionally, as the non-DLL branch above does.', + " // Waiting for another 'data' event leaks the conout worker on every", + ' // self-exiting shell, because no more data ever arrives (F24).', + ' this._conoutSocketWorker.dispose();', + '' + ].join('\n'), + [ + " this._outSocket.on('data', function () {", + ' _this._conoutSocketWorker.dispose();', + ' });', + '' + ].join('\n') + ] + ], + 'windowsTerminal.js': [ + [ + ' // Attach before readiness so a broken ConPTY output pipe cannot be unhandled.', + null + ], + [' // A ConPTY input-pipe error must retire only this terminal.', null] + ] +} + +function desktopPath(file) { + return join(projectDir, 'node_modules', 'node-pty', 'lib', file) +} + +afterEach(() => { + for (const dir of cleanupDirs.splice(0)) { + rmSync(dir, { recursive: true, force: true }) + } +}) + +describe('Windows SSH relay node-pty ConPTY teardown patch', () => { + // Why reconstruct rather than vendor upstream: the installed tree IS the published file plus the + // desktop's hunks, so un-applying them yields upstream exactly -- and pinning that against this + // asset's own hashes is what fails loudly if either side of the pair moves. + it('takes the desktop error listeners verbatim', () => { + const fixture = writeNodePtyFixture('1.1.0') + patchNodePtyWindowsTeardown(fixture.root) + + expect(readFileSync(join(fixture.libDir, 'windowsTerminal.js'), 'utf8')).toBe( + readFileSync(desktopPath('windowsTerminal.js'), 'utf8') + ) + }) + + // The one hunk that must NOT match the desktop patch, and the reason is measured, not stylistic: + // on the branch a relay runs, releasing conin before `_getConsoleProcessList()` forks aborts + // teardown partway. Desktop terminal panes take the other branch, so no pane is affected either + // way; what this guards is a patch sync putting the early placement onto the relay's branch. + it('releases conin after the console-list fork, unlike the desktop patch placement', () => { + const fixture = writeNodePtyFixture('1.1.0') + patchNodePtyWindowsTeardown(fixture.root) + const patched = readFileSync(join(fixture.libDir, 'windowsPtyAgent.js'), 'utf8') + + const branch = patched.slice( + patched.indexOf('if (!this._useConptyDll) {'), + patched.indexOf('else {', patched.indexOf('if (!this._useConptyDll) {')) + ) + expect(branch).toContain('this._inSocket.destroy();') + expect(branch.indexOf('this._inSocket.destroy();')).toBeGreaterThan( + branch.indexOf('this._conoutSocketWorker.dispose();') + ) + expect(branch.indexOf('this._inSocket.destroy();')).toBeGreaterThan( + branch.indexOf('this._getConsoleProcessList()') + ) + // Pinned so a future "sync the relay asset to config/patches" cannot copy the early placement + // onto the relay's branch, where it costs +2 File and +1 Process per terminal. + expect(patched).not.toBe(readFileSync(desktopPath('windowsPtyAgent.js'), 'utf8')) + }) + + it('installs and verifies idempotently', () => { + const fixture = writeNodePtyFixture('1.1.0') + + patchNodePtyWindowsTeardown(fixture.root) + const once = PATCHED_FILES.map((file) => readFileSync(join(fixture.libDir, file), 'utf8')) + for (const file of PATCHED_FILES) { + expect(existsSync(`${join(fixture.libDir, file)}.orca-patch-${process.pid}`)).toBe(false) + } + expect(() => assertPatchedNodePtyWindowsTeardown(fixture.root)).not.toThrow() + + patchNodePtyWindowsTeardown(fixture.root) + expect(PATCHED_FILES.map((file) => readFileSync(join(fixture.libDir, file), 'utf8'))).toEqual( + once + ) + }) + + it('refuses a different package version or unexpected source', () => { + const wrongVersion = writeNodePtyFixture('1.2.0-beta.11') + expect(() => patchNodePtyWindowsTeardown(wrongVersion.root)).toThrow('expected 1.1.0') + + for (const file of PATCHED_FILES) { + const drifted = writeNodePtyFixture('1.1.0') + const path = join(drifted.libDir, file) + writeFileSync(path, `${readFileSync(path, 'utf8')}\n// drift`) + expect(() => patchNodePtyWindowsTeardown(drifted.root)).toThrow('unexpected node-pty') + } + }) + + it('refuses a half-applied tree, so one file cannot pass for both', () => { + for (const file of PATCHED_FILES) { + const partial = writeNodePtyFixture('1.1.0') + const fixture = writeNodePtyFixture('1.1.0') + patchNodePtyWindowsTeardown(fixture.root) + writeFileSync(join(partial.libDir, file), readFileSync(join(fixture.libDir, file), 'utf8')) + expect(() => assertPatchedNodePtyWindowsTeardown(partial.root)).toThrow('is not installed') + } + }) +}) + +/** A published node-pty tree, rebuilt by un-applying the desktop hunks from the installed one. */ +function writeNodePtyFixture(version) { + const root = mkdtempSync(join(projectDir, '.node-pty-teardown-patch-test-')) + cleanupDirs.push(root) + const libDir = join(root, 'node_modules', 'node-pty', 'lib') + mkdirSync(libDir, { recursive: true }) + writeFileSync(join(root, 'node_modules', 'node-pty', 'package.json'), JSON.stringify({ version })) + for (const file of PATCHED_FILES) { + const desktop = readFileSync(desktopPath(file), 'utf8') + for (const [marker] of DESKTOP_HUNKS[file]) { + expect(desktop).toContain(marker) + } + writeFileSync(join(libDir, file), unapplyDesktopHunks(file, desktop)) + } + return { root, libDir } +} + +/** + * Reverse of the published-to-desktop transform. + * + * `windowsTerminal.js` is taken verbatim from the desktop, so the asset's own replacement table is + * the transform and reversing it is exact. `windowsPtyAgent.js` deliberately diverges, so its + * published form is rebuilt from the desktop hunk instead -- which is also what makes this file the + * place that notices if the desktop hunk itself ever moves. + */ +function unapplyDesktopHunks(file, desktop) { + if (file === 'windowsPtyAgent.js') { + let published = desktop + for (const [patched, original] of DESKTOP_HUNKS[file]) { + expect(published.split(patched).length - 1).toBe(1) + published = published.replace(patched, original) + } + return published + } + const asset = readFileSync( + join(projectDir, 'config', 'relay-assets', 'node-pty-1.1.0-windows-pty-teardown-patch.cjs'), + 'utf8' + ) + const { PATCH_TARGETS } = loadPatchTargets(asset) + const target = PATCH_TARGETS.find((entry) => entry.relativePath.at(-1) === file) + expect(target).toBeDefined() + let published = desktop + for (const [from, to] of target.replacements.toReversed()) { + expect(published.split(to).length - 1).toBe(1) + published = published.replace(to, from) + } + return published +} + +function loadPatchTargets(assetSource) { + const module = { exports: {} } + const factory = new Function( + 'module', + 'exports', + 'require', + `${assetSource}\nmodule.exports.PATCH_TARGETS = PATCH_TARGETS` + ) + factory(module, module.exports, require) + return module.exports +} diff --git a/config/scripts/patched-dependencies-frozen-install.test.mjs b/config/scripts/patched-dependencies-frozen-install.test.mjs new file mode 100644 index 00000000000..89f98ef074b --- /dev/null +++ b/config/scripts/patched-dependencies-frozen-install.test.mjs @@ -0,0 +1,155 @@ +import { + cpSync, + copyFileSync, + existsSync, + mkdirSync, + mkdtempSync, + readFileSync, + writeFileSync +} from 'node:fs' +import { tmpdir } from 'node:os' +import { isAbsolute, join, parse, resolve } from 'node:path' +import { describe, expect, it } from 'vitest' +import { runProcessSync } from '../../src/shared/child-process/run-process.ts' +import { resolveCliCommand } from '../../src/shared/node-cli-command-resolution.ts' +import { removeTreeSync } from '../../src/shared/windows-transient-lock-removal.ts' +import { resolvePnpmCliInvocation } from './pnpm-cli-invocation.mjs' + +/** + * Run the command that actually consumes the patch hashes. + * + * A hash comparison is not this check. `@vscode/windows-process-tree@0.8.0` shipped + * twice with a hand-computed `sha256(patchBytes)` in the lockfile, and two separate + * reviews "verified" it by recomputing the same number the same wrong way. pnpm + * hashes the **LF-normalized** content, so a CRLF patch makes the raw digest a value + * pnpm will never produce, and `--frozen-lockfile` dies with + * ERR_PNPM_LOCKFILE_CONFIG_MISMATCH on every runner. An independent check that + * repeats the original assumption is not independent; only the installer is. + * + * `--lockfile-only --ignore-scripts` keeps it to the resolution pnpm rejects on, + * with no node_modules and no native builds. + */ +const PROJECT_DIR = resolve(import.meta.dirname, '../..') +const WINDOWS_PROCESS_TREE_PATCH = '@vscode__windows-process-tree@0.8.0.patch' + +/** + * Which pnpm to run belongs to pnpm-cli-invocation.mjs, not to this file: naming + * the Windows shim here is what windows-cmd-shim-spawn-boundary.test.mjs rejects. + * Its `shell` is dropped on purpose -- runProcessSync refuses that flag and + * already drives a shim through the interpreter itself. + */ +function resolvePnpmInvocation() { + const { command, prefixArgs } = resolvePnpmCliInvocation() + if (isAbsolute(command)) { + return existsSync(command) ? { program: command, prefixArgs } : null + } + // Bare name only when npm_execpath is unset (bare `vitest`, not `pnpm test`). + // Drop the extension so the shared resolver tries every executable form of it. + const resolved = resolveCliCommand(parse(command).name) + return isAbsolute(resolved) ? { program: resolved, prefixArgs } : null +} + +describe('patched dependencies', () => { + it('installs with --frozen-lockfile, which is what validates every patch hash', () => { + const pnpm = resolvePnpmInvocation() + expect(pnpm, 'pnpm must be installed; it is the only thing that can check this').not.toBeNull() + + // A copy, because a --frozen-lockfile run still rewrites parts of the + // lockfile this repo does not track, and the real one must not move. + const scratch = mkdtempSync(join(tmpdir(), 'orca-frozen-install-')) + try { + for (const file of ['package.json', 'pnpm-lock.yaml', 'pnpm-workspace.yaml']) { + copyFileSync(join(PROJECT_DIR, file), join(scratch, file)) + } + mkdirSync(join(scratch, 'config'), { recursive: true }) + cpSync(join(PROJECT_DIR, 'config', 'patches'), join(scratch, 'config', 'patches'), { + recursive: true + }) + + const result = runProcessSync({ + program: pnpm.program, + args: [ + ...pnpm.prefixArgs, + 'install', + '--frozen-lockfile', + '--lockfile-only', + '--ignore-scripts' + ], + cwd: scratch, + timeoutMs: 300_000 + }) + + expect(result.code, `${result.stdout}\n${result.stderr}`).toBe(0) + } finally { + removeTreeSync(scratch) + } + // The 300s spawn budget is only reachable if the case is allowed to take it; + // config/vitest.config.ts caps every case at 30s by default. + }, 300_000) + + /** + * `--lockfile-only` resolves; it never applies a patch. So the case above is + * bounded to hash consistency, and the actual question -- can pnpm still put + * the patched reader on disk? -- had nothing covering it. + * + * One package, patch applied for real, assert the marker landed. Scoped to the + * single dependency so it stays a ~2s check rather than a full install. + */ + it('materializes the patched command-line reader on a real install', () => { + const pnpm = resolvePnpmInvocation() + expect(pnpm, 'pnpm must be installed; it is the only thing that can check this').not.toBeNull() + + const scratch = mkdtempSync(join(tmpdir(), 'orca-patch-apply-')) + try { + mkdirSync(join(scratch, 'config', 'patches'), { recursive: true }) + copyFileSync( + join(PROJECT_DIR, 'config', 'patches', WINDOWS_PROCESS_TREE_PATCH), + join(scratch, 'config', 'patches', WINDOWS_PROCESS_TREE_PATCH) + ) + writeFileSync( + join(scratch, 'package.json'), + `${JSON.stringify( + { + name: 'orca-patch-apply-probe', + version: '1.0.0', + dependencies: { '@vscode/windows-process-tree': '0.8.0' } + }, + null, + 2 + )}\n` + ) + writeFileSync( + join(scratch, 'pnpm-workspace.yaml'), + 'packages: []\n' + + 'patchedDependencies:\n' + + ` '@vscode/windows-process-tree@0.8.0': config/patches/${WINDOWS_PROCESS_TREE_PATCH}\n` + ) + + const result = runProcessSync({ + program: pnpm.program, + args: [...pnpm.prefixArgs, 'install', '--no-frozen-lockfile', '--ignore-scripts'], + cwd: scratch, + timeoutMs: 300_000 + }) + expect(result.code, `${result.stdout}\n${result.stderr}`).toBe(0) + + const materialized = readFileSync( + join( + scratch, + 'node_modules', + '@vscode', + 'windows-process-tree', + 'src', + 'process_commandline.cc' + ), + 'utf8' + ) + expect(materialized).toContain('kProcessCommandLineInformation') + // The whole point of the patch: the upstream reader is gone, not merely + // supplemented. + expect(materialized).not.toContain('ReadProcessMemory') + } finally { + removeTreeSync(scratch) + } + }, 300_000) +}) diff --git a/config/scripts/pr-code-change-scope.mjs b/config/scripts/pr-code-change-scope.mjs index f5a73f6239f..3089d376b2a 100644 --- a/config/scripts/pr-code-change-scope.mjs +++ b/config/scripts/pr-code-change-scope.mjs @@ -213,12 +213,18 @@ const LINUX_PACKAGE_TESTS = [ const WINDOWS_PACKAGE_TESTS = [ ...LINUX_PACKAGE_TESTS, 'config/scripts/rebuild-native-deps.test.mjs', + 'config/scripts/rebuild-native-deps-windows-process-tree.test.mjs', 'src/main/providers/windows-conpty-wide-char-duplication.node-pty.test.ts', 'src/main/providers/pty-repaint-wide-char-buffer.node-pty.test.ts', 'src/shared/child-process/windows-command-line.win32.test.ts', + 'src/shared/child-process/windows-cmd-shim-resolution.test.ts', + 'src/shared/child-process/windows-cmd-shim-resolution.win32.test.ts', 'src/main/agent-hooks/windows-hook-payload-delivery.test.ts', + 'src/main/agent-hooks/windows-direct-cmd-hook-command.test.ts', 'src/main/windows/windows-pty-job.win32.test.ts', 'src/main/windows/windows-host-job.win32.test.ts', + 'src/main/windows/windows-process-tree-command-line-patch.test.ts', + 'src/main/windows-live-tree-kill.win32.test.ts', 'src/main/wsl/wsl-runner.test.ts', 'src/main/wsl/wsl-guest-environment.test.ts', 'src/main/wsl/wsl-invocation-boundary.test.ts', @@ -226,14 +232,18 @@ const WINDOWS_PACKAGE_TESTS = [ 'src/main/wsl/wsl-w1-w3-contract.test.ts', 'src/shared/source-scan/source-tree-scan.test.ts', 'src/main/cli/wsl-cli-powershell-boundary.test.ts', + 'src/main/computer/desktop-script-runtime-host.win32.test.ts', 'src/main/cursor/hook-service.test.ts', 'src/main/orca-profiles/profile-index-store.test.ts', 'src/main/startup/windows-install-dir-acl-repair.win32.test.ts', 'src/main/runtime/repo-worktree-admin-fingerprint.test.ts', 'src/main/runtime/worktree-scan-admin-fingerprint-gate.test.ts', 'src/shared/secure-file-fsync-flags.test.ts', + 'src/shared/secure-path-windows-acl.win32.test.ts', + 'src/main/runtime/unreadable-secret-store-preservation.win32.test.ts', 'src/main/ipc/pty-codex-account-attribution.test.ts', - 'src/main/ipc/pty-spawn-env-codex-resume-provenance.test.ts' + 'src/main/ipc/pty-spawn-env-codex-resume-provenance.test.ts', + 'src/relay/windows-port-scan.win32.test.ts' ] const DESKTOP_IRRELEVANT_PREFIXES = [ diff --git a/config/scripts/pr-code-change-scope.test.mjs b/config/scripts/pr-code-change-scope.test.mjs index 4642372135c..f31822e5b93 100644 --- a/config/scripts/pr-code-change-scope.test.mjs +++ b/config/scripts/pr-code-change-scope.test.mjs @@ -414,10 +414,11 @@ describe('PR Checks skip wiring', () => { }) it('skips e2e detection on docs-only PRs without dropping the draft gate', () => { - expect(prWorkflow.jobs['e2e-paths'].needs).toEqual(['code_paths']) - expect(prWorkflow.jobs['e2e-paths'].if).toBe( - "github.event.pull_request.draft != true && needs.code_paths.outputs.should_run == 'true'" + const filter = prWorkflow.jobs.code_paths.steps.find((step) => step.id === 'e2e_filter') + expect(filter.if).toBe( + "github.event.pull_request.draft != true && steps.filter.outputs.should_run == 'true'" ) + expect(prWorkflow.jobs['e2e-paths']).toBeUndefined() }) it('lets verify pass skipped jobs the classifier turned off', () => { diff --git a/config/scripts/pr-e2e-gate-contract.test.mjs b/config/scripts/pr-e2e-gate-contract.test.mjs index ceac6b8cc6e..41f9338ab75 100644 --- a/config/scripts/pr-e2e-gate-contract.test.mjs +++ b/config/scripts/pr-e2e-gate-contract.test.mjs @@ -39,7 +39,7 @@ const nativeImeSpec = readFileSync( 'utf8' ) -const filterStep = prWorkflow.jobs['e2e-paths'].steps.find( +const filterStep = prWorkflow.jobs.code_paths.steps.find( (step) => step.name === 'Filter changed E2E specs' ) const rollbackStep = prWorkflow.jobs.static_analysis.steps.find( @@ -106,16 +106,16 @@ describe('PR E2E gate contract', () => { // Why: without this the job could lose its filter and run on every PR — the // cost the path filter exists to avoid — while the gate assertions above // stay green. - expect(prWorkflow.jobs.e2e.needs).toBe('e2e-paths') - expect(prWorkflow.jobs.e2e.if).toBe("needs.e2e-paths.outputs.should_run == 'true'") - expect(prWorkflow.jobs['e2e-paths'].outputs.should_run).toBe( - '${{ steps.filter.outputs.should_run }}' + expect(prWorkflow.jobs.e2e.needs).toBe('code_paths') + expect(prWorkflow.jobs.e2e.if).toBe("needs.code_paths.outputs.e2e_should_run == 'true'") + expect(prWorkflow.jobs.code_paths.outputs.e2e_should_run).toBe( + '${{ steps.e2e_filter.outputs.should_run }}' ) - expect(prWorkflow.jobs['e2e-paths'].outputs.test_files).toBe( - '${{ steps.filter.outputs.test_files }}' + expect(prWorkflow.jobs.code_paths.outputs.test_files).toBe( + '${{ steps.e2e_filter.outputs.test_files }}' ) expect(prWorkflow.jobs.e2e.with.ref).toBe('${{ github.event.pull_request.head.sha }}') - expect(prWorkflow.jobs.e2e.with.test_files).toBe('${{ needs.e2e-paths.outputs.test_files }}') + expect(prWorkflow.jobs.e2e.with.test_files).toBe('${{ needs.code_paths.outputs.test_files }}') }) it('enforces every job verify depends on', () => { @@ -360,11 +360,11 @@ describe('PR E2E gate contract', () => { expect(sshLaneCondition).toContain("inputs.ssh_source_changed == 'true' ||") expect(e2eWorkflow.on.workflow_call.inputs.ssh_source_changed.type).toBe('string') - expect(prWorkflow.jobs['e2e-paths'].outputs.ssh_source_changed).toBe( - '${{ steps.filter.outputs.ssh_source_changed }}' + expect(prWorkflow.jobs.code_paths.outputs.ssh_source_changed).toBe( + '${{ steps.e2e_filter.outputs.ssh_source_changed }}' ) expect(prWorkflow.jobs.e2e.with.ssh_source_changed).toBe( - '${{ needs.e2e-paths.outputs.ssh_source_changed }}' + '${{ needs.code_paths.outputs.ssh_source_changed }}' ) expect(filterStep.run).toContain('pr-e2e-source-routing.mjs --ssh-source') expect(filterStep.run).toContain('ssh_source_changed=$SSH_SOURCE_CHANGED') @@ -565,12 +565,12 @@ describe('PR E2E gate contract', () => { expect(prWorkflow.jobs.terminal_ime_native.uses).toBe( './.github/workflows/terminal-ime-e2e.yml' ) - expect(prWorkflow.jobs.terminal_ime_native.needs).toBe('e2e-paths') + expect(prWorkflow.jobs.terminal_ime_native.needs).toBe('code_paths') expect(prWorkflow.jobs.terminal_ime_native.if).toBe( - "needs.e2e-paths.outputs.native_ime_source_changed == 'true'" + "needs.code_paths.outputs.native_ime_source_changed == 'true'" ) - expect(prWorkflow.jobs['e2e-paths'].outputs.native_ime_source_changed).toBe( - '${{ steps.filter.outputs.native_ime_source_changed }}' + expect(prWorkflow.jobs.code_paths.outputs.native_ime_source_changed).toBe( + '${{ steps.e2e_filter.outputs.native_ime_source_changed }}' ) expect(filterStep.run).toContain('pr-e2e-source-routing.mjs --native-ime-source') expect(filterStep.run).toContain('native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED') @@ -636,13 +636,8 @@ describe('PR E2E gate contract', () => { .filter((spec) => nativeGateExpression.test(readFileSync(join(projectDir, spec), 'utf8'))) expect(nativeGatedSpecs.length).toBeGreaterThan(0) - // Why exempt: the digit repro needs a nested gnome-shell, which no hosted runner provides - // (headless mutter never answers RemoteDesktop.CreateSession); the macOS spec needs a real - // macOS input source, and no macOS runner exists on any PR or scheduled lane. - const unreachableSpecs = new Set([ - 'tests/e2e/terminal-hangul-terminating-digit-native.spec.ts', - 'tests/e2e/terminal-macos-2set-korean-native.spec.ts' - ]) + // The macOS spec needs a native input source; PR and scheduled IME lanes use Linux. + const unreachableSpecs = new Set(['tests/e2e/terminal-macos-2set-korean-native.spec.ts']) const unclaimed = nativeGatedSpecs.filter( (spec) => !unreachableSpecs.has(spec) && !nativeImeRunner.includes(spec) ) @@ -682,8 +677,13 @@ describe('PR E2E gate contract', () => { // Why pin the titles: the runner requires one receipt per name, so a rename that nobody // mirrored here would fail the lane loudly instead of quietly halving it. + const nativeDigitSpec = readFileSync( + join(projectDir, 'tests/e2e/terminal-hangul-terminating-digit-native.spec.ts'), + 'utf8' + ) + expect(nativeDigitSpec).toContain('appendImeEngagementReceipt(testInfo.title, trace)') for (const title of EXPECTED_NATIVE_IME_TESTS) { - expect(nativeImeSpec, title).toContain(title) + expect(nativeImeSpec + nativeDigitSpec, title).toContain(title) } }) diff --git a/config/scripts/pr-e2e-native-only-routing.test.mjs b/config/scripts/pr-e2e-native-only-routing.test.mjs new file mode 100644 index 00000000000..b6c3662cd1d --- /dev/null +++ b/config/scripts/pr-e2e-native-only-routing.test.mjs @@ -0,0 +1,33 @@ +import { readFileSync } from 'node:fs' +import { describe, expect, it } from 'vitest' +import { parse } from 'yaml' +import { hasNativeImeSourceChange, shouldRunReusablePrE2e } from './pr-e2e-source-routing.mjs' + +const workflow = parse(readFileSync('.github/workflows/pr.yml', 'utf8')) +const filterStep = workflow.jobs.code_paths.steps.find((step) => step.id === 'e2e_filter') + +describe('native-only PR E2E routing', () => { + it('avoids generic E2E allocation for native-only changes while preserving its IME lane', () => { + for (const file of [ + 'tests/e2e/terminal-ibus-hangul-native.spec.ts', + 'config/scripts/run-terminal-ibus-hangul-e2e.mjs' + ]) { + expect(hasNativeImeSourceChange([file])).toBe(true) + expect(shouldRunReusablePrE2e([file])).toBe(false) + } + expect(shouldRunReusablePrE2e([])).toBe(false) + for (const spec of [ + 'tests/e2e/ssh-startup-exec-readiness.spec.ts', + 'tests/e2e/paired-startup-exec-readiness.spec.ts', + 'tests/e2e/terminal-ime-exact-byte.spec.ts', + 'tests/e2e/future.spec.ts' + ]) { + expect(shouldRunReusablePrE2e([spec])).toBe(true) + expect(shouldRunReusablePrE2e(['tests/e2e/terminal-ibus-hangul-native.spec.ts', spec])).toBe( + true + ) + } + expect(filterStep.run).toContain('pr-e2e-source-routing.mjs --reusable-workflow') + expect(filterStep.run).toContain('if [ "$SHOULD_RUN" = true ]; then') + }) +}) diff --git a/config/scripts/pr-e2e-source-routing.mjs b/config/scripts/pr-e2e-source-routing.mjs index 78814b663cb..5b698fb0b42 100644 --- a/config/scripts/pr-e2e-source-routing.mjs +++ b/config/scripts/pr-e2e-source-routing.mjs @@ -217,6 +217,16 @@ export function hasNativeImeSourceChange(changedPaths) { ).some((route) => changedPaths.some(route.matches)) } +export function shouldRunReusablePrE2e(changedPaths) { + // Native IME has its own workflow; SSH still runs inside the reusable workflow. + return ( + hasSshSourceChange(changedPaths) || + selectPrE2eSpecs(changedPaths).some( + (spec) => spec !== 'tests/e2e/terminal-ibus-hangul-native.spec.ts' + ) + ) +} + if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { let input = '' process.stdin.setEncoding('utf8') @@ -226,6 +236,8 @@ if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) const changedPaths = input.split(/\r?\n/).filter(Boolean) if (process.argv.includes('--ssh-source')) { process.stdout.write(`${hasSshSourceChange(changedPaths)}\n`) + } else if (process.argv.includes('--reusable-workflow')) { + process.stdout.write(`${shouldRunReusablePrE2e(changedPaths)}\n`) } else if (process.argv.includes('--native-ime-source')) { process.stdout.write(`${hasNativeImeSourceChange(changedPaths)}\n`) } else { diff --git a/config/scripts/quick-open-exclusion-benchmark.mjs b/config/scripts/quick-open-exclusion-benchmark.mjs new file mode 100644 index 00000000000..399302c5a2b --- /dev/null +++ b/config/scripts/quick-open-exclusion-benchmark.mjs @@ -0,0 +1,60 @@ +import assert from 'node:assert/strict' +import { performance } from 'node:perf_hooks' +import { build } from 'esbuild' + +const bundled = await build({ + entryPoints: ['src/shared/quick-open-filter.ts'], + bundle: true, + platform: 'node', + format: 'esm', + write: false, + logLevel: 'silent' +}) +const { shouldExcludeQuickOpenRelPath: after } = await import( + `data:text/javascript;base64,${Buffer.from(bundled.outputFiles[0].text).toString('base64')}` +) +// Original production predicate, including its exact boundary check. +function before(relPath, prefixes) { + for (const prefix of prefixes) { + if (relPath === prefix) { + return true + } + if (relPath.length > prefix.length && relPath.startsWith(`${prefix}/`)) { + return true + } + } + return false +} +const files = Array.from( + { length: 100000 }, + (_, index) => `src/components/group-${index % 100}/file-${index}.tsx` +) +function run(fn, prefixes) { + let excluded = 0 + for (const file of files) { + excluded += Number(fn(file, prefixes)) + } + return excluded +} +function measure(fn, prefixes) { + run(fn, prefixes) + const samples = [] + for (let index = 0; index < 5; index++) { + const start = performance.now() + run(fn, prefixes) + samples.push(performance.now() - start) + } + return samples.sort((a, b) => a - b)[2] +} +const results = [] +for (const count of [0, 10, 100, 500]) { + const prefixes = Array.from({ length: count }, (_, index) => `nested-worktrees/worktree-${index}`) + assert.equal(run(after, prefixes), run(before, prefixes)) + results.push({ + files: files.length, + exclusions: count, + beforeMs: measure(before, prefixes), + afterMs: measure(after, prefixes) + }) +} +console.log(JSON.stringify({ node: process.version, platform: process.platform, results }, null, 2)) diff --git a/config/scripts/rebuild-native-deps-node-pty.test.mjs b/config/scripts/rebuild-native-deps-node-pty.test.mjs index 09e38853371..871732dd53d 100644 --- a/config/scripts/rebuild-native-deps-node-pty.test.mjs +++ b/config/scripts/rebuild-native-deps-node-pty.test.mjs @@ -4,6 +4,8 @@ import { join } from 'node:path' import { describe, expect, it } from 'vitest' import { + gitLineEndingEnv, + initGitWorkTree, mkTempProject, runRebuildScript, writeFakeElectronRebuild, @@ -14,7 +16,8 @@ import { writeFakeWindowsProcessTreeWithNodeAddonApi, writeFakeWindowsRegistry, writeNodePtyPatchFile, - writePatchedNodePtyBuildArtifacts + writePatchedNodePtyBuildArtifacts, + writeWindowsProcessTreePatchFile } from './rebuild-native-deps-test-fixtures.mjs' describe('rebuild-native-deps patched node-pty rebuild', () => { @@ -85,6 +88,91 @@ describe('rebuild-native-deps patched node-pty rebuild', () => { } }) + const commandLineSourcePath = (projectDir) => + join( + projectDir, + 'node_modules', + '@vscode', + 'windows-process-tree', + 'src', + 'process_commandline.cc' + ) + + // Why inside a git work tree: `git apply` run under one prefixes patch paths + // with the cwd-relative prefix, silently skips what does not match, and still + // exits 0. The package dir is always under the project root in production, so + // a fixture in %TEMP% alone would pass while the real repair did nothing. + // + // Why both line-ending modes: the patch is stored LF while upstream ships this + // source CRLF, so whether the pre-image matches depends on `core.autocrlf` -- + // and under `false`, Git's own built-in default, it did not. The repair blinds + // git to the repo, so that value comes from global config, i.e. from whichever + // option the developer's installer wrote. Pinning both makes the case cover the + // host that breaks rather than the host that happens to run it. + for (const autocrlf of ['false', 'true']) { + it(`repairs an un-applied command-line patch in a work tree (autocrlf=${autocrlf})`, () => { + const projectDir = mkTempProject() + + try { + initGitWorkTree(projectDir) + writeFakeUsableElectronPackage(projectDir, { platform: 'win32' }) + writeFakeElectronRebuild(projectDir) + writeFakeNodePtyConptyPayload(projectDir, 'x64') + writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir, { + commandLinePatchApplied: false + }) + writeWindowsProcessTreePatchFile(projectDir) + + const result = runRebuildScript( + projectDir, + { + npm_config_platform: 'win32', + npm_config_arch: 'x64', + ...gitLineEndingEnv(autocrlf) + }, + ['--platform=win32', '--arch=x64', '--force'] + ) + + expect(result.status, result.stderr).toBe(0) + expect(readFileSync(commandLineSourcePath(projectDir), 'utf8')).toContain( + 'kProcessCommandLineInformation' + ) + } finally { + removeTreeSync(projectDir) + } + }) + } + + // Why fail rather than build: an unpatched command-line reader compiles fine + // and then opens every process with PROCESS_VM_READ to walk its PEB, which is + // the primitive the patch exists to remove. + it('refuses a Windows rebuild when the command-line patch cannot be applied', () => { + const projectDir = mkTempProject() + + try { + initGitWorkTree(projectDir) + writeFakeUsableElectronPackage(projectDir, { platform: 'win32' }) + writeFakeElectronRebuild(projectDir) + writeFakeNodePtyConptyPayload(projectDir, 'x64') + writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir, { commandLinePatchApplied: false }) + // No patch file, so the repair has nothing to apply. + + const result = runRebuildScript( + projectDir, + { npm_config_platform: 'win32', npm_config_arch: 'x64' }, + ['--platform=win32', '--arch=x64', '--force'] + ) + + expect(result.status).not.toBe(0) + expect(result.stderr).toContain('process_commandline.cc') + expect(readFileSync(commandLineSourcePath(projectDir), 'utf8')).not.toContain( + 'kProcessCommandLineInformation' + ) + } finally { + removeTreeSync(projectDir) + } + }) + it('restores the ConPTY runtime payload after a Windows Electron rebuild', () => { const projectDir = mkTempProject() @@ -256,4 +344,37 @@ describe('rebuild-native-deps patched node-pty rebuild', () => { } } ) + + // The binary this step produces is the one copied into the packaged app. The + // relay build checks its own artifact and ensure-native-runtime checks what it + // loads; nothing checked this one, so a rebuild that quietly emitted the + // upstream reader shipped. Both non-clean states have to fail, which is the + // caller the tri-state was missing: after a rebuild that reported success, an + // absent binary is a broken build, not an absence to shrug at. + for (const [addon, expected] of [ + ['unpatched', 'still imports ReadProcessMemory'], + ['none', 'is not there'] + ]) { + it(`fails a Windows rebuild that leaves ${addon} windows-process-tree bytes`, () => { + const projectDir = mkTempProject() + + try { + writeFakeUsableElectronPackage(projectDir, { platform: 'win32' }) + writeFakeElectronRebuild(projectDir, { addon }) + writeFakeNodePtyConptyPayload(projectDir, 'x64') + writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir) + + const result = runRebuildScript( + projectDir, + { npm_config_platform: 'win32', npm_config_arch: 'x64' }, + ['--platform=win32', '--arch=x64', '--force'] + ) + + expect(result.status).not.toBe(0) + expect(result.stderr).toContain(expected) + } finally { + removeTreeSync(projectDir) + } + }) + } }) diff --git a/config/scripts/rebuild-native-deps-test-fixtures.mjs b/config/scripts/rebuild-native-deps-test-fixtures.mjs index 585e7a58ef2..2cb7d8ba8b4 100644 --- a/config/scripts/rebuild-native-deps-test-fixtures.mjs +++ b/config/scripts/rebuild-native-deps-test-fixtures.mjs @@ -1,5 +1,12 @@ import { spawnSync } from 'node:child_process' -import { chmodSync, copyFileSync, mkdirSync, mkdtempSync, writeFileSync } from 'node:fs' +import { + chmodSync, + copyFileSync, + mkdirSync, + mkdtempSync, + readFileSync, + writeFileSync +} from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' import { fileURLToPath } from 'node:url' @@ -15,6 +22,68 @@ const sourceNodePtyJobOwnershipPath = fileURLToPath( const sourceWindowsProcessTreeGypRebuildPath = fileURLToPath( new URL('./windows-process-tree-gyp-rebuild.mjs', import.meta.url) ) +const sourceWindowsProcessTreePatchPath = fileURLToPath( + new URL('../patches/@vscode__windows-process-tree@0.8.0.patch', import.meta.url) +) + +/** + * The command-line reader as it is *before* the patch, taken from the patch's + * own pre-image so no upstream copy has to be vendored. + * + * Written back as **CRLF**, which is what `@vscode/windows-process-tree@0.8.0` + * actually ships: all 67 pre-image lines of this file carried a CR before the + * patch was normalized to LF. Rebuilding it with the patch's current newline + * instead would make fixture and patch agree by construction, on any encoding — + * which is exactly how a repair that cannot apply to the real package passed + * this suite. + */ +function unpatchedWindowsProcessTreeCommandLineSource() { + const lines = readFileSync(sourceWindowsProcessTreePatchPath, 'utf8').split('\n') + const start = lines.findIndex((line) => + line.startsWith('diff --git a/src/process_commandline.cc ') + ) + const rest = lines.slice(start + 1) + const end = rest.findIndex((line) => line.startsWith('diff --git ')) + const preImage = (end === -1 ? rest : rest.slice(0, end)) + .filter((line) => line.startsWith(' ') || line.startsWith('-')) + .filter((line) => !line.startsWith('---')) + .map((line) => line.slice(1).replace(/\r$/, '')) + .join('\r\n') + // Splitting drops the file's own trailing newline as an empty element, and + // `git apply` needs the bytes exact. + return `${preImage}\r\n` +} + +/** + * Pin `core.autocrlf` for a spawned repair, whatever the host is set to. + * + * The repair blinds git to the surrounding repo with `GIT_DIR`, so the value it + * sees comes from global/system config — on a Git for Windows box that is + * whichever line-ending option the installer wrote, and `false` (Git's built-in + * default, "checkout as-is") is the one the repair used to fail under. A global + * config in a temp HOME outranks the system file, so this is deterministic + * rather than whatever the developer happens to have. + */ +export function gitLineEndingEnv(autocrlf) { + const home = mkdtempSync(join(tmpdir(), `orca-git-home-${autocrlf}-`)) + writeFileSync(join(home, '.gitconfig'), `[core]\n\tautocrlf = ${autocrlf}\n`) + return { HOME: home, USERPROFILE: home } +} + +/** Production always runs the repair from inside a work tree; `git apply` behaves differently there. */ +export function initGitWorkTree(projectDir) { + for (const args of [['init'], ['config', 'user.email', 'a@b.c'], ['config', 'user.name', 't']]) { + spawnSync('git', args, { cwd: projectDir, encoding: 'utf8' }) + } +} + +export function writeWindowsProcessTreePatchFile(projectDir) { + mkdirSync(join(projectDir, 'config', 'patches'), { recursive: true }) + copyFileSync( + sourceWindowsProcessTreePatchPath, + join(projectDir, 'config', 'patches', '@vscode__windows-process-tree@0.8.0.patch') + ) +} export function mkTempProject() { const projectDir = mkdtempSync(join(tmpdir(), 'orca-rebuild-native-deps-')) @@ -143,17 +212,46 @@ if (${JSON.stringify(createExecutable)}) { ) } -export function writeFakeElectronRebuild(projectDir, { logPathEnv = null } = {}) { +/** Bytes that stand in for a compiled addon's import table. */ +const FAKE_ADDON_BYTES = { + clean: 'MZ\0ntdll.dll\0NtQueryInformationProcess\0', + unpatched: 'MZ\0KERNEL32.dll\0ReadProcessMemory\0' +} + +/** + * A rebuild that produces nothing leaves no addon to inspect, and the script now + * asserts the binary it just built is a patched one. Emit a stand-in so the + * fixture models a rebuild that actually succeeded. `addon` picks which kind, + * because "produced the upstream reader" and "produced nothing" are both real + * outcomes that assertion has to tell apart. + */ +export function writeFakeElectronRebuild(projectDir, { logPathEnv = null, addon = 'clean' } = {}) { const rebuildDir = join(projectDir, 'node_modules', '@electron', 'rebuild') mkdirSync(rebuildDir, { recursive: true }) writeFileSync(join(rebuildDir, 'package.json'), JSON.stringify({ type: 'module' })) + const emitAddon = + addon === 'none' + ? '' + : ` + const packageDir = join('node_modules', '@vscode', 'windows-process-tree') + if (existsSync(join(packageDir, 'package.json'))) { + mkdirSync(join(packageDir, 'build', 'Release'), { recursive: true }) + writeFileSync( + join(packageDir, 'build', 'Release', 'windows_process_tree.node'), + ${JSON.stringify(FAKE_ADDON_BYTES[addon])} + ) + }` + const emitImports = + addon === 'none' + ? '' + : "import { existsSync, mkdirSync, writeFileSync } from 'node:fs'\nimport { join } from 'node:path'\n" writeFileSync( join(rebuildDir, 'index.js'), logPathEnv ? ` import { appendFileSync } from 'node:fs' - -export async function rebuild(options) { +${emitImports} +export async function rebuild(options) {${emitAddon} const logPath = process.env[${JSON.stringify(logPathEnv)}] if (!logPath) { return @@ -171,7 +269,10 @@ export async function rebuild(options) { ) } ` - : 'export async function rebuild() {}\n' + : `${emitImports} +export async function rebuild() {${emitAddon} +} +` ) } @@ -271,12 +372,22 @@ export function writeFakeWindowsProcessTree(projectDir) { writeFileSync(join(processTreeDir, 'index.js'), 'module.exports = {}\n') } -export function writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir) { +export function writeFakeWindowsProcessTreeWithNodeAddonApi( + projectDir, + { commandLinePatchApplied = true } = {} +) { const processTreeDir = join(projectDir, 'node_modules', '@vscode', 'windows-process-tree') const nodeAddonApiDir = join(processTreeDir, 'node_modules', 'node-addon-api') mkdirSync(nodeAddonApiDir, { recursive: true }) writeFileSync(join(processTreeDir, 'package.json'), '{"dependencies":{"node-addon-api":"*"}}\n') writeFileSync(join(processTreeDir, 'index.js'), 'module.exports = {}\n') + mkdirSync(join(processTreeDir, 'src'), { recursive: true }) + writeFileSync( + join(processTreeDir, 'src', 'process_commandline.cc'), + commandLinePatchApplied + ? '// kProcessCommandLineInformation = 60\n' + : unpatchedWindowsProcessTreeCommandLineSource() + ) writeFileSync(join(nodeAddonApiDir, 'package.json'), '{"name":"node-addon-api"}\n') writeFileSync(join(nodeAddonApiDir, 'napi.h'), '// napi.h\n') writeFileSync(join(nodeAddonApiDir, 'napi-inl.h'), '// napi-inl.h\n') diff --git a/config/scripts/rebuild-native-deps-windows-process-tree.test.mjs b/config/scripts/rebuild-native-deps-windows-process-tree.test.mjs new file mode 100644 index 00000000000..4f98c1b092d --- /dev/null +++ b/config/scripts/rebuild-native-deps-windows-process-tree.test.mjs @@ -0,0 +1,103 @@ +import { spawn } from 'node:child_process' +import { appendFileSync, copyFileSync, existsSync, mkdirSync } from 'node:fs' +import { createRequire } from 'node:module' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { removeTreeSync } from '../../src/shared/windows-transient-lock-removal.ts' + +import { + mkTempProject, + runRebuildScript, + writeFakeElectronRebuild, + writeFakeNodePtyConptyPayload, + writeFakeUsableElectronPackage, + writeFakeWindowsProcessTreeWithNodeAddonApi +} from './rebuild-native-deps-test-fixtures.mjs' + +const require = createRequire(import.meta.url) + +/** A real loadable addon, so the OS holds the same lock a running Orca holds. */ +function repoAddonPath() { + try { + const entry = require.resolve('@vscode/windows-process-tree') + const built = join(entry, '..', '..', 'build', 'Release', 'windows_process_tree.node') + return existsSync(built) ? built : null + } catch { + return null + } +} + +/** + * Stage a stale addon and keep it loaded, exactly as a running Orca does. + * + * The bytes are the repo's own patched build with the flagged import appended, + * because the guard keys on that symbol and the patched binary does not carry + * it. Trailing bytes are PE overlay, so the file still loads. + */ +async function stageLoadedStaleAddon(projectDir) { + const source = repoAddonPath() + const releaseDir = join( + projectDir, + 'node_modules', + '@vscode', + 'windows-process-tree', + 'build', + 'Release' + ) + mkdirSync(releaseDir, { recursive: true }) + const stale = join(releaseDir, 'windows_process_tree.node') + copyFileSync(source, stale) + appendFileSync(stale, 'ReadProcessMemory') + + const holder = spawn( + process.execPath, + ['-e', 'require(process.argv[1]); process.send("held"); setInterval(() => {}, 1000)', stale], + { stdio: ['ignore', 'ignore', 'ignore', 'ipc'] } + ) + await new Promise((resolve, reject) => { + holder.once('message', resolve) + holder.once('exit', () => reject(new Error('the addon holder exited before loading'))) + }) + return holder +} + +// Why an end-to-end run: the defect was purely one of placement. The guard threw +// a real EPERM, and the classifier that turns that into "close running Orca" +// already existed -- the throw simply happened before the try that reaches it. +// Only the whole script exercises that. +describe.runIf(process.platform === 'win32')('rebuild-native-deps stale addon under lock', () => { + it.skipIf(!repoAddonPath())( + 'reports a locked stale addon as a Windows file lock instead of an EPERM stack', + async () => { + const projectDir = mkTempProject() + let holder + + try { + writeFakeUsableElectronPackage(projectDir, { platform: 'win32' }) + writeFakeElectronRebuild(projectDir) + writeFakeNodePtyConptyPayload(projectDir, process.arch) + writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir) + holder = await stageLoadedStaleAddon(projectDir) + + const result = runRebuildScript( + projectDir, + { + npm_lifecycle_event: 'postinstall', + npm_config_platform: 'win32', + npm_config_arch: process.arch + }, + ['--platform=win32', `--arch=${process.arch}`, '--force'] + ) + + expect(result.stderr).toContain( + 'Close running Orca/Electron/dev processes for this worktree' + ) + // Non-strict postinstall soft-exits on a lock; the next dev/start re-checks. + expect(result.status, result.stderr).toBe(0) + } finally { + holder?.kill() + removeTreeSync(projectDir) + } + } + ) +}) diff --git a/config/scripts/rebuild-native-deps.mjs b/config/scripts/rebuild-native-deps.mjs index 3b17683e831..863aac850a1 100644 --- a/config/scripts/rebuild-native-deps.mjs +++ b/config/scripts/rebuild-native-deps.mjs @@ -20,7 +20,12 @@ import { rebuild } from '@electron/rebuild' import { execFileSync, spawnSync } from 'node:child_process' -import { stageWindowsProcessTreeNodeAddonApiHeaders } from './windows-process-tree-gyp-rebuild.mjs' +import { + ensureWindowsProcessTreeCommandLinePatch, + inspectWindowsProcessTreeAddon, + stageWindowsProcessTreeNodeAddonApiHeaders, + windowsProcessTreeAddonPath +} from './windows-process-tree-gyp-rebuild.mjs' import { copyFileSync, existsSync, @@ -141,15 +146,21 @@ if (!ignoreModules.includes('cpu-features')) { } } -if ( - rebuildPlatform === 'win32' && - modulesToRebuild.includes('@vscode/windows-process-tree') && - existsSync(join(projectDir, 'node_modules', '@vscode', 'windows-process-tree', 'package.json')) -) { - stageWindowsProcessTreeNodeAddonApiHeaders() -} - try { + // Why inside the try: the patch guard deletes a stale addon binary, and that + // delete fails EPERM when the addon is loaded -- exactly the running-Orca case + // the catch below is written for. Outside, it aborted `pnpm install` with a + // raw stack instead of the "close running Orca/Electron processes" message. + if ( + rebuildPlatform === 'win32' && + modulesToRebuild.includes('@vscode/windows-process-tree') && + existsSync(join(projectDir, 'node_modules', '@vscode', 'windows-process-tree', 'package.json')) + ) { + stageWindowsProcessTreeNodeAddonApiHeaders() + if (ensureWindowsProcessTreeCommandLinePatch()) { + console.warn('[rebuild] Repaired the un-applied windows-process-tree command-line patch.') + } + } await rebuild({ buildPath: projectDir, electronVersion, @@ -165,6 +176,7 @@ try { force: true }) restoreNodePtyWindowsConptyRuntime() + assertWindowsProcessTreeAddonIsPatched() } catch (/** @type {any} */ err) { console.error('[rebuild] Native module rebuild failed:', err?.message ?? err) if (isWindowsNativeLockError(err)) { @@ -184,6 +196,40 @@ try { process.exit(1) } +/** + * The binary this rebuild just produced is the one the packaged app ships. + * + * The relay build asserts its own artifact and `ensure-native-runtime.mjs` + * asserts what it loads, but nothing checked the addon that gets copied into the + * packaged `node_modules` -- so a rebuild that silently produced the upstream + * reader would reach users. Anything but `clean` fails: after a rebuild that + * reported success the binary must exist, so `missing` is a broken build, not an + * absence to shrug at. This is the caller that needs the state to be a state and + * not a boolean. + */ +function assertWindowsProcessTreeAddonIsPatched() { + if ( + rebuildPlatform !== 'win32' || + !modulesToRebuild.includes('@vscode/windows-process-tree') || + !existsSync(join(projectDir, 'node_modules', '@vscode', 'windows-process-tree', 'package.json')) + ) { + return + } + const addonPath = windowsProcessTreeAddonPath() + const state = inspectWindowsProcessTreeAddon(addonPath) + if (state === 'clean') { + return + } + throw new Error( + state === 'missing' + ? `the rebuild reported success but ${addonPath} is not there, so the packaged app would ` + + 'ship no windows-process-tree addon at all.' + : `${addonPath} still imports ReadProcessMemory, so it was not built from the patched ` + + 'command-line reader. The packaged app would carry the primitive MDE scores as ' + + 'credential dumping.' + ) +} + function restoreNodePtyWindowsConptyRuntime() { if (rebuildPlatform !== 'win32' || !onlyModules.includes('node-pty')) { return diff --git a/config/scripts/redactor-environment-lines-benchmark.mjs b/config/scripts/redactor-environment-lines-benchmark.mjs new file mode 100644 index 00000000000..71aebf9fe88 --- /dev/null +++ b/config/scripts/redactor-environment-lines-benchmark.mjs @@ -0,0 +1,47 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import { stripTypeScriptTypes } from 'node:module' +import { performance } from 'node:perf_hooks' +import { redactString } from '../../src/main/observability/redactor.ts' + +// Supply an unchanged redactor.ts snapshot to measure the actual previous production function. +const baselinePath = process.argv[2] +if (!baselinePath) { + throw new Error( + 'Usage: node config/scripts/redactor-environment-lines-benchmark.mjs ' + ) +} +const baselineSource = stripTypeScriptTypes(readFileSync(baselinePath, 'utf8')) +const { redactString: before } = await import( + `data:text/javascript;base64,${Buffer.from(baselineSource).toString('base64')}` +) +function median(fn, input, repeats) { + const samples = [] + for (let run = 0; run < repeats; run++) { + const started = performance.now() + fn(input) + samples.push(performance.now() - started) + } + return samples.sort((a, b) => a - b)[Math.floor(samples.length / 2)] +} +const rows = [] +for (const [shape, input] of [ + ['8KiB blank lines', '\n'.repeat(8192)], + ['16KiB blank lines', '\n'.repeat(16384)], + ['32KiB blank lines', '\n'.repeat(32768)], + ['32KiB blank lines then invalid key', `${'\n'.repeat(32768)}lowercase`], + ['ordinary env', 'FOO=value\nBAR=other\n'], + ['ordinary message', 'Cannot read directory /workspace/source: file not found'] +]) { + assert.equal(redactString(input), before(input)) + const beforeMs = median(before, input, 3) + const afterMs = median(redactString, input, 15) + rows.push({ + shape, + bytes: Buffer.byteLength(input), + beforeMs, + afterMs, + speedup: beforeMs / afterMs + }) +} +console.log(JSON.stringify({ node: process.version, platform: process.platform, rows }, null, 2)) diff --git a/config/scripts/relay-asset-line-ending-pin.test.mjs b/config/scripts/relay-asset-line-ending-pin.test.mjs new file mode 100644 index 00000000000..3384aa9b88a --- /dev/null +++ b/config/scripts/relay-asset-line-ending-pin.test.mjs @@ -0,0 +1,82 @@ +import { execFileSync } from 'node:child_process' +import { resolve } from 'node:path' +import { RELAY_ARTIFACTS } from '../../src/shared/relay-artifacts.ts' +import { describe, expect, it } from 'vitest' + +/** + * Guard the `.gitattributes` pin that keeps `config/relay-assets` on LF. + * + * `core.autocrlf=true` ships in the Git-for-Windows system config, so without a + * pin a Windows runner checks these out as CRLF. build-relay.mjs copies them + * verbatim into the bundle and hashes them byte-for-byte into `.version`, which + * names the immutable remote relay directory -- so a Windows-built client and a + * mac/Linux-built one disagree on the same release, and one SSH host ends up with + * two relay trees, each paying its own remote native-dep compile. + * + * Measured on v1.4.197: master-cloexec-patch.cjs shipped at 11229 bytes from the + * mac runner and 11547 (= 11229 + 318 lines) from the Windows one. + */ +const projectDir = resolve(import.meta.dirname, '../..') + +function git(args) { + return execFileSync('git', args, { cwd: projectDir, encoding: 'utf8' }) +} + +/** `git check-attr -z` emits NUL-separated path/attr/value triples. */ +function eolAttributes(paths) { + const fields = git(['check-attr', '-z', 'eol', '--', ...paths]).split('\0') + const found = new Map() + for (let index = 0; index + 2 < fields.length; index += 3) { + found.set(fields[index], fields[index + 2]) + } + return found +} + +/** + * Keyed off the manifest, not a directory: build-relay refuses to emit an + * artifact absent from RELAY_ARTIFACTS, so relocating an asset cannot slip + * past this the way a path glob would. esbuild bundles have no tracked + * source and contribute no hits, so they need no classifying. + */ +function trackedManifestSources() { + const paths = new Set() + for (const { filename } of RELAY_ARTIFACTS) { + const hits = git(['ls-files', '-z', '--', `*/${filename}`]) + .split('\0') + .filter(Boolean) + for (const path of hits) { + paths.add(path) + } + } + return [...paths] +} + +describe('config/relay-assets line-ending pin', () => { + it('pins every tracked relay artifact source to LF', () => { + const assets = trackedManifestSources() + expect(assets.length).toBeGreaterThan(0) + + const attributes = eolAttributes(assets) + const unpinned = assets.filter((path) => attributes.get(path) !== 'lf') + + expect( + unpinned, + 'A relay asset left on the platform default gets CRLF on a Windows runner, ' + + 'which changes the .version hash and splits one release across two remote ' + + 'relay directories. Pin it in .gitattributes.' + ).toEqual([]) + }) + + // Why: the assertion above only sees files that exist today. These fix the + // pattern itself -- broad enough to cover a file added tomorrow, narrow enough + // not to claim neighbours. + it.each([ + ['config/relay-assets/example.cjs', 'lf'], + ['config/relay-assets/nested/deeper/example.cjs', 'lf'], + ['config/relay-assets/example.txt', 'lf'], + ['config/relay-assets-extra/example.cjs', 'unspecified'], + ['vendor/config/relay-assets/example.cjs', 'unspecified'] + ])('resolves %s to eol=%s', (path, expected) => { + expect(eolAttributes([path]).get(path)).toBe(expected) + }) +}) diff --git a/config/scripts/relay-frame-buffer-benchmark.mjs b/config/scripts/relay-frame-buffer-benchmark.mjs new file mode 100644 index 00000000000..24d7b565400 --- /dev/null +++ b/config/scripts/relay-frame-buffer-benchmark.mjs @@ -0,0 +1,62 @@ +#!/usr/bin/env node +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import { stripTypeScriptTypes } from 'node:module' +import { performance } from 'node:perf_hooks' + +// Pass the pre-change source saved with git show :src/shared/relay-frame-buffer.ts. +const baselinePath = process.argv[2] +if (!baselinePath) { + throw new Error('Usage: node config/scripts/relay-frame-buffer-benchmark.mjs ') +} +async function load(source) { + return ( + await import( + `data:text/javascript;base64,${Buffer.from(stripTypeScriptTypes(source)).toString('base64')}` + ) + ).RelayFrameBuffer +} +const Before = await load(readFileSync(baselinePath, 'utf8')) +const After = await load( + readFileSync(new URL('../../src/shared/relay-frame-buffer.ts', import.meta.url), 'utf8') +) +function median(values) { + return values.sort((a, b) => a - b)[Math.floor(values.length / 2)] +} +for (const count of [1, 256, 16384, 65536]) { + const chunks = Array.from({ length: count }, (_, index) => Buffer.alloc(64, index % 256)) + const expected = Buffer.concat(chunks) + for (const mode of ['take', 'discard']) { + const times = [[], []] + for (let round = 0; round < 9; round += 1) { + for (const arm of round % 2 === 0 ? [0, 1] : [1, 0]) { + const FrameBuffer = arm === 0 ? Before : After + const buffer = new FrameBuffer() + for (const chunk of chunks) { + buffer.append(chunk) + } + const start = performance.now() + const output = buffer[mode](expected.length) + times[arm].push(performance.now() - start) + if (mode === 'take') { + assert.deepEqual(output, expected) + } + assert.equal(buffer.length, 0) + buffer.append(Buffer.from('tail')) + assert.equal(buffer.drain().toString(), 'tail') + } + } + const beforeMs = median(times[0]), + afterMs = median(times[1]) + console.log( + JSON.stringify({ + mode, + chunks: count, + bytes: expected.length, + beforeMs, + afterMs, + speedup: beforeMs / afterMs + }) + ) + } +} diff --git a/config/scripts/renderer-quadratic-scan-benchmark.mjs b/config/scripts/renderer-quadratic-scan-benchmark.mjs new file mode 100644 index 00000000000..681b6cf8550 --- /dev/null +++ b/config/scripts/renderer-quadratic-scan-benchmark.mjs @@ -0,0 +1,364 @@ +#!/usr/bin/env node +// Benchmarks four renderer projections that scaled worse than linearly with user data, each on a +// path that reruns per keystroke or per store write. +// +// Scenarios 1, 3 and 4 time the production export against a hand-written reproduction of the +// pre-change shape and assert both agree first. Scenario 2 is MODELLED on both sides: the +// projection lives inside the `useTabGroupItemProjections` React hook and cannot be imported +// without a renderer, so it reproduces the before/after loops rather than driving production. +import { spawnSync } from 'node:child_process' +import { transformSync } from 'esbuild' +import { performance } from 'node:perf_hooks' +import fs from 'node:fs' +import nodeModule from 'node:module' +import path from 'node:path' +import process from 'node:process' +import { fileURLToPath, pathToFileURL } from 'node:url' + +if (!process.execArgv.includes('--experimental-transform-types')) { + const result = spawnSync( + process.execPath, + ['--experimental-transform-types', '--no-warnings', import.meta.filename], + { stdio: 'inherit' } + ) + process.exit(result.status ?? 1) +} + +const ROOT = path.resolve(import.meta.dirname, '../..') +const RENDERER = path.join(ROOT, 'src/renderer/src') + +nodeModule.registerHooks({ + resolve(specifier, context, nextResolve) { + if (!context.parentURL) { + return nextResolve(specifier, context) + } + const candidates = specifier.startsWith('@/') + ? ['.ts', '.tsx', '/index.ts', '/index.tsx', ''].map( + (suffix) => path.join(RENDERER, specifier.slice(2)) + suffix + ) + : specifier.startsWith('.') && !/\.[cm]?[jt]sx?$/.test(specifier) + ? ['.ts', '.tsx'].map((suffix) => + fileURLToPath(new URL(specifier + suffix, context.parentURL)) + ) + : [] + const resolved = candidates.find((file) => fs.existsSync(file) && fs.statSync(file).isFile()) + return resolved + ? { url: pathToFileURL(resolved).href, shortCircuit: true } + : nextResolve(specifier, context) + }, + // Node strips types from .ts but not .tsx; the sidebar row model transitively imports icons. + load(url, context, nextLoad) { + if (url.endsWith('.tsx')) { + const source = fs.readFileSync(fileURLToPath(url), 'utf8') + const { code } = transformSync(source, { loader: 'tsx', format: 'esm', jsx: 'automatic' }) + return { format: 'module', source: code, shortCircuit: true } + } + if (url.endsWith('.json') && !url.includes('/node_modules/')) { + const source = fs.readFileSync(fileURLToPath(url), 'utf8') + return { format: 'module', source: `export default ${source}`, shortCircuit: true } + } + return nextLoad(url, context) + } +}) + +const importRenderer = (relativePath) => + import(pathToFileURL(path.join(RENDERER, relativePath)).href) + +function envInt(name, fallback) { + const value = Number(process.env[name] ?? fallback) + if (!Number.isSafeInteger(value) || value <= 0) { + throw new Error(`${name} must be a positive integer, got ${value}`) + } + return value +} + +const KEYSTROKES = envInt('ORCA_QUADRATIC_BENCH_KEYSTROKES', 12) +const WORKTREES = envInt('ORCA_QUADRATIC_BENCH_WORKTREES', 300) +const TABS = envInt('ORCA_QUADRATIC_BENCH_TABS', 60) +const OPEN_FILES = envInt('ORCA_QUADRATIC_BENCH_OPEN_FILES', 120) +const CHANGED_FILES = envInt('ORCA_QUADRATIC_BENCH_CHANGED_FILES', 5000) +const SIDEBAR_ROWS = envInt('ORCA_QUADRATIC_BENCH_SIDEBAR_ROWS', 600) +const SIDEBAR_REPOS = envInt('ORCA_QUADRATIC_BENCH_SIDEBAR_REPOS', 80) +if (SIDEBAR_REPOS > SIDEBAR_ROWS) { + throw new Error( + 'ORCA_QUADRATIC_BENCH_SIDEBAR_REPOS must not exceed ORCA_QUADRATIC_BENCH_SIDEBAR_ROWS' + ) +} + +function timeRounds(run, rounds = 7) { + run() + const samples = Array.from({ length: rounds }, () => { + const start = performance.now() + run() + return performance.now() - start + }).sort((left, right) => left - right) + return samples[Math.floor(rounds / 2)] +} + +function repeat(times, run) { + return () => { + let last + for (let round = 0; round < times; round += 1) { + last = run() + } + return last + } +} + +const results = [] +function compare({ label, scale, drives, before, after }) { + if (JSON.stringify(before()) !== JSON.stringify(after())) { + throw new Error(`${label}: baseline disagreed with the indexed shape`) + } + results.push({ label, scale, drives, beforeMs: timeRounds(before), afterMs: timeRounds(after) }) +} + +// ------------------------------------------------- 1. workspace board search index + +const { buildWorkspaceBoardPaletteDocuments, matchWorkspaceBoardWorktrees } = await importRenderer( + 'components/sidebar/workspace-kanban-search.ts' +) + +const repoMap = new Map([ + ['repo-1', { id: 'repo-1', name: 'orca', path: '/tmp/orca', branch: 'main' }] +]) +const boardWorktrees = Array.from({ length: WORKTREES }, (_, index) => ({ + id: `repo-1::/tmp/worktree-${index}`, + repoId: 'repo-1', + path: `/tmp/worktree-${index}`, + branch: `feature/search-target-${index}`, + title: `Workspace ${index} search target`, + isMain: false +})) +const queries = Array.from({ length: KEYSTROKES }, (_, index) => 'search'.slice(0, (index % 6) + 1)) +const matchAll = (documents) => + queries.map((query) => [ + ...matchWorkspaceBoardWorktrees({ worktrees: boardWorktrees, query, repoMap, documents }) + ]) + +compare({ + label: 'workspace board filter (per keystroke burst)', + scale: `${WORKTREES} worktrees x ${KEYSTROKES} keystrokes`, + drives: 'production', + // Omitting `documents` is the pre-change shape: the index is rebuilt inside every match. + before: () => matchAll(undefined), + // The hook memoizes the index on [worktrees, repoMap]; only the match reruns per keystroke. + after: () => matchAll(buildWorkspaceBoardPaletteDocuments({ worktrees: boardWorktrees, repoMap })) +}) + +// ------------------------------------------------- 2. tab-group projections (modelled) + +const groupTabs = Array.from({ length: TABS }, (_, index) => ({ + id: `tab-${index}`, + entityId: `entity-${index}`, + contentType: index % 3 === 0 ? 'editor' : 'terminal' +})) +const openFiles = Array.from({ length: OPEN_FILES }, (_, index) => ({ + id: `entity-${index}`, + path: `/tmp/file-${index}.ts` +})) +const tabOrder = groupTabs.map((tab) => tab.id) +// Production memoizes each index on its own source list, so a unified-tab write reuses it. +const openFileById = new Map(openFiles.map((item) => [item.id, item])) +const groupTabById = new Map(groupTabs.map((item) => [item.id, item])) + +function tabProjections(findOpenFile, findGroupTab) { + const editorItems = groupTabs + .filter((item) => item.contentType === 'editor') + .map((item) => findOpenFile(item.entityId)) + .filter((file) => file !== undefined) + const order = tabOrder.map((itemId) => findGroupTab(itemId)?.entityId ?? itemId) + return [editorItems, order] +} + +compare({ + label: 'tab-group projections (per unified-tab write)', + scale: `${TABS} tabs x ${OPEN_FILES} open files`, + drives: 'modelled', + before: repeat(200, () => + tabProjections( + (id) => openFiles.find((candidate) => candidate.id === id), + (id) => groupTabs.find((candidate) => candidate.id === id) + ) + ), + after: repeat(200, () => + tabProjections( + (id) => openFileById.get(id), + (id) => groupTabById.get(id) + ) + ) +}) + +// ------------------------------------------------- 3. source-control tree build + +const { buildSourceControlTree } = await importRenderer( + 'components/right-sidebar/source-control-tree.ts' +) +const { normalizeRelativePath } = await importRenderer('lib/path.ts') +const { splitPathSegments } = await importRenderer('components/right-sidebar/path-tree.ts') +const { compareFileNames } = await import( + pathToFileURL(path.join(ROOT, 'src/shared/file-name-sort.ts')).href +) + +const changedEntries = Array.from({ length: CHANGED_FILES }, (_, index) => ({ + path: `src/area-${index % 20}/module-${index % 60}/nested/deep/part-${index % 7}/file-${index}.ts` +})) + +// Pre-change `buildSourceControlTree`: identical except each ancestor path is re-joined. +function buildSourceControlTreeBefore(area, entries) { + const makeDirectory = (dirPath, name, depth) => ({ + type: 'directory', + key: `dir::${area}::${dirPath}`, + name, + path: dirPath, + area, + depth, + fileCount: 0, + children: [], + directoryChildren: new Map() + }) + const root = makeDirectory('', '', -1) + for (const entry of entries) { + const normalizedPath = normalizeRelativePath(entry.path) + const segments = splitPathSegments(normalizedPath) + if (segments.length === 0) { + continue + } + let parent = root + for (let index = 0; index < segments.length - 1; index += 1) { + const name = segments[index] + const dirPath = segments.slice(0, index + 1).join('/') + let dir = parent.directoryChildren.get(name) + if (!dir) { + dir = makeDirectory(dirPath, name, index) + parent.directoryChildren.set(name, dir) + parent.children.push(dir) + } + parent = dir + } + parent.children.push({ + type: 'file', + key: `${area}::${entry.path}`, + name: segments.at(-1), + path: normalizedPath, + entry, + area, + depth: segments.length - 1 + }) + } + const finalize = (node) => { + const directories = node.children.filter((child) => child.type === 'directory').map(finalize) + const files = node.children.filter((child) => child.type === 'file') + directories.sort((a, b) => compareFileNames(a.name, b.name)) + files.sort((a, b) => compareFileNames(a.entry.path, b.entry.path)) + const { directoryChildren: _, ...rest } = node + return { + ...rest, + fileCount: files.length + directories.reduce((count, dir) => count + dir.fileCount, 0), + children: [...directories, ...files] + } + } + return finalize(root).children +} + +compare({ + label: 'source-control tree build (per filter keystroke)', + scale: `${CHANGED_FILES} changed files`, + drives: 'production', + before: () => buildSourceControlTreeBefore('unstaged', changedEntries), + after: () => buildSourceControlTree('unstaged', changedEntries) +}) + +// ------------------------------------------------- 4. sidebar header boundaries + +const { getRepoHeaderSectionEndByRepoId } = await importRenderer( + 'components/sidebar/worktree-header-section-boundaries.ts' +) +const { estimateRenderRowSize } = await importRenderer( + 'components/sidebar/worktree-list/viewport/virtual-rows.ts' +) + +const headerRowIndexes = new Set( + Array.from({ length: SIDEBAR_REPOS }, (_, repo) => + Math.floor((repo * SIDEBAR_ROWS) / SIDEBAR_REPOS) + ) +) +const sidebarRows = Array.from({ length: SIDEBAR_ROWS }, (_, index) => + headerRowIndexes.has(index) + ? { + type: 'header', + key: `repo:${index}`, + label: '', + count: 0, + tone: '', + repo: { id: `repo-${index}` } + } + : { type: 'item', rowKey: `wt:${index}`, sectionKey: '', depth: 0, groupDepth: 0 } +) +const headerRepoIds = sidebarRows.filter((row) => row.type === 'header').map((row) => row.repo.id) +const boundaryArgs = { + rows: sidebarRows, + firstHeaderIndex: 0, + // What `getSidebarOrderedRepoHeaderIdsByBucket` yields for repos outside any project group. + sidebarRepoHeaderIdsByBucket: new Map([['ungrouped', headerRepoIds]]), + repoHeaderBucketByRepoId: new Map(headerRepoIds.map((id) => [id, 'ungrouped'])) +} + +// Pre-change `getRepoHeaderSectionEndByRepoId`: a findIndex and an indexOf per header row. +function getRepoHeaderSectionEndByRepoIdBefore(args) { + const rowStarts = [] + let offset = 0 + for (let index = 0; index < args.rows.length; index += 1) { + rowStarts[index] = offset + offset += estimateRenderRowSize(args.rows, index, args.firstHeaderIndex, null) + } + rowStarts[args.rows.length] = offset + const sectionEndByRepoId = new Map() + for (let index = 0; index < args.rows.length; index += 1) { + const row = args.rows[index] + const repoId = row?.type === 'header' ? row.repo?.id : undefined + if (!repoId) { + continue + } + const bucketKey = args.repoHeaderBucketByRepoId.get(repoId) + const bucketRepoIds = bucketKey ? args.sidebarRepoHeaderIdsByBucket.get(bucketKey) : undefined + const bucketIndex = bucketRepoIds?.indexOf(repoId) ?? -1 + const nextRepoId = bucketIndex >= 0 ? bucketRepoIds?.[bucketIndex + 1] : undefined + let endIndex = -1 + if (nextRepoId) { + endIndex = args.rows.findIndex((r) => r.type === 'header' && r.repo?.id === nextRepoId) + } else { + endIndex = args.rows.length + for (let next = index + 1; next < args.rows.length; next += 1) { + if (args.rows[next]?.type === 'header' || args.rows[next]?.type === 'host-header') { + endIndex = next + break + } + } + } + sectionEndByRepoId.set( + repoId, + rowStarts[endIndex >= 0 ? endIndex : args.rows.length] ?? rowStarts[args.rows.length] ?? 0 + ) + } + return sectionEndByRepoId +} + +compare({ + label: 'sidebar header boundaries (per row-model rebuild)', + scale: `${SIDEBAR_REPOS} repos x ${SIDEBAR_ROWS} rows`, + drives: 'production', + before: repeat(50, () => [...getRepoHeaderSectionEndByRepoIdBefore(boundaryArgs)]), + after: repeat(50, () => [...getRepoHeaderSectionEndByRepoId(boundaryArgs)]) +}) + +// ------------------------------------------------- + +console.log('Renderer quadratic-scan removals\n') +console.log('| projection | drives | scale | before | after | |') +console.log('| --- | --- | --- | --- | --- | --- |') +for (const row of results) { + console.log( + `| ${row.label} | ${row.drives} | ${row.scale} | ${row.beforeMs.toFixed(2)} ms | ${row.afterMs.toFixed(2)} ms | ${(row.beforeMs / row.afterMs).toFixed(1)}x |` + ) +} diff --git a/config/scripts/replace-cached-nsis-elevate.mjs b/config/scripts/replace-cached-nsis-elevate.mjs new file mode 100644 index 00000000000..fcd1a7323d4 --- /dev/null +++ b/config/scripts/replace-cached-nsis-elevate.mjs @@ -0,0 +1,260 @@ +#!/usr/bin/env node + +// Why: electron-builder re-runs `CopyElevateHelper.copy` on every NSIS pack, so the +// release rebuild overwrites the SignPath-signed `resources/elevate.exe` with the +// unsigned copy sitting in the electron-builder toolset cache. The release workflow +// swapped the cached copy first, but searched `/nsis` — a directory no current +// app-builder-lib layout creates (real ones are `/nsis-3.0.4.1/nsis-3.0.4.1-/` +// and `/nsis@/nsis-bundle--/`), so the swap silently found +// nothing and v1.4.193/v1.4.194 shipped an unsigned UAC elevation helper. + +import { copyFileSync, readdirSync, statSync } from 'node:fs' +import { createRequire } from 'node:module' +import { homedir, platform as osPlatform, tmpdir } from 'node:os' +import { join, parse, resolve } from 'node:path' + +const require = createRequire(import.meta.url) + +const ELEVATE_EXE = 'elevate.exe' + +// `nsis` (the layout the old hardcoded path assumed), `nsis-3.0.4.1` (legacy bundle via +// `getBinFromUrl`), `nsis@1.2.1` (unified bundle). Not `customNsisBinary`: the +// `nsis-` key `getBinFromCustomLoc` builds is only `getBin`'s in-process promise +// key, and the extract dir is named for the custom URL's parent segment, which need not +// start with `nsis` at all. Only the app-builder-lib probe covers that layout — which is +// why the probe, not this scan, is what decides whether the swap succeeded. +const NSIS_RELEASE_DIR = /^nsis(?:[-@].*)?$/i + +// elevate.exe lives at the bundle root, one level under the release dir. The legacy +// bundle carries thousands of files under Contrib/, so an unbounded walk is both slow +// and a way to match something that is not a toolset copy. +const MAX_DEPTH = 3 + +function isFile(path) { + try { + return statSync(path).isFile() + } catch { + return false + } +} + +/** + * Mirrors `getCacheDirectory` in app-builder-lib's `out/util/electronGet.js`, which is what + * decides where the NSIS bundle is unpacked. Kept as a local port rather than an import + * because the swap must still resolve a cache root when app-builder-lib cannot be loaded. + */ +export function resolveElectronBuilderCacheDir({ + env = process.env, + platform = osPlatform(), + home = homedir(), + temp = tmpdir() +} = {}) { + const override = env.ELECTRON_BUILDER_CACHE?.trim() + if (override && parse(override).root) { + return override + } + if (platform === 'darwin') { + return join(home, 'Library', 'Caches', 'electron-builder') + } + if (platform === 'win32') { + const localAppData = env.LOCALAPPDATA?.trim() + // https://github.com/electron-userland/electron-builder/issues/1164 + const isSystemUser = + localAppData?.toLowerCase().includes('\\windows\\system32\\') === true || + env.USERNAME?.trim().toLowerCase() === 'system' + if (!localAppData || isSystemUser) { + return join(temp, 'electron-builder-cache') + } + return join(localAppData, 'electron-builder', 'Cache') + } + const xdgCache = env.XDG_CACHE_HOME + return xdgCache && parse(xdgCache).root + ? join(xdgCache, 'electron-builder') + : join(home, '.cache', 'electron-builder') +} + +function collectElevateFiles(dir, depth, found) { + let entries + try { + entries = readdirSync(dir, { withFileTypes: true }) + } catch { + return found + } + for (const entry of entries) { + const path = join(dir, entry.name) + if (entry.isFile()) { + if (entry.name.toLowerCase() === ELEVATE_EXE) { + found.push(path) + } + } else if (entry.isDirectory() && depth > 1) { + collectElevateFiles(path, depth - 1, found) + } + } + return found +} + +/** + * Every cached `elevate.exe` under an NSIS release directory of `cacheDir`, plus the + * `ELECTRON_BUILDER_NSIS_DIR` override copy when that is set. + */ +export function findCachedElevatePaths(cacheDir, { env = process.env } = {}) { + const found = [] + const overrideDir = env.ELECTRON_BUILDER_NSIS_DIR?.trim() + if (overrideDir && isFile(join(overrideDir, ELEVATE_EXE))) { + found.push(join(overrideDir, ELEVATE_EXE)) + } + let entries + try { + entries = readdirSync(cacheDir, { withFileTypes: true }) + } catch { + return found + } + for (const entry of entries) { + if (entry.isDirectory() && NSIS_RELEASE_DIR.test(entry.name)) { + collectElevateFiles(join(cacheDir, entry.name), MAX_DEPTH, found) + } + } + return found +} + +/** + * The exact path `CopyElevateHelper` will pack, asked of app-builder-lib itself. Returns the + * failure instead of logging it: an unavailable probe leaves the directory scan as the only + * signal, and the caller has to say that out loud rather than quietly passing. + */ +export async function resolveToolsetElevatePath(projectDir = process.cwd()) { + try { + const configPath = require.resolve(resolve(projectDir, 'config/electron-builder.config.cjs')) + const config = require(configPath) + const { getNsisElevatePath } = require('app-builder-lib/out/toolsets/windows.js') + const path = await getNsisElevatePath(config.toolsets?.nsis, config.nsis?.customNsisBinary) + return { path, error: null } + } catch (error) { + return { path: null, error: error.message } + } +} + +/** + * Replaces every cached copy rather than picking one. Which bundle the rebuild packs + * depends on the toolset version resolved at pack time, and each cached copy is an + * unsigned `elevate.exe` that a later pack could reach for; the helper is a standalone + * UAC shim, not coupled to the NSIS version around it, so overwriting all of them is safe. + * + * `toolsetReplaced` is the signal that matters. A non-empty `replaced` only says that some + * cached copy was rewritten, which a stale release directory carried in by the + * `electron-builder-win-` prefix restore can satisfy on its own. + */ +export async function replaceCachedElevateHelpers({ + signedPath, + cacheDir = resolveElectronBuilderCacheDir(), + projectDir = process.cwd(), + env = process.env, + probe = resolveToolsetElevatePath +} = {}) { + if (!isFile(signedPath)) { + throw new Error(`Signed elevate.exe not found: ${signedPath}`) + } + const targets = new Set(findCachedElevatePaths(cacheDir, { env })) + const { path: toolsetPath, error: toolsetError } = await probe(projectDir) + if (toolsetPath != null && isFile(toolsetPath)) { + targets.add(toolsetPath) + } + + const replaced = [] + for (const target of targets) { + copyFileSync(signedPath, target) + replaced.push(target) + } + return { + replaced, + cacheDir, + toolsetPath, + toolsetError, + toolsetReplaced: toolsetPath != null && replaced.includes(toolsetPath) + } +} + +/** + * The annotations and exit code a swap result earns. Split out so every branch is testable + * without a subprocess — including the one that made this defect class possible, where the + * step passes because *a* cached copy was replaced while the copy the rebuild packs was not. + */ +export function summarizeSwap({ replaced, cacheDir, toolsetPath, toolsetError, toolsetReplaced }) { + if (toolsetPath != null && !toolsetReplaced) { + return { + annotations: [ + { + level: 'error', + message: + `app-builder-lib resolves the elevate.exe the NSIS rebuild will pack to ${toolsetPath}, ` + + 'but that path could not be replaced, so the installer will ship an unsigned UAC ' + + 'elevation helper.' + } + ], + exitCode: 1 + } + } + if (replaced.length === 0) { + return { + annotations: [ + { + level: 'error', + message: + `No cached elevate.exe found under ${cacheDir}; the NSIS rebuild will pack the unsigned ` + + 'helper and ship an unsigned UAC elevation binary. The electron-builder toolset cache ' + + 'layout has changed — update config/scripts/replace-cached-nsis-elevate.mjs.' + } + ], + exitCode: 1 + } + } + if (toolsetPath == null) { + // A green step must never quietly mean "the authoritative check did not run". The scan + // alone is satisfiable by a stale release directory that the `electron-builder-win-` + // prefix restore carried across a lockfile change, while the bundle the rebuild actually + // packs sits in a directory this scan does not match. + return { + annotations: [ + { + level: 'warning', + message: + 'Could not ask app-builder-lib which elevate.exe the NSIS rebuild will pack ' + + `(${toolsetError}); replaced ${replaced.length} copies found by scanning ${cacheDir} ` + + 'alone, which a stale release directory can satisfy while the packed copy stays unsigned.' + } + ], + exitCode: 0 + } + } + return { annotations: [], exitCode: 0 } +} + +// Why an exit code and not a warning: a swap that misses the copy the rebuild packs exits +// before that rebuild restores the unsigned helper, so a silent success here is +// indistinguishable from a release that shipped a signed one — which is how this went +// unnoticed for two releases. The workflow step is `continue-on-error`, so this annotates +// loudly without making a release unbuildable. +if (import.meta.filename === process.argv[1]) { + const signedPath = process.argv[2] + if (!signedPath) { + process.stderr.write('Usage: replace-cached-nsis-elevate.mjs \n') + process.exit(2) + } + try { + const result = await replaceCachedElevateHelpers({ signedPath }) + const { annotations, exitCode } = summarizeSwap(result) + for (const { level, message } of annotations) { + process.stdout.write(`::${level}::${message}\n`) + } + if (exitCode === 0) { + for (const path of result.replaced) { + const role = path === result.toolsetPath ? ' (the copy app-builder-lib will pack)' : '' + process.stdout.write(`Replaced ${path} with the SignPath-signed copy.${role}\n`) + } + } + process.exit(exitCode) + } catch (error) { + process.stdout.write(`::error::Could not replace the cached elevate.exe: ${error.message}\n`) + process.exit(1) + } +} diff --git a/config/scripts/replace-cached-nsis-elevate.test.mjs b/config/scripts/replace-cached-nsis-elevate.test.mjs new file mode 100644 index 00000000000..a88461703c3 --- /dev/null +++ b/config/scripts/replace-cached-nsis-elevate.test.mjs @@ -0,0 +1,364 @@ +import { spawnSync } from 'node:child_process' +import { + existsSync, + mkdirSync, + mkdtempSync, + readdirSync, + readFileSync, + rmSync, + writeFileSync +} from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { parse } from 'yaml' + +import { + findCachedElevatePaths, + replaceCachedElevateHelpers, + resolveElectronBuilderCacheDir, + summarizeSwap +} from './replace-cached-nsis-elevate.mjs' + +// The probe is app-builder-lib asking itself where the packed elevate.exe lives; injected +// here so no test needs the network or a warm toolset cache. +const probeFound = (path) => async () => ({ path, error: null }) +const probeUnavailable = async () => ({ path: null, error: 'app-builder-lib not loadable' }) + +const projectRoot = resolve(import.meta.dirname, '../..') +const scriptPath = join(projectRoot, 'config/scripts/replace-cached-nsis-elevate.mjs') + +let scratch + +beforeEach(() => { + scratch = mkdtempSync(join(tmpdir(), 'orca elevate swap ')) +}) + +afterEach(() => { + rmSync(scratch, { recursive: true, force: true }) +}) + +function makeCache(...relativeFiles) { + const cacheDir = join(scratch, 'Cache') + for (const relative of relativeFiles) { + const path = join(cacheDir, ...relative.split('/')) + mkdirSync(join(path, '..'), { recursive: true }) + writeFileSync(path, 'unsigned-elevate') + } + mkdirSync(cacheDir, { recursive: true }) + return cacheDir +} + +describe('cached elevate.exe swap covers the real electron-builder layouts', () => { + // Why these exact shapes: `downloadBuilderToolset` unpacks to + // `//-/`, and `releaseName` is + // `nsis-3.0.4.1` on the legacy bundle (`getBinFromUrl`) and `nsis@` on the + // unified bundle. The release workflow searched `/nsis`, which matches none of + // them. `customNsisBinary` is deliberately absent — see the probe suite below. + it.each([ + ['legacy bundle', 'nsis-3.0.4.1/nsis-3.0.4.1-1mx3n/elevate.exe'], + ['unified bundle', 'nsis@1.2.1/nsis-bundle-3.12-k4d9x/elevate.exe'], + ['bare nsis release dir', 'nsis/nsis-3.0.4.1/elevate.exe'] + ])('finds the cached helper in the %s layout', (_label, relative) => { + const cacheDir = makeCache(relative) + expect(findCachedElevatePaths(cacheDir, { env: {} })).toEqual([ + join(cacheDir, ...relative.split('/')) + ]) + }) + + it('leaves other toolsets and the raw download dir alone', () => { + const cacheDir = makeCache( + 'winCodeSign/winCodeSign-2.6.0-abc12/elevate.exe', + 'downloads/nsis/elevate.exe' + ) + expect(findCachedElevatePaths(cacheDir, { env: {} })).toEqual([]) + }) + + // `nsis-resources-3.4.1` matches the release-dir pattern and is scanned. Documented + // rather than excluded: `getLegacyNsisResourcesBin` ships plugins, never an elevate.exe, + // so the over-match costs one cheap directory read and nothing else. Narrowing the + // pattern to exclude it would be a guess about a name app-builder-lib owns. + it('scans the resources bundle too, which ships no helper to find', () => { + expect( + findCachedElevatePaths(makeCache('nsis-resources-3.4.1/plugins/x86-unicode/nsProcess.dll'), { + env: {} + }) + ).toEqual([]) + + const planted = 'nsis-resources-3.4.1/nsis-resources-3.4.1-p8w1z/elevate.exe' + const cacheDir = makeCache(planted) + expect(findCachedElevatePaths(cacheDir, { env: {} })).toEqual([ + join(cacheDir, ...planted.split('/')) + ]) + }) + + // The rebuild picks one bundle, and nothing outside app-builder-lib knows which. + // Replacing every cached copy is the deliberate answer to that ambiguity. + it('replaces every cached copy when several bundles are present', async () => { + const cacheDir = makeCache( + 'nsis-3.0.4.1/nsis-3.0.4.1-1mx3n/elevate.exe', + 'nsis@1.2.1/nsis-bundle-3.12-k4d9x/elevate.exe' + ) + const signed = join(scratch, 'signed-elevate.exe') + writeFileSync(signed, 'signpath-signed-elevate') + + const { replaced } = await replaceCachedElevateHelpers({ + signedPath: signed, + cacheDir, + env: {}, + probe: probeUnavailable + }) + + expect(replaced).toHaveLength(2) + for (const path of replaced) { + expect(readFileSync(path, 'utf8')).toBe('signpath-signed-elevate') + } + }) + + it('covers the ELECTRON_BUILDER_NSIS_DIR override copy', () => { + const overrideDir = join(scratch, 'nsis-override') + mkdirSync(overrideDir, { recursive: true }) + writeFileSync(join(overrideDir, 'elevate.exe'), 'unsigned-elevate') + const cacheDir = makeCache() + + expect( + findCachedElevatePaths(cacheDir, { env: { ELECTRON_BUILDER_NSIS_DIR: overrideDir } }) + ).toEqual([join(overrideDir, 'elevate.exe')]) + }) + + it('resolves the cache root the same way app-builder-lib does', () => { + expect( + resolveElectronBuilderCacheDir({ + env: { LOCALAPPDATA: 'C:\\Users\\runneradmin\\AppData\\Local' }, + platform: 'win32' + }) + ).toBe(join('C:\\Users\\runneradmin\\AppData\\Local', 'electron-builder', 'Cache')) + expect(resolveElectronBuilderCacheDir({ env: {}, platform: 'darwin', home: '/Users/a' })).toBe( + join('/Users/a', 'Library', 'Caches', 'electron-builder') + ) + expect(resolveElectronBuilderCacheDir({ env: { ELECTRON_BUILDER_CACHE: '/mnt/cache' } })).toBe( + '/mnt/cache' + ) + }) + + // Proof against the layout actually on disk, not just the fixtures. Cross-checked + // against an independent unbounded walk so a search that scopes itself wrongly + // cannot pass by finding nothing — which is exactly how the inline path passed. + // Skipped only where no NSIS bundle has been downloaded into the cache yet. + it('finds every elevate.exe the real electron-builder cache holds', (ctx) => { + const cacheDir = resolveElectronBuilderCacheDir() + if (!existsSync(cacheDir)) { + // Reported as skipped, never as passed: this is the one test that checks the scan + // against a layout nobody wrote down, and a silent no-op here is the suite + // confirming itself. The Linux unit-test job has no electron-builder cache. + ctx.skip() + return + } + const walk = (dir) => + readdirSync(dir, { withFileTypes: true }).flatMap((entry) => { + const path = join(dir, entry.name) + if (entry.isDirectory()) { + return walk(path) + } + return entry.name.toLowerCase() === 'elevate.exe' ? [path] : [] + }) + const onDisk = walk(cacheDir) + if (onDisk.length === 0) { + ctx.skip() + return + } + expect(findCachedElevatePaths(cacheDir, { env: {} }).sort()).toEqual(onDisk.sort()) + }) +}) + +describe('the probe, not the scan, decides whether the swap worked', () => { + // Why the probe is load-bearing: `getBinFromCustomLoc` passes `nsis-` to `getBin` + // as its in-process promise key only — the extract dir is named for the custom URL's parent + // segment, so a customNsisBinary bundle can sit outside `nsis*` entirely. + it('covers a custom bundle the directory scan cannot match', async () => { + const relative = 'orca-nsis-mirror/nsis-custom-3.11-0zqp2/elevate.exe' + const cacheDir = makeCache(relative) + const packed = join(cacheDir, ...relative.split('/')) + const signed = join(scratch, 'signed-elevate.exe') + writeFileSync(signed, 'signpath-signed-elevate') + + expect(findCachedElevatePaths(cacheDir, { env: {} })).toEqual([]) + + const result = await replaceCachedElevateHelpers({ + signedPath: signed, + cacheDir, + env: {}, + probe: probeFound(packed) + }) + + expect(result.toolsetReplaced).toBe(true) + expect(readFileSync(packed, 'utf8')).toBe('signpath-signed-elevate') + expect(summarizeSwap(result)).toEqual({ annotations: [], exitCode: 0 }) + }) + + // The shape that reproduced the hole: release-cut.yml restores the toolset cache with + // `restore-keys: electron-builder-win-`, so a stale release directory survives a lockfile + // change. Replacing that stale copy satisfies `replaced.length > 0` on its own while the + // bundle the rebuild packs sits in a directory the scan never matches. + it('does not call a stale directory a success when the packed bundle is unmatched', async () => { + const stale = 'nsis-3.0.4.1/nsis-3.0.4.1-1mx3n/elevate.exe' + const packed = 'builder-nsis@4.0.0/nsis-bundle-4.0-k4d9x/elevate.exe' + const cacheDir = makeCache(stale, packed) + const signed = join(scratch, 'signed-elevate.exe') + writeFileSync(signed, 'signpath-signed-elevate') + + const result = await replaceCachedElevateHelpers({ + signedPath: signed, + cacheDir, + env: {}, + probe: probeUnavailable + }) + + // The scan rewrote only the stale copy; the one that would be packed is untouched. + expect(result.replaced).toEqual([join(cacheDir, ...stale.split('/'))]) + expect(readFileSync(join(cacheDir, ...packed.split('/')), 'utf8')).toBe('unsigned-elevate') + + // So the run must not look clean. + const { annotations, exitCode } = summarizeSwap(result) + expect(exitCode).toBe(0) + expect(annotations).toHaveLength(1) + expect(annotations[0].level).toBe('warning') + expect(annotations[0].message).toContain('Could not ask app-builder-lib') + }) + + it('fails when the probe names a copy that could not be replaced', () => { + const summary = summarizeSwap({ + replaced: ['C:/cache/nsis-3.0.4.1/nsis-3.0.4.1-1mx3n/elevate.exe'], + cacheDir: 'C:/cache', + toolsetPath: 'C:/cache/nsis@2.0.0/nsis-bundle-4.0-k4d9x/elevate.exe', + toolsetError: null, + toolsetReplaced: false + }) + + expect(summary.exitCode).toBe(1) + expect(summary.annotations[0].level).toBe('error') + expect(summary.annotations[0].message).toContain('will pack') + }) + + it('fails when nothing at all was replaced', () => { + const summary = summarizeSwap({ + replaced: [], + cacheDir: 'C:/cache', + toolsetPath: null, + toolsetError: 'app-builder-lib not loadable', + toolsetReplaced: false + }) + + expect(summary.exitCode).toBe(1) + expect(summary.annotations[0].level).toBe('error') + expect(summary.annotations[0].message).toContain('No cached elevate.exe found') + }) +}) + +describe('a cached elevate.exe miss is not silent', () => { + // ELECTRON_BUILDER_NSIS_DIR short-circuits app-builder-lib's own resolution before + // any download, so the probe fails offline instead of fetching the NSIS bundle. + function runScript(cacheDir, nsisDir, signedPath) { + return spawnSync(process.execPath, [scriptPath, signedPath], { + cwd: projectRoot, + encoding: 'utf8', + env: { + ...process.env, + ELECTRON_BUILDER_CACHE: cacheDir, + ELECTRON_BUILDER_NSIS_DIR: nsisDir + } + }) + } + + it('exits non-zero with an ::error:: annotation when no cached copy is found', () => { + const cacheDir = makeCache() + const emptyNsisDir = join(scratch, 'empty-nsis') + mkdirSync(emptyNsisDir, { recursive: true }) + const signed = join(scratch, 'signed-elevate.exe') + writeFileSync(signed, 'signpath-signed-elevate') + + const result = runScript(cacheDir, emptyNsisDir, signed) + + expect(result.status).toBe(1) + expect(result.stdout).toContain('::error::No cached elevate.exe found') + }) + + it('warns on the scan-only path so green never means the probe was skipped', () => { + const cacheDir = makeCache('nsis-3.0.4.1/nsis-3.0.4.1-1mx3n/elevate.exe') + const emptyNsisDir = join(scratch, 'empty-nsis') + mkdirSync(emptyNsisDir, { recursive: true }) + const signed = join(scratch, 'signed-elevate.exe') + writeFileSync(signed, 'signpath-signed-elevate') + + const result = runScript(cacheDir, emptyNsisDir, signed) + + expect(result.status).toBe(0) + expect(result.stdout).not.toContain('::error::') + expect(result.stdout).toContain('::warning::Could not ask app-builder-lib') + expect( + readFileSync(join(cacheDir, 'nsis-3.0.4.1', 'nsis-3.0.4.1-1mx3n', 'elevate.exe'), 'utf8') + ).toBe('signpath-signed-elevate') + }) + + // The healthy release-job path: app-builder-lib answers, so the copy it will pack is the + // one that gets replaced and there is nothing to warn about. + it('exits clean when the probe resolves the copy the rebuild will pack', () => { + const cacheDir = makeCache() + const nsisDir = join(scratch, 'nsis-bundle') + mkdirSync(nsisDir, { recursive: true }) + writeFileSync(join(nsisDir, 'elevate.exe'), 'unsigned-elevate') + const signed = join(scratch, 'signed-elevate.exe') + writeFileSync(signed, 'signpath-signed-elevate') + + const result = runScript(cacheDir, nsisDir, signed) + + expect(result.status).toBe(0) + expect(result.stdout).not.toContain('::error::') + expect(result.stdout).not.toContain('::warning::') + expect(result.stdout).toContain('the copy app-builder-lib will pack') + expect(readFileSync(join(nsisDir, 'elevate.exe'), 'utf8')).toBe('signpath-signed-elevate') + }) +}) + +describe('release-cut.yml swaps the cached elevate.exe through the resolver', () => { + function swapStep() { + const workflow = parse( + readFileSync(join(projectRoot, '.github/workflows/release-cut.yml'), 'utf8') + ) + const step = workflow.jobs.build.steps.find( + (candidate) => candidate.name === 'Replace cached elevate.exe with the signed copy' + ) + expect(step).toBeDefined() + return step + } + + it('delegates the cache lookup to the script instead of an inline path', () => { + const step = swapStep() + expect(step.run).toContain('node config/scripts/replace-cached-nsis-elevate.mjs $signed') + // The hardcoded miss that shipped v1.4.193/v1.4.194 unsigned. + expect(step.run).not.toContain('electron-builder\\Cache\\nsis') + expect(step.run).not.toContain('-ErrorAction SilentlyContinue') + }) + + it('fails the step when the swap reports a miss', () => { + const step = swapStep() + // Matched as an executed statement: downgrading this to a Write-Host restores + // the silent fail-open that let the unsigned helper ship. + expect(step.run).toMatch(/if \(\$LASTEXITCODE -ne 0\) \{/) + expect(step.run).toMatch(/^\s*throw \$message\s*$/m) + expect(step.run).toContain('GITHUB_STEP_SUMMARY') + }) + + // Why kept: windows-signing-rehearsal.yml shares the electron-builder-win- + // cache key, so dropping this guard would let a test certificate reach a release cache. + it('still refuses to stage anything but a SignPath-signed helper', () => { + const step = swapStep() + expect(step.run).toContain("$signature.Status -ne 'Valid'") + expect(step.run).toContain("$subject -notlike '*CN=SignPath Foundation*'") + }) + + // The inner-signing chain stays fail-open: a loud red step, not an unbuildable release. + it('keeps the step unable to fail the release job', () => { + expect(swapStep()['continue-on-error']).toBe(true) + }) +}) diff --git a/config/scripts/repo-icon-source-href-benchmark.mjs b/config/scripts/repo-icon-source-href-benchmark.mjs new file mode 100644 index 00000000000..76c42d261b4 --- /dev/null +++ b/config/scripts/repo-icon-source-href-benchmark.mjs @@ -0,0 +1,55 @@ +import assert from 'node:assert/strict' +import { performance } from 'node:perf_hooks' +import { extractIconHref } from '../../src/main/repo-icon-source-href.ts' + +// Original production expressions, preserved for the before/after measurement. +const html = + /]*\brel=["'](?:icon|shortcut icon)["'])(?=[^>]*\bhref=["']([^"'?]+))[^>]*>/i +const object = + /(?=[^}]*\brel\s*:\s*["'](?:icon|shortcut icon)["'])(?=[^}]*\bhref\s*:\s*["']([^"'?]+))[^}]*/i +const original = (source) => source.match(html)?.[1] ?? source.match(object)?.[1] ?? null + +function measurePair(source) { + original(source) + extractIconHref(source) + const beforeSamples = [] + const afterSamples = [] + for (let run = 0; run < 5; run++) { + const measurements = [ + [original, beforeSamples], + [extractIconHref, afterSamples] + ] + if (run % 2 === 1) { + measurements.reverse() + } + for (const [fn, samples] of measurements) { + const started = performance.now() + fn(source) + samples.push(performance.now() - started) + } + } + return { + beforeMs: beforeSamples.sort((a, b) => a - b)[2], + afterMs: afterSamples.sort((a, b) => a - b)[2] + } +} + +const results = [] +for (const size of [8192, 16384, 32768]) { + for (const shape of ['no icon', 'rel without href', 'unterminated link starts']) { + const source = + shape === 'unterminated link starts' + ? ' left - right) + return sorted[Math.floor(sorted.length / 2)] +} + +function timeRounds(run) { + const samples = [] + run() + for (let round = 0; round < ROUNDS; round += 1) { + const start = performance.now() + run() + samples.push(performance.now() - start) + } + return median(samples) +} + +function report(label, baselineMs, currentMs, extra = '') { + const speedup = baselineMs / currentMs + console.log( + `${label}\n before ${baselineMs.toFixed(3)} ms → after ${currentMs.toFixed(3)} ms (${speedup.toFixed(1)}x)${extra}` + ) + return speedup +} + +// ---------------------------------------------------------------- scenario 1 + +// Verbatim pre-change capTerminalScrollbackSessionBuffer; measureUtf8ByteLength itself is unchanged. +function baselineCapScrollbackBuffer(buffer) { + if ( + buffer.length <= TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT && + !measureUtf8ByteLength(buffer, { + stopAfterBytes: TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT + }).exceededLimit + ) { + return buffer + } + return clampUtf8TextTail(buffer, TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT).text +} + +// A terminal that has been running a while sits at the cap, which is the case that scanned in full. +const scrollbackLine = `${''}build output line with a path /Users/dev/project/src/index.ts and a status ok\n` +let atCapBuffer = '' +while (atCapBuffer.length < TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT) { + atCapBuffer += scrollbackLine +} +atCapBuffer = atCapBuffer.slice(0, TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT) + +if (capTerminalScrollbackSessionBuffer(atCapBuffer) !== baselineCapScrollbackBuffer(atCapBuffer)) { + throw new Error('scrollback cap disagreed with the baseline implementation') +} + +// The session write runs the prune twice, once per retained leaf. +const CAP_CALLS_PER_WRITE = LEAVES * 2 +const capBaselineMs = timeRounds(() => { + for (let call = 0; call < CAP_CALLS_PER_WRITE; call += 1) { + baselineCapScrollbackBuffer(atCapBuffer) + } +}) +const capCurrentMs = timeRounds(() => { + for (let call = 0; call < CAP_CALLS_PER_WRITE; call += 1) { + capTerminalScrollbackSessionBuffer(atCapBuffer) + } +}) + +console.log( + `Session-write hot path — ${LEAVES} retained scrollback leaves, ${PANE_KEYS} accumulated pane keys\n` +) +report( + `1. scrollback UTF-8 budget scan (${CAP_CALLS_PER_WRITE} calls/write @ ${(atCapBuffer.length / 1024).toFixed(0)} KB)`, + capBaselineMs, + capCurrentMs +) + +// ---------------------------------------------------------------- scenario 2 + +const paneKeys = {} +const leafIdByInputLeafIdByTabId = new Map() +for (let index = 0; index < PANE_KEYS; index += 1) { + const tabId = `tab-${index % 64}` + const leafId = `${(index % 64).toString(16).padStart(8, '0')}-0000-4000-8000-${index.toString(16).padStart(12, '0')}` + paneKeys[makePaneKey(tabId, leafId)] = index + let leaves = leafIdByInputLeafIdByTabId.get(tabId) + if (!leaves) { + leaves = new Map() + leafIdByInputLeafIdByTabId.set(tabId, leaves) + } + // Steady state: a stable UUID leaf maps to itself. + leaves.set(leafId, leafId) +} + +// Verbatim pre-change remapPaneKeys: parses every key, then rebuilds the object regardless. +function baselineRemapPaneKeys(values, remap) { + if (!values || Object.keys(values).length === 0) { + return { values, changed: false } + } + let changed = false + const next = {} + const setValue = (paneKey, value) => { + const existing = next[paneKey] + next[paneKey] = existing === undefined ? value : Math.max(existing, value) + } + for (const [paneKey, value] of Object.entries(values)) { + if (parsePaneKey(paneKey)) { + setValue(paneKey, value) + continue + } + const delimiter = paneKey.indexOf(':') + if (delimiter <= 0 || delimiter === paneKey.length - 1) { + setValue(paneKey, value) + continue + } + const tabId = paneKey.slice(0, delimiter) + const remappedLeafId = remap.get(tabId)?.get(paneKey.slice(delimiter + 1)) + if (!remappedLeafId || !isTerminalLeafId(remappedLeafId)) { + setValue(paneKey, value) + continue + } + try { + setValue(makePaneKey(tabId, remappedLeafId), value) + changed = true + } catch { + setValue(paneKey, value) + } + } + return { values: next, changed } +} + +// The write remaps three of these maps: acknowledgements, activity cutoffs, manual unread. +const REMAP_CALLS_PER_WRITE = 3 +const remapBaselineMs = timeRounds(() => { + for (let call = 0; call < REMAP_CALLS_PER_WRITE; call += 1) { + baselineRemapPaneKeys(paneKeys, leafIdByInputLeafIdByTabId) + } +}) +const remapCurrentMs = timeRounds(() => { + for (let call = 0; call < REMAP_CALLS_PER_WRITE; call += 1) { + remapAcknowledgedAgentPaneKeys(paneKeys, leafIdByInputLeafIdByTabId) + } +}) +const remapResult = remapAcknowledgedAgentPaneKeys(paneKeys, leafIdByInputLeafIdByTabId) +if (remapResult.changed || remapResult.acknowledgements !== paneKeys) { + throw new Error('steady-state remap should return the input map untouched') +} +report( + `2. pane-key remap (${REMAP_CALLS_PER_WRITE} maps/write @ ${PANE_KEYS} keys)`, + remapBaselineMs, + remapCurrentMs, + ' — and 3 discarded objects/write become 0' +) diff --git a/config/scripts/skill-description-length.test.mjs b/config/scripts/skill-description-length.test.mjs new file mode 100644 index 00000000000..e7a9db79541 --- /dev/null +++ b/config/scripts/skill-description-length.test.mjs @@ -0,0 +1,39 @@ +import { readdirSync, readFileSync } from 'node:fs' +import { join, resolve } from 'node:path' +import { describe, expect, it } from 'vitest' +import { parse } from 'yaml' + +const skillsDir = resolve(import.meta.dirname, '../../skills') +// Why: the Agent Skills spec caps `description` at 1024 chars and conforming installers +// reject the whole skill (#17935); the frontmatter is what the installer parses, so check it. +const MAX_DESCRIPTION_LENGTH = 1024 + +function readDescription(skillName) { + const skillMarkdown = readFileSync(join(skillsDir, skillName, 'SKILL.md'), 'utf8') + const frontmatter = /^---\r?\n([\s\S]*?)\r?\n---\r?\n/u.exec(skillMarkdown)?.[1] + + expect(frontmatter, `${skillName}: missing frontmatter`).toBeDefined() + + return parse(frontmatter ?? '').description +} + +describe('bundled skill descriptions', () => { + const skillNames = readdirSync(skillsDir, { withFileTypes: true }) + .filter((entry) => entry.isDirectory()) + .map((entry) => entry.name) + + it('discovers the bundled skills', () => { + expect(skillNames).toContain('orchestration') + }) + + it.each(skillNames)('%s keeps description within the Agent Skills spec limit', (name) => { + const description = readDescription(name) + + expect(typeof description, `${name}: description must be a string`).toBe('string') + expect(description.trim().length, `${name}: description is empty`).toBeGreaterThan(0) + expect( + description.length, + `${name}: description is ${description.length} chars` + ).toBeLessThanOrEqual(MAX_DESCRIPTION_LENGTH) + }) +}) diff --git a/config/scripts/sort-comparator-performance-plugin.test.mjs b/config/scripts/sort-comparator-performance-plugin.test.mjs new file mode 100644 index 00000000000..a9319c6238d --- /dev/null +++ b/config/scripts/sort-comparator-performance-plugin.test.mjs @@ -0,0 +1,45 @@ +import path from 'node:path' +import { describe, expect, it } from 'vitest' +import { runOxlintPluginOnSource } from './oxlint-plugin-test-runner.mjs' + +function lint(source) { + return runOxlintPluginOnSource({ + pluginName: 'sort-comparator-performance', + pluginPath: path.resolve('config/oxlint-plugins/sort-comparator-performance.mjs'), + rules: { 'sort-comparator-performance/no-repeated-collator': 'warn' }, + source + }) +} + +describe('sort comparator performance', () => { + it('reports repeated collation setup in inline sort and toSorted callbacks', () => { + const findings = lint(` + rows.sort((a, b) => a.name.localeCompare(b.name, locale, { sensitivity: 'base' })) + rows.toSorted(function (a, b) { return new Intl.Collator('sv').compare(a, b) }) + rows['sort']((a, b) => Intl.Collator('en', { numeric: true }).compare(a, b)) + rows.sort((a, b) => a['localeCompare'](b, undefined, options)) + `) + expect(findings).toHaveLength(4) + expect( + findings.every( + (finding) => finding.code === 'sort-comparator-performance(no-repeated-collator)' + ) + ).toBe(true) + }) + + it('allows one collator per sort, bare comparisons, and unrelated callbacks', () => { + expect( + lint(` + const collator = new Intl.Collator(locale, options) + rows.sort((a, b) => collator.compare(a.name, b.name) || a.id.localeCompare(b.id)) + rows.toSorted(collator.compare) + const equal = a.localeCompare(b, undefined, { sensitivity: 'accent' }) === 0 + rows.map(a => new Intl.Collator(a.locale)) + rows.sort((a, b) => { + function deferred() { return new Intl.Collator(locale) } + return a - b + }) + `) + ).toEqual([]) + }) +}) diff --git a/config/scripts/source-string-blanking-benchmark.mjs b/config/scripts/source-string-blanking-benchmark.mjs new file mode 100644 index 00000000000..de677b9baa9 --- /dev/null +++ b/config/scripts/source-string-blanking-benchmark.mjs @@ -0,0 +1,79 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { stripTypeScriptTypes } from 'node:module' +import { performance } from 'node:perf_hooks' +import { blankStringContents as after } from '../../src/shared/source-scan/source-tree-scan.ts' + +const ref = process.argv[2] +if (!ref) { + throw new Error('Usage: node config/scripts/source-string-blanking-benchmark.mjs ') +} +const source = execFileSync('git', ['show', `${ref}:src/shared/source-scan/source-tree-scan.ts`], { + encoding: 'utf8' +}) +const { blankStringContents: before } = await import( + `data:text/javascript;base64,${Buffer.from(stripTypeScriptTypes(source)).toString('base64')}` +) +const tokens = [ + 'a', + '/', + '*', + ' ', + '\n', + '\r', + '\t', + '\u00a0', + '\u2028', + '"', + "'", + '`', + '${', + '}', + '{', + '\\', + '(', + ')', + '[', + ']', + '=', + '+', + '-', + ';' +] +let seed = 173 +for (let sample = 0; sample < 3000; sample++) { + let input = '' + for (let token = 0; token < 40; token++) { + seed = (Math.imul(seed, 1664525) + 1013904223) >>> 0 + input += tokens[seed % tokens.length] + } + assert.equal(after(input), before(input), JSON.stringify(input)) + assert.equal(after(input, true), before(input, true), JSON.stringify(input)) +} +function measure(fn, input) { + const samples = [] + for (let run = 0; run < 3; run++) { + const start = performance.now() + fn(input) + samples.push(performance.now() - start) + } + return samples.sort((a, b) => a - b)[1] +} +const results = [] +for (const lines of [100, 1000, 5000, 10000]) { + const input = 'const x = value / 2;\n'.repeat(lines) + assert.equal(after(input), before(input)) + results.push({ + lines, + bytes: Buffer.byteLength(input), + beforeMs: measure(before, input), + afterMs: measure(after, input) + }) +} +console.log( + JSON.stringify( + { node: process.version, platform: process.platform, differentialCases: 3000, results }, + null, + 2 + ) +) diff --git a/config/scripts/terminal-ime-engagement-receipt.mjs b/config/scripts/terminal-ime-engagement-receipt.mjs index 9ad5255d235..8f0732908c1 100644 --- a/config/scripts/terminal-ime-engagement-receipt.mjs +++ b/config/scripts/terminal-ime-engagement-receipt.mjs @@ -13,7 +13,8 @@ export const IME_ENGAGEMENT_RECEIPT_ENV = 'ORCA_E2E_IME_ENGAGEMENT_RECEIPT' /** The tests that must each leave a receipt. Pinned so deleting one cannot quietly shrink the lane. */ export const EXPECTED_NATIVE_IME_TESTS = [ 'forwards the issue exact-byte sequence without loss or duplication', - 'forwards the issue sentence stress sequence without leaked ASCII' + 'forwards the issue sentence stress sequence without leaked ASCII', + 'a digit typed right after a Hangul syllable reaches the pty' ] function parseReceipts(text) { diff --git a/config/scripts/terminal-ime-engagement-receipt.test.mjs b/config/scripts/terminal-ime-engagement-receipt.test.mjs index 04161339a0f..613abc2ffae 100644 --- a/config/scripts/terminal-ime-engagement-receipt.test.mjs +++ b/config/scripts/terminal-ime-engagement-receipt.test.mjs @@ -4,7 +4,7 @@ import { verifyImeEngagementReceipts } from './terminal-ime-engagement-receipt.mjs' -const [firstTest, secondTest] = EXPECTED_NATIVE_IME_TESTS +const [firstTest, secondTest, thirdTest] = EXPECTED_NATIVE_IME_TESTS function receipt(test, overrides = {}) { return JSON.stringify({ @@ -18,9 +18,11 @@ function receipt(test, overrides = {}) { describe('verifyImeEngagementReceipts', () => { it('accepts a run where every expected test observed real composition', () => { - expect(verifyImeEngagementReceipts(`${receipt(firstTest)}\n${receipt(secondTest)}\n`)).toEqual( - [] - ) + expect( + verifyImeEngagementReceipts( + `${receipt(firstTest)}\n${receipt(secondTest)}\n${receipt(thirdTest)}\n` + ) + ).toEqual([]) }) // The failure this whole mechanism exists for: Playwright reports a skipped test as a pass, so @@ -35,13 +37,20 @@ describe('verifyImeEngagementReceipts', () => { it('rejects a partial run where only one test reached the engine', () => { expect(verifyImeEngagementReceipts(`${receipt(firstTest)}\n`)).toEqual([ - `no engagement receipt for "${secondTest}" — it was skipped, filtered out, or renamed` + `no engagement receipt for "${secondTest}" — it was skipped, filtered out, or renamed`, + `no engagement receipt for "${thirdTest}" — it was skipped, filtered out, or renamed` + ]) + }) + + it('requires the digit receipt even when both original native tests passed', () => { + expect(verifyImeEngagementReceipts(`${receipt(firstTest)}\n${receipt(secondTest)}\n`)).toEqual([ + `no engagement receipt for "${thirdTest}" — it was skipped, filtered out, or renamed` ]) }) it('rejects a run that typed keys but never opened a composition', () => { const problems = verifyImeEngagementReceipts( - `${receipt(firstTest, { compositionStart: 0 })}\n${receipt(secondTest)}\n` + `${receipt(firstTest, { compositionStart: 0 })}\n${receipt(secondTest)}\n${receipt(thirdTest)}\n` ) expect(problems).toEqual([ `"${firstTest}" recorded no compositionstart — the IME never engaged` @@ -50,7 +59,7 @@ describe('verifyImeEngagementReceipts', () => { it('rejects a composition that produced no Hangul, which a latin passthrough would satisfy', () => { const problems = verifyImeEngagementReceipts( - `${receipt(firstTest, { hangulComposition: 0 })}\n${receipt(secondTest)}\n` + `${receipt(firstTest, { hangulComposition: 0 })}\n${receipt(secondTest)}\n${receipt(thirdTest)}\n` ) expect(problems).toEqual([ `"${firstTest}" recorded no Hangul composition data — the engine produced no syllables` @@ -59,7 +68,7 @@ describe('verifyImeEngagementReceipts', () => { it('rejects a renamed test rather than counting it toward coverage', () => { const problems = verifyImeEngagementReceipts( - `${receipt(firstTest)}\n${receipt(secondTest)}\n${receipt('some new scenario')}\n` + `${receipt(firstTest)}\n${receipt(secondTest)}\n${receipt(thirdTest)}\n${receipt('some new scenario')}\n` ) expect(problems).toEqual([ 'unexpected engagement receipt for "some new scenario" — update EXPECTED_NATIVE_IME_TESTS' @@ -68,7 +77,7 @@ describe('verifyImeEngagementReceipts', () => { it('reports a truncated receipt rather than parsing around it', () => { const problems = verifyImeEngagementReceipts( - `${receipt(firstTest)}\n{"test":"trunc\n${receipt(secondTest)}\n` + `${receipt(firstTest)}\n{"test":"trunc\n${receipt(secondTest)}\n${receipt(thirdTest)}\n` ) expect(problems).toEqual(['malformed receipt line: {"test":"trunc']) }) diff --git a/config/scripts/terminal-partial-escape-tail-benchmark.mjs b/config/scripts/terminal-partial-escape-tail-benchmark.mjs new file mode 100644 index 00000000000..0337daa4e9d --- /dev/null +++ b/config/scripts/terminal-partial-escape-tail-benchmark.mjs @@ -0,0 +1,88 @@ +#!/usr/bin/env node +// Times the partial-escape-tail fold that runs once per PTY chunk for every terminal against a +// baseline with the pre-change shape (unconditional concat + per-code-unit walk). Equivalence is +// proven over a corpus first, so the reported speedup cannot come from the gate changing the answer. +import { performance } from 'node:perf_hooks' +import { + advancePartialEscapeTail, + extractPartialEscapeTail, + MAX_PARTIAL_ESCAPE_TAIL_LENGTH +} from '../../src/shared/terminal-partial-escape-tail.ts' + +const CHUNK_BYTES = 16 * 1024 +const CHUNKS = 640 +const ROUNDS = 7 + +function baselineAdvance(pendingTail, chunk) { + const tail = extractPartialEscapeTail(pendingTail + chunk) + return tail.length > MAX_PARTIAL_ESCAPE_TAIL_LENGTH ? '' : tail +} + +const chunkOf = (line) => line.repeat(Math.ceil(CHUNK_BYTES / line.length)).slice(0, CHUNK_BYTES) +const escFreeChunk = chunkOf('[build] compiled src/renderer/src/components/thing.tsx in 12ms\n') +const colouredChunk = chunkOf( + '\x1b[32m[build]\x1b[0m compiled src/renderer/src/components/thing.tsx in 12ms\n' +) + +// Every state the scanner can be left in, plus the boundaries the gate must not swallow. +const PIECES = [ + '', + 'plain output\n', + '\x1b[32mgreen\x1b[0m', + '\x1b[3', + '\x1b]0;title\x07', + '\x1b]0;partial', + '\x1bP dcs payload', + '\x1b', + '\x18', + '\x1a', + '\x1b]8;;https://example.com\x1b\\', + '\x1b]8;;https://example.com\x1b', + '\x1b(B', + '\x1b(', + '\x1b[1;2;3', + escFreeChunk +] +let checked = 0 +for (const pending of PIECES.map((piece) => extractPartialEscapeTail(piece))) { + for (const chunk of PIECES) { + const expected = baselineAdvance(pending, chunk) + const actual = advancePartialEscapeTail(pending, chunk) + if (expected !== actual) { + throw new Error( + `gate changed the tracked tail: ${JSON.stringify({ pending, chunk, expected, actual })}` + ) + } + checked += 1 + } +} + +function medianMs(advance, chunk) { + // First sample is the warm-up and is discarded. + const samples = Array.from({ length: ROUNDS + 1 }, () => { + const start = performance.now() + let tail = '' + for (let index = 0; index < CHUNKS; index += 1) { + tail = advance(tail, chunk) + } + return performance.now() - start + }) + return samples.slice(1).sort((left, right) => left - right)[Math.floor(ROUNDS / 2)] +} + +const megabytes = ((CHUNK_BYTES * CHUNKS) / 1024 / 1024).toFixed(1) +console.log( + `Partial-escape-tail fold: ${CHUNKS} x ${CHUNK_BYTES / 1024} KB chunks (${megabytes} MB), ${checked} equivalence cases verified\n` +) +console.log('| stream shape | before | after | |') +console.log('| --- | --- | --- | --- |') +for (const [label, chunk] of [ + ['ESC-free (build logs, `cat`, piped output)', escFreeChunk], + ['SGR-coloured output (gate does not apply)', colouredChunk] +]) { + const before = medianMs(baselineAdvance, chunk) + const after = medianMs(advancePartialEscapeTail, chunk) + console.log( + `| ${label} | ${before.toFixed(2)} ms | ${after.toFixed(2)} ms | ${(before / after).toFixed(1)}x |` + ) +} diff --git a/config/scripts/verify-dev-channel-packaging.test.mjs b/config/scripts/verify-dev-channel-packaging.test.mjs index 63e1c7d5b0c..8e5a00f48e1 100644 --- a/config/scripts/verify-dev-channel-packaging.test.mjs +++ b/config/scripts/verify-dev-channel-packaging.test.mjs @@ -53,6 +53,19 @@ describe('electron-builder dev-channel identity', () => { expect(config.win.verifyUpdateCodeSignature).toBe(false) }) + // Why on every channel: the hook is the only handle electron-builder gives on + // the NSIS uninstaller, and it signs nothing — it relays the file to and from + // the CI SignPath request. Carrying it must not drag a publisherName onto a + // dev build, which is the failure the split above exists to prevent. + it('carries the uninstaller sign hook without changing publisherName semantics', () => { + for (const env of [{}, WIN_ADHOC_ENV]) { + const config = loadConfigWithEnv(env) + expect(typeof config.win.signtoolOptions.sign).toBe('function') + } + expect(loadConfigWithEnv({}).win.signtoolOptions.publisherName).toBe('SignPath Foundation') + expect(loadConfigWithEnv(WIN_ADHOC_ENV).win.signtoolOptions.publisherName).toBeUndefined() + }) + it.each([ ['hourly', { ORCA_WIN_HOURLY: '1' }, 'orca-hourly'], ['daily', { ORCA_WIN_DAILY: '1' }, 'orca-daily'], diff --git a/config/scripts/verify-localization-catalog.mjs b/config/scripts/verify-localization-catalog.mjs index 002a84360f6..a73e9d5e3cc 100644 --- a/config/scripts/verify-localization-catalog.mjs +++ b/config/scripts/verify-localization-catalog.mjs @@ -11,7 +11,12 @@ import { repairTranslatedValue } from './locale-translation-policy.mjs' const SOURCE_EXTENSIONS = new Set(['.ts', '.tsx', '.js', '.jsx', '.mts', '.cts']) const SKIP_PATH_PARTS = new Set(['.git', 'dist', 'node_modules', 'out', '__snapshots__', 'assets']) -const LOCALIZATION_FUNCTION_NAMES = new Set(['t', 'translate', 'translateMain', 'translateSearchKeyword']) +const LOCALIZATION_FUNCTION_NAMES = new Set([ + 't', + 'translate', + 'translateMain', + 'translateSearchKeyword' +]) const PLACEHOLDER_RE = /\{\{[^}]+\}\}/g const LOCALES_RELATIVE_DIR = path.join('src', 'renderer', 'src', 'i18n', 'locales') export const LOCALIZATION_SOURCE_ROOTS = [ diff --git a/config/scripts/windows-process-tree-gyp-rebuild.mjs b/config/scripts/windows-process-tree-gyp-rebuild.mjs index c815407d6d0..20d91e55497 100644 --- a/config/scripts/windows-process-tree-gyp-rebuild.mjs +++ b/config/scripts/windows-process-tree-gyp-rebuild.mjs @@ -9,7 +9,8 @@ * hop escapes the store and configure fails with "node_addon_api.gyp not * found" (run 32999886072). */ -import { copyFileSync, mkdirSync, realpathSync } from 'node:fs' +import { execFileSync } from 'node:child_process' +import { copyFileSync, existsSync, mkdirSync, readFileSync, realpathSync, rmSync } from 'node:fs' import { createRequire } from 'node:module' import { dirname, join, resolve } from 'node:path' @@ -22,6 +23,16 @@ export const WINDOWS_PROCESS_TREE_PACKAGE_DIR = join( 'windows-process-tree' ) +export const WINDOWS_PROCESS_TREE_PATCH_PATH = join( + ROOT, + 'config', + 'patches', + '@vscode__windows-process-tree@0.8.0.patch' +) + +/** Only the patched reader defines this; the upstream one walks the PEB. */ +const COMMAND_LINE_PATCH_MARKER = 'kProcessCommandLineInformation' + export const WINDOWS_PROCESS_TREE_NODE_ADDON_API_HEADERS = [ 'napi.h', 'napi-inl.h', @@ -39,6 +50,119 @@ export function nodeGypRebuildInvocation(arch, packageDir = WINDOWS_PROCESS_TREE } } +/** The binary the addon actually loads. */ +export function windowsProcessTreeAddonPath(packageDir = WINDOWS_PROCESS_TREE_PACKAGE_DIR) { + return join(packageDir, 'build', 'Release', 'windows_process_tree.node') +} + +/** The import whose absence tells the patched binary from the published prebuilt. */ +const FLAGGED_IMPORT = 'ReadProcessMemory' + +/** + * Does this compiled addon still carry the flagged primitive? + * + * The patched reader never calls `ReadProcessMemory`, so the symbol is absent + * from its import table; the upstream build imports it. That makes this a + * property of the binary rather than of the source next to it, which matters + * because the published tarball ships a *loadable* prebuilt built from + * unpatched source: it is node-addon-api, so it satisfies a bare `require()` + * under both Node and Electron, and a skipped rebuild would use it. + * + * Tri-state, not a predicate: a binary that is not there has not been cleared, + * and a boolean makes "absent" indistinguishable from "verified clean" at every + * call site. Takes the binary path so the relay's staged addon -- which sits + * beside the bundle, with no package around it -- gets the same check. + * + * @param {string} addonPath + * @returns {'clean' | 'unpatched' | 'missing'} + */ +export function inspectWindowsProcessTreeAddon(addonPath) { + if (!existsSync(addonPath)) { + return 'missing' + } + return readFileSync(addonPath).includes(FLAGGED_IMPORT) ? 'unpatched' : 'clean' +} + +/** + * Refuse to compile or load the upstream command-line reader. + * + * Unpatched, it opens every process with `PROCESS_VM_READ` and walks the PEB to + * recover the command line -- the primitive MDE scores as credential dumping, + * and the reason this package is patched at all. pnpm has been seen + * materializing this CRLF package with its patch missing, so repair the source + * from the patch file, and drop any binary that predates the repair. + */ +export function ensureWindowsProcessTreeCommandLinePatch( + packageDir = WINDOWS_PROCESS_TREE_PACKAGE_DIR +) { + const source = join(packageDir, 'src', 'process_commandline.cc') + if (!existsSync(source)) { + throw new Error( + `${source} is missing, so the command-line patch cannot be verified. Run pnpm install.` + ) + } + let repaired = false + + if (!readFileSync(source, 'utf8').includes(COMMAND_LINE_PATCH_MARKER)) { + try { + execFileSync( + 'git', + [ + // Why force the line-ending mode: the patch is stored LF (a contract + // test forbids CR bytes in it), but upstream ships this source CRLF, + // so its pre-image lines and the file's differ by a CR. Under + // `core.autocrlf=false` -- Git's own built-in default, and what + // "checkout as-is" selects in the Git for Windows installer -- git + // compares them literally, the hunk does not match, and the repair + // throws. `input` normalizes line endings for that comparison and + // nothing else, so a hunk whose real content drifted is still + // rejected. Measured: without it, apply exits 1 at autocrlf=false and + // 0 at true/input; with it, 0 for CRLF and LF sources under all three. + '-c', + 'core.autocrlf=input', + 'apply', + '--include=src/process_commandline.cc', + WINDOWS_PROCESS_TREE_PATCH_PATH + ], + { + cwd: realpathSync(packageDir), + stdio: 'pipe', + // Why blind git to the repo: run inside a work tree, `git apply` + // prefixes patch paths with the cwd-relative prefix, silently skips + // everything that does not match -- and still exits 0. The package + // dir is always under the project root, so without this the repair + // reports success and changes nothing. + env: { ...process.env, GIT_DIR: join(packageDir, '.orca-no-such-git-dir') } + } + ) + } catch (error) { + throw new Error( + 'src/process_commandline.cc still reads the PEB, and repairing it from ' + + `${WINDOWS_PROCESS_TREE_PATCH_PATH} failed: ${error?.message ?? error}. Run pnpm install.` + ) + } + if (!readFileSync(source, 'utf8').includes(COMMAND_LINE_PATCH_MARKER)) { + throw new Error( + 'src/process_commandline.cc still reads the PEB after repair, so the patch did not ' + + 'apply. Run pnpm install.' + ) + } + repaired = true + } + + // A binary from before the repair -- or the tarball's own prebuilt -- would + // otherwise survive a skipped rebuild and load the flagged reader anyway. + // Deleting it can fail EPERM against a loaded (memory-mapped) addon, which + // `force: true` does not cover -- it only swallows ENOENT. That throw is the + // caller's to classify as a Windows file lock, so it must not be swallowed. + if (inspectWindowsProcessTreeAddon(windowsProcessTreeAddonPath(packageDir)) === 'unpatched') { + rmSync(windowsProcessTreeAddonPath(packageDir), { force: true }) + repaired = true + } + + return repaired +} + // Patched binding.gyp includes deps/node-addon-api; the tarball does not ship those headers. export function stageWindowsProcessTreeNodeAddonApiHeaders( packageDir = WINDOWS_PROCESS_TREE_PACKAGE_DIR diff --git a/config/scripts/windows-process-tree-gyp-rebuild.test.mjs b/config/scripts/windows-process-tree-gyp-rebuild.test.mjs index f4820e9430a..f2939b71179 100644 --- a/config/scripts/windows-process-tree-gyp-rebuild.test.mjs +++ b/config/scripts/windows-process-tree-gyp-rebuild.test.mjs @@ -10,8 +10,9 @@ import { } from 'node:fs' import { tmpdir } from 'node:os' import { join, resolve } from 'node:path' -import { describe, expect, it } from 'vitest' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' import { + inspectWindowsProcessTreeAddon, nodeGypRebuildInvocation, stageWindowsProcessTreeNodeAddonApiHeaders, WINDOWS_PROCESS_TREE_NODE_ADDON_API_HEADERS, @@ -59,3 +60,40 @@ describe('windows-process-tree node-gyp rebuild', () => { } }) }) + +describe('inspecting a compiled windows-process-tree addon', () => { + let dir + + beforeEach(() => { + dir = mkdtempSync(join(tmpdir(), 'orca-windows-process-tree-addon-')) + }) + afterEach(() => { + rmSync(dir, { recursive: true, force: true }) + }) + + it('reports a binary that still imports ReadProcessMemory as unpatched', () => { + const addonPath = join(dir, 'windows_process_tree.node') + writeFileSync(addonPath, Buffer.from('MZ\0\0KERNEL32.dll\0ReadProcessMemory\0', 'binary')) + expect(inspectWindowsProcessTreeAddon(addonPath)).toBe('unpatched') + }) + + it('reports a binary without the import as clean', () => { + const addonPath = join(dir, 'windows_process_tree.node') + writeFileSync(addonPath, Buffer.from('MZ\0\0ntdll.dll\0NtQueryInformationProcess\0', 'binary')) + expect(inspectWindowsProcessTreeAddon(addonPath)).toBe('clean') + }) + + // The whole point of the tri-state: absence is not evidence of safety, and a + // boolean made "there is no binary" indistinguishable from "checked, clean". + it('reports an absent binary as missing rather than clean', () => { + expect(inspectWindowsProcessTreeAddon(join(dir, 'windows_process_tree.node'))).toBe('missing') + }) + + it('inspects whatever path it is handed, including a relay-staged addon', () => { + // The relay loads `./windows-process-tree.node` beside its bundle, which is + // nowhere near a node_modules package directory. + const staged = join(dir, 'windows-process-tree.node') + writeFileSync(staged, Buffer.from('MZ\0\0ReadProcessMemory\0', 'binary')) + expect(inspectWindowsProcessTreeAddon(staged)).toBe('unpatched') + }) +}) diff --git a/config/scripts/windows-signing-workflow-contract.test.mjs b/config/scripts/windows-signing-workflow-contract.test.mjs index 37edc2196d4..c321db8cfd2 100644 --- a/config/scripts/windows-signing-workflow-contract.test.mjs +++ b/config/scripts/windows-signing-workflow-contract.test.mjs @@ -1,4 +1,5 @@ import { readFileSync } from 'node:fs' +import { createRequire } from 'node:module' import { join, resolve } from 'node:path' import { describe, expect, it } from 'vitest' import { parse } from 'yaml' @@ -212,6 +213,7 @@ describe('Windows signing workflow contract', () => { 'Notify Slack that inner-binary signing is waiting for approval', 'Download signed inner binaries from SignPath', 'Restore signed inner binaries into unpacked app', + 'Restore signed uninstaller for the installer rebuild', 'Replace cached elevate.exe with the signed copy', 'Rebuild NSIS installer from signed unpacked app' ] @@ -222,3 +224,235 @@ describe('Windows signing workflow contract', () => { } }) }) + +// Why these exist: the NSIS uninstaller is generated inside electron-builder's +// uninstaller pass and deleted immediately after being embedded, so the only way +// CI can sign it is the export/import relay through win.signtoolOptions.sign. +// Every link is asserted here the way Orca.exe and conpty_console_list.node are. +describe('Windows NSIS uninstaller signing', () => { + const releaseSteps = () => readWorkflow('.github/workflows/release-cut.yml').jobs.build.steps + const stepNamed = (steps, name) => steps.find((step) => step.name === name) + + const EXPORT_ENV = 'ORCA_WIN_UNINSTALLER_EXPORT_PATH' + const SIGNED_ENV = 'ORCA_WIN_UNINSTALLER_SIGNED_PATH' + + it('exports the uninstaller from the first Windows build', () => { + const build = stepNamed(releaseSteps(), 'Build Windows release artifacts') + + expect(build.env[EXPORT_ENV]).toContain('uninstaller-signing') + expect(build.env[EXPORT_ENV]).toContain('orca-uninstaller.exe') + }) + + // Why this is a test and not a comment: `files` in the electron-builder config + // is all-negation, so app-builder packs whatever is left in the checkout root. + // These steps retry, and a retried attempt would pack an unsigned .exe into + // app.asar — the very defect this chain removes. Every relay path must live + // outside the checkout. + it('keeps every relay path out of the packed checkout', () => { + const relayEnvValues = [ + ...releaseSteps(), + ...readWorkflow('.github/workflows/windows-signing-rehearsal.yml').jobs.rehearse.steps + ].flatMap((step) => [step.env?.[EXPORT_ENV], step.env?.[SIGNED_ENV]].filter(Boolean)) + + expect(relayEnvValues.length).toBe(4) + for (const value of relayEnvValues) { + expect(value).toContain('runner.temp') + expect(value).not.toContain('github.workspace') + } + + const relayScripts = [ + ...releaseSteps(), + ...readWorkflow('.github/workflows/windows-signing-rehearsal.yml').jobs.rehearse.steps + ] + .map((step) => step.run ?? '') + .filter((run) => run.includes('uninstaller-signing')) + + expect(relayScripts.length).toBeGreaterThan(0) + for (const run of relayScripts) { + // Why count occurrences rather than assert `toContain` once: a step + // carrying two relay paths could root the first in RUNNER_TEMP and leave + // the second bare-relative — which resolves against the checkout, and is + // exactly the shape of the defect this test exists to catch. + const mentions = run.match(/uninstaller-signing/g) ?? [] + const rooted = run.match(/Join-Path \$env:RUNNER_TEMP 'uninstaller-signing/g) ?? [] + + expect(rooted.length, run).toBe(mentions.length) + expect(run).not.toContain('$env:GITHUB_WORKSPACE') + } + }) + + it('stages the uninstaller into the same request as the inner binaries', () => { + const stage = stepNamed(releaseSteps(), 'Stage unsigned inner PE files for signing') + + expect(stage.run).toContain('uninstaller-signing\\unsigned\\orca-uninstaller.exe') + expect(stage.run).toContain('uninstaller\\orca-uninstaller.exe') + // No third SignPath request: exactly two submissions, as budgeted for the + // 1h + 4h approval waits inside the 360-minute job cap. + const submissions = releaseSteps().filter( + (step) => step.uses === 'signpath/github-action-submit-signing-request@v2' + ) + expect(submissions).toHaveLength(2) + }) + + // A staged-but-unreturned uninstaller must not fail the inner chain, or a + // SignPath artifact-configuration gap would cost the inner-binary signatures. + it('keeps the uninstaller out of the inner-binary copy-back list', () => { + const stage = stepNamed(releaseSteps(), 'Stage unsigned inner PE files for signing') + const restoreInner = stepNamed( + releaseSteps(), + 'Restore signed inner binaries into unpacked app' + ) + + expect(stage.run).not.toMatch(/\$list\.Add\(['"]uninstaller/) + expect(restoreInner.run).not.toContain('orca-uninstaller.exe') + }) + + // This step's outcome gates the upload of every inner binary, so a filesystem + // error while staging the uninstaller must not escape — otherwise one + // uninstaller-specific failure costs every inner-binary signature, which is + // strictly worse than the behaviour before this chain existed. + it('cannot let an uninstaller staging failure cost the inner-binary signatures', () => { + const stage = stepNamed(releaseSteps(), 'Stage unsigned inner PE files for signing') + const uninstallerBlock = stage.run.slice(stage.run.indexOf('$exportedUninstaller')) + + expect(stage.run).toMatch(/try \{[\s\S]*\$exportedUninstaller[\s\S]*\} catch \{/) + expect(uninstallerBlock).toContain('::warning::Could not stage the NSIS uninstaller') + expect(uninstallerBlock).not.toContain('throw') + // Explicit, so the catch does not silently depend on GitHub's + // $ErrorActionPreference='Stop' default for `shell: pwsh`. + expect(uninstallerBlock).toContain('New-Item -ItemType Directory -Force -Path (Split-Path') + expect(uninstallerBlock).toMatch(/New-Item[^\r\n]*-ErrorAction Stop/) + expect(uninstallerBlock).toMatch(/Copy-Item[^\r\n]*-ErrorAction Stop/) + // The upload it gates still keys off this step, so the catch is load-bearing. + expect(stepNamed(releaseSteps(), 'Upload unsigned inner binaries for SignPath').if).toContain( + "steps.stage-inner.outcome == 'success'" + ) + }) + + it('re-injects the signed uninstaller into the rebuilt installer', () => { + const steps = releaseSteps() + const restore = stepNamed(steps, 'Restore signed uninstaller for the installer rebuild') + const rebuild = stepNamed(steps, 'Rebuild NSIS installer from signed unpacked app') + const names = steps.map((step) => step.name) + + expect(restore.if).toContain('github.run_attempt == 1') + expect(restore.if).toContain("steps.restore-signed-inner.outcome == 'success'") + expect(restore.run).toContain('orca-uninstaller.exe') + expect(names.indexOf(restore.name)).toBeLessThan(names.indexOf(rebuild.name)) + expect(rebuild.env[SIGNED_ENV]).toContain('uninstaller-signing') + // The rebuild must not depend on the uninstaller leg: a missing signed + // uninstaller ships today's installer, it does not skip the rebuild. + expect(rebuild.if).not.toContain('restore-signed-uninstaller') + }) + + // NSIS hides the uninstaller in a compressed data section the bundled 7za + // cannot read, so the gate proves it from the sign hook's digest receipt + // instead of extracting it — and only when the relay actually ran. + it('reports the embedded uninstaller in the inner-binary evidence gate', () => { + const gate = stepNamed(releaseSteps(), 'Verify Windows inner binary signatures') + + expect(gate.env.UNINSTALLER_SIGNING_COMPLETED).toBe( + "${{ steps.restore-signed-uninstaller.outcome == 'success' }}" + ) + expect(gate.run).toContain('.embedded-sha256') + expect(gate.run).toContain("$env:UNINSTALLER_SIGNING_COMPLETED -eq 'true'") + expect(gate.run).toContain('not signed by SignPath Foundation: Uninstall Orca.exe') + // The uninstaller must not join the 7z payload loop, which cannot see it. + expect(gate.run).not.toContain("$targets += 'Uninstall Orca.exe'") + }) + + it('rehearses the uninstaller leg end to end', () => { + const steps = readWorkflow('.github/workflows/windows-signing-rehearsal.yml').jobs.rehearse + .steps + const names = steps.map((step) => step.name) + const pack = stepNamed(steps, 'Package Windows app and export the NSIS uninstaller') + const rebuild = stepNamed(steps, 'Build NSIS installer from signed unpacked app') + const verify = stepNamed(steps, 'Verify signatures end to end') + + // --dir never produces an uninstaller, so the rehearsal has to build the + // installer the way release-cut's first Windows pass does. + expect(pack.run).toContain('--win --publish never') + expect(pack.run).not.toContain('--dir') + expect(pack.env[EXPORT_ENV]).toContain('orca-uninstaller.exe') + expect(names).toContain('Restore signed uninstaller for the installer rebuild') + expect(rebuild.env[SIGNED_ENV]).toContain('orca-uninstaller.exe') + expect(verify.run).toContain('.embedded-sha256') + // The receipt only proves the import leg ran. The rehearsal is where the + // shipped uninstaller itself gets checked — the release job cannot install + // onto the runner it publishes from. + expect(verify.run).toContain('shipped: Uninstall Orca.exe') + expect(verify.run).toContain('-tnsis') + expect(verify.run).toContain("-ArgumentList '/S'") + }) + + // This workflow is the merge gate, so it must not be able to fail on its own + // artefact: 7-Zip's NSIS handler is unreliable enough that its output has to + // be corroborated before a signature verdict is drawn from it. + it('never lets an unreliable extract fail the rehearsal', () => { + const steps = readWorkflow('.github/workflows/windows-signing-rehearsal.yml').jobs.rehearse + .steps + const verify = stepNamed(steps, 'Verify signatures end to end') + + // The 7-Zip route is only trusted when it reproduces the relayed bytes; + // otherwise it falls through to the install route rather than failing. + expect(verify.run).toContain( + 'Write-Host "7-Zip\'s NSIS output did not match the relayed digest; falling back to a silent install."' + ) + expect(verify.run).toMatch(/\$installedUninstaller = \$null\r?\n\s*\}/) + + // The comparison that is not tautological: a file NSIS wrote out, against + // the digest the sign hook recorded. + expect(verify.run).toContain('$shippedDigest -ne $expectedDigest') + expect(verify.run).toContain('the uninstaller the installer ships is not the relayed one') + + // An installer that prompts must not hang to the 360-minute job cap, and + // the app it launches must not outlive the step holding install-dir handles. + expect(verify.run).toContain('-PassThru') + expect(verify.run).toContain('$installerProcess.WaitForExit(300000)') + expect(verify.run).toContain('the silent install did not exit within 5 minutes') + expect(verify.run).toMatch(/for \(\$attempt = 0; \$attempt -lt 20; \$attempt\+\+\)/) + expect(verify.run).toContain("Get-Process -Name 'orca-terminal-daemon'") + }) + + // resources\elevate.exe is downgraded to advisory because app-builder-lib's + // CopyElevateHelper clobbers it on every nsis pack — a pre-existing defect + // that predates the uninstaller relay and is being tracked separately. The + // escape hatch it needed is the kind that quietly grows until the gate + // asserts nothing, so pin it to exactly that one file. + it('confines the advisory escape hatch to elevate.exe', () => { + const steps = readWorkflow('.github/workflows/windows-signing-rehearsal.yml').jobs.rehearse + .steps + const verify = stepNamed(steps, 'Verify signatures end to end') + const advisoryCalls = verify.run + .split('\n') + .filter((line) => line.includes('-Advisory') && line.includes('Test-Signature')) + + expect(advisoryCalls).toHaveLength(1) + expect(advisoryCalls[0]).toContain('installed: $relative') + expect(verify.run).toContain("if ($relative -eq 'resources\\elevate.exe')") + + // Both uninstaller verdicts stay fatal — the whole point of the gate. + for (const call of ['relayed: orca-uninstaller.exe', 'shipped: Uninstall Orca.exe']) { + const line = verify.run + .split('\n') + .find((it) => it.includes(`Test-Signature`) && it.includes(call)) + expect(line, call).toBeDefined() + expect(line, call).not.toContain('-Advisory') + } + + // An advisory must still reach the evidence artifact, or downgrading it + // becomes indistinguishable from deleting the check. + expect(verify.run).toContain('ADVISORY (known pre-existing') + expect(verify.run).toContain('$script:advisories.Add($problem)') + }) + + it('wires the electron-builder sign hook that the relay depends on', () => { + const require = createRequire(import.meta.url) + const configPath = resolve(projectDir, 'config/electron-builder.config.cjs') + delete require.cache[require.resolve(configPath)] + const config = require(configPath) + + expect(typeof config.win.signtoolOptions.sign).toBe('function') + delete require.cache[require.resolve(configPath)] + }) +}) diff --git a/config/scripts/windows-uninstaller-signing.cjs b/config/scripts/windows-uninstaller-signing.cjs new file mode 100644 index 00000000000..c3243b4581a --- /dev/null +++ b/config/scripts/windows-uninstaller-signing.cjs @@ -0,0 +1,111 @@ +// Why this exists: the NSIS uninstaller is the one Orca binary SignPath never +// saw. app-builder-lib builds it in a separate makensis pass, hands it to the +// packager's sign hook, embeds it in the installer, then deletes it +// (NsisTarget.computeScriptAndSignUninstaller → packager.signIf(uninstallerPath), +// then `unlink(defines.UNINSTALLER_OUT_FILE)`). That hook is the only moment the +// file exists on disk, so it is the only place a post-hoc signer can reach it. +// +// Orca does not sign during electron-builder — SignPath signs afterwards, behind +// a human approval — so instead of signing, this hook relays: build 1 exports the +// unsigned uninstaller so CI can put it in the existing inner-binaries SignPath +// request, and the rebuild-from-signed-tree pass swaps the signed bytes back in +// before makensis embeds them. +// +// Trap for whoever adds a real certificate to the Windows build: a custom sign +// hook *replaces* signtool rather than running alongside it — windowsSignToolManager +// does `const executor = customSign || (config => this.doSign(config))`. Inert +// today (no CSC_LINK/WIN_CSC_LINK anywhere in the Windows workflows), but setting +// one would silently sign nothing until this hook learns to delegate. +// +// Trap for whoever adds a second NSIS target or arch: app-builder-lib names the +// intermediate uninstaller per target *and* arch, while the relay is a single +// pair of env vars. Two targets would race — last write wins on export, every +// installer would embed the same uninstaller, and the receipt could not tell. +// Release is x64-only `--win` with `win.target` unset (so `["nsis"]`) today. +const { createHash } = require('node:crypto') +const { copyFileSync, existsSync, mkdirSync, readFileSync, writeFileSync } = require('node:fs') +const { basename, dirname } = require('node:path') + +// app-builder-lib names the intermediate uninstaller `__uninstaller.exe`. +const UNINSTALLER_BASENAME_SUFFIX = '__uninstaller.exe' + +// Why a receipt: NSIS embeds the uninstaller in its own compressed data section, +// not in the app 7z payload the evidence gate extracts, so the shipped installer +// cannot be inspected for it with the bundled 7za. The receipt records the digest +// of the exact bytes handed to makensis, which the gate compares against the +// SignPath-returned file — proving what was embedded without extracting it. +const EMBEDDED_RECEIPT_SUFFIX = '.embedded-sha256' + +const isNsisUninstallerArtifact = (filePath) => + typeof filePath === 'string' && basename(filePath).endsWith(UNINSTALLER_BASENAME_SUFFIX) + +/** + * Pure relay. Returns a short verdict string for logging and tests. + * Never throws: a relay failure must ship today's installer, not break the build. + */ +function relayNsisUninstaller({ + filePath, + exportPath, + signedPath, + fs = { copyFileSync, existsSync, mkdirSync, readFileSync, writeFileSync } +}) { + if (!isNsisUninstallerArtifact(filePath)) { + return 'not-uninstaller' + } + try { + // Import wins over export: the rebuild pass must embed the signed bytes even + // though it also regenerates an unsigned uninstaller of its own. + if (signedPath) { + if (!fs.existsSync(signedPath)) { + return 'signed-missing' + } + fs.copyFileSync(signedPath, filePath) + const digest = createHash('sha256').update(fs.readFileSync(filePath)).digest('hex') + fs.writeFileSync(`${signedPath}${EMBEDDED_RECEIPT_SUFFIX}`, digest) + return 'imported' + } + if (exportPath) { + fs.mkdirSync(dirname(exportPath), { recursive: true }) + fs.copyFileSync(filePath, exportPath) + return 'exported' + } + return 'idle' + } catch (error) { + return `failed: ${error.message}` + } +} + +const VERDICT_MESSAGES = { + imported: (paths) => `embedded the SignPath-signed uninstaller from ${paths.signedPath}`, + exported: (paths) => `exported the unsigned uninstaller to ${paths.exportPath}`, + 'signed-missing': (paths) => + `no signed uninstaller at ${paths.signedPath}; embedding the unsigned one (fail-open)` +} + +/** + * electron-builder `win.signtoolOptions.sign` hook. Called for every Windows + * executable, twice per file (once per signing hash), so it must be cheap for + * non-uninstaller paths and idempotent for the uninstaller. + */ +function signWindowsUninstallerViaSignPath(configuration) { + const paths = { + filePath: configuration?.path, + exportPath: process.env.ORCA_WIN_UNINSTALLER_EXPORT_PATH || undefined, + signedPath: process.env.ORCA_WIN_UNINSTALLER_SIGNED_PATH || undefined + } + const verdict = relayNsisUninstaller(paths) + const message = VERDICT_MESSAGES[verdict] + if (message) { + console.log(`[win-uninstaller-signing] ${message(paths)}`) + } else if (verdict.startsWith('failed')) { + console.warn(`[win-uninstaller-signing] ${verdict}; embedding the unsigned uninstaller.`) + } +} + +module.exports = { + EMBEDDED_RECEIPT_SUFFIX, + UNINSTALLER_BASENAME_SUFFIX, + isNsisUninstallerArtifact, + relayNsisUninstaller, + signWindowsUninstallerViaSignPath +} diff --git a/config/scripts/windows-uninstaller-signing.test.mjs b/config/scripts/windows-uninstaller-signing.test.mjs new file mode 100644 index 00000000000..57ebfbdf786 --- /dev/null +++ b/config/scripts/windows-uninstaller-signing.test.mjs @@ -0,0 +1,235 @@ +import { createHash } from 'node:crypto' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { createRequire } from 'node:module' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' + +const require = createRequire(import.meta.url) +const { + EMBEDDED_RECEIPT_SUFFIX, + isNsisUninstallerArtifact, + relayNsisUninstaller, + signWindowsUninstallerViaSignPath +} = require('./windows-uninstaller-signing.cjs') + +const makeDir = () => mkdtempSync(join(tmpdir(), 'orca-uninstaller-signing-')) + +describe('isNsisUninstallerArtifact', () => { + // The name app-builder-lib's NsisTarget.computeScriptAndSignUninstaller gives + // the intermediate uninstaller; the hook keys off nothing else. + it('matches only electron-builder intermediate uninstallers', () => { + expect(isNsisUninstallerArtifact('C:\\dist\\orca-windows-setup.__uninstaller.exe')).toBe(true) + expect(isNsisUninstallerArtifact('/dist/orca-windows-setup.__uninstaller.exe')).toBe(true) + expect(isNsisUninstallerArtifact('C:\\dist\\win-unpacked\\Orca.exe')).toBe(false) + expect(isNsisUninstallerArtifact('C:\\dist\\orca-windows-setup.exe')).toBe(false) + expect(isNsisUninstallerArtifact(undefined)).toBe(false) + }) +}) + +describe('relayNsisUninstaller', () => { + const writeUninstaller = (dir, contents) => { + const filePath = join(dir, 'orca-windows-setup.__uninstaller.exe') + writeFileSync(filePath, contents) + return filePath + } + + it('ignores every file that is not the uninstaller', () => { + const dir = makeDir() + const filePath = join(dir, 'Orca.exe') + writeFileSync(filePath, 'app') + expect(relayNsisUninstaller({ filePath, exportPath: join(dir, 'out', 'x.exe') })).toBe( + 'not-uninstaller' + ) + }) + + it('exports the unsigned uninstaller, creating the destination directory', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'unsigned-uninstaller') + const exportPath = join(dir, 'uninstaller-signing', 'unsigned', 'orca-uninstaller.exe') + + expect(relayNsisUninstaller({ filePath, exportPath })).toBe('exported') + expect(readFileSync(exportPath, 'utf8')).toBe('unsigned-uninstaller') + }) + + it('overwrites the freshly built uninstaller with the signed bytes', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'rebuild-unsigned') + const signedPath = join(dir, 'signed', 'orca-uninstaller.exe') + mkdirSync(join(dir, 'signed')) + writeFileSync(signedPath, 'signpath-signed') + + expect(relayNsisUninstaller({ filePath, signedPath })).toBe('imported') + expect(readFileSync(filePath, 'utf8')).toBe('signpath-signed') + }) + + // The receipt is the evidence gate's only handle on the embedded uninstaller: + // NSIS hides it in a compressed section the bundled 7za cannot read. + it('records the digest of the bytes it handed makensis', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'rebuild-unsigned') + const signedPath = join(dir, 'signed', 'orca-uninstaller.exe') + mkdirSync(join(dir, 'signed')) + writeFileSync(signedPath, 'signpath-signed') + + relayNsisUninstaller({ filePath, signedPath }) + + const expected = createHash('sha256').update('signpath-signed').digest('hex') + expect(readFileSync(`${signedPath}${EMBEDDED_RECEIPT_SUFFIX}`, 'utf8')).toBe(expected) + }) + + it('leaves no receipt when the signed uninstaller never came back', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'unsigned-uninstaller') + const signedPath = join(dir, 'absent', 'orca-uninstaller.exe') + + relayNsisUninstaller({ filePath, signedPath }) + + expect(existsSync(`${signedPath}${EMBEDDED_RECEIPT_SUFFIX}`)).toBe(false) + }) + + // Import wins so the rebuild pass embeds the signed bytes even though it also + // regenerates an unsigned uninstaller of its own. + it('prefers importing over exporting when both are configured', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'rebuild-unsigned') + const signedPath = join(dir, 'signed', 'orca-uninstaller.exe') + mkdirSync(join(dir, 'signed')) + writeFileSync(signedPath, 'signpath-signed') + + expect( + relayNsisUninstaller({ filePath, signedPath, exportPath: join(dir, 'out', 'x.exe') }) + ).toBe('imported') + expect(readFileSync(filePath, 'utf8')).toBe('signpath-signed') + }) + + // Fail-open: a missing or unwritable relay must leave the build with today's + // unsigned uninstaller, never throw. + it('leaves the unsigned uninstaller in place when no signed copy came back', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'unsigned-uninstaller') + + expect( + relayNsisUninstaller({ filePath, signedPath: join(dir, 'absent', 'orca-uninstaller.exe') }) + ).toBe('signed-missing') + expect(readFileSync(filePath, 'utf8')).toBe('unsigned-uninstaller') + }) + + it('swallows filesystem errors instead of failing the build', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'unsigned-uninstaller') + const fs = { + existsSync: () => true, + mkdirSync: () => {}, + copyFileSync: () => { + throw new Error('EACCES') + } + } + + expect(relayNsisUninstaller({ filePath, exportPath: join(dir, 'x.exe'), fs })).toBe( + 'failed: EACCES' + ) + }) + + it('does nothing when neither relay path is configured (local builds)', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'unsigned-uninstaller') + + expect(relayNsisUninstaller({ filePath })).toBe('idle') + expect(readFileSync(filePath, 'utf8')).toBe('unsigned-uninstaller') + }) +}) + +// Why a suite of its own: this is the function electron-builder actually calls, +// and it runs inside `Build Windows release artifacts`, which has no +// continue-on-error. If it throws, the release job dies before a single +// SignPath request is made. Nothing else in the chain guards that. +describe('signWindowsUninstallerViaSignPath', () => { + const RELAY_VARS = ['ORCA_WIN_UNINSTALLER_EXPORT_PATH', 'ORCA_WIN_UNINSTALLER_SIGNED_PATH'] + + const withEnv = (env, run) => { + const saved = Object.fromEntries(RELAY_VARS.map((key) => [key, process.env[key]])) + const apply = (values) => { + for (const key of RELAY_VARS) { + if (values[key] === undefined) { + delete process.env[key] + } else { + process.env[key] = values[key] + } + } + } + apply({ ...Object.fromEntries(RELAY_VARS.map((key) => [key, undefined])), ...env }) + try { + return run() + } finally { + apply(saved) + } + } + + const writeBuiltUninstaller = (dir) => { + const filePath = join(dir, 'orca-windows-setup.__uninstaller.exe') + writeFileSync(filePath, 'built-by-makensis') + return filePath + } + + it.each([ + ['a missing configuration', undefined], + ['a configuration with no path', {}], + ['a non-uninstaller path', { path: 'C:\\dist\\win-unpacked\\Orca.exe' }] + ])('never throws on %s', (_label, configuration) => { + withEnv({ ORCA_WIN_UNINSTALLER_EXPORT_PATH: join(makeDir(), 'out', 'x.exe') }, () => { + expect(() => signWindowsUninstallerViaSignPath(configuration)).not.toThrow() + }) + }) + + // electron-builder calls the hook once per signing hash (sha1 then sha256), + // so both legs have to survive running twice over the same file. + it('is idempotent across the sha1 and sha256 invocations on both legs', () => { + const dir = makeDir() + const filePath = writeBuiltUninstaller(dir) + const exportPath = join(dir, 'relay', 'unsigned', 'orca-uninstaller.exe') + + withEnv({ ORCA_WIN_UNINSTALLER_EXPORT_PATH: exportPath }, () => { + signWindowsUninstallerViaSignPath({ path: filePath }) + signWindowsUninstallerViaSignPath({ path: filePath }) + }) + expect(readFileSync(exportPath, 'utf8')).toBe('built-by-makensis') + + const signedPath = join(dir, 'relay', 'signed', 'orca-uninstaller.exe') + mkdirSync(join(dir, 'relay', 'signed'), { recursive: true }) + writeFileSync(signedPath, 'signpath-signed') + + withEnv({ ORCA_WIN_UNINSTALLER_SIGNED_PATH: signedPath }, () => { + signWindowsUninstallerViaSignPath({ path: filePath }) + signWindowsUninstallerViaSignPath({ path: filePath }) + }) + expect(readFileSync(filePath, 'utf8')).toBe('signpath-signed') + expect(readFileSync(`${signedPath}${EMBEDDED_RECEIPT_SUFFIX}`, 'utf8')).toBe( + createHash('sha256').update('signpath-signed').digest('hex') + ) + }) + + // An unwritable destination is the realistic filesystem failure, and it must + // cost the uninstaller signature rather than the release job. + it('never throws when the export destination cannot be created', () => { + const dir = makeDir() + const filePath = writeBuiltUninstaller(dir) + const blocker = join(dir, 'blocker') + writeFileSync(blocker, 'not a directory') + + withEnv({ ORCA_WIN_UNINSTALLER_EXPORT_PATH: join(blocker, 'sub', 'x.exe') }, () => { + expect(() => signWindowsUninstallerViaSignPath({ path: filePath })).not.toThrow() + }) + expect(readFileSync(filePath, 'utf8')).toBe('built-by-makensis') + }) + + it('does nothing when neither relay variable is set (local Windows builds)', () => { + const dir = makeDir() + const filePath = writeBuiltUninstaller(dir) + + withEnv({}, () => { + expect(() => signWindowsUninstallerViaSignPath({ path: filePath })).not.toThrow() + }) + expect(readFileSync(filePath, 'utf8')).toBe('built-by-makensis') + }) +}) diff --git a/config/scripts/workflow-ref-mirror-case-safety.test.mjs b/config/scripts/workflow-ref-mirror-case-safety.test.mjs new file mode 100644 index 00000000000..31366f5e489 --- /dev/null +++ b/config/scripts/workflow-ref-mirror-case-safety.test.mjs @@ -0,0 +1,44 @@ +import { readFileSync } from 'node:fs' +import { join, resolve } from 'node:path' +import { describe, expect, it } from 'vitest' +import { parse } from 'yaml' + +const projectDir = resolve(import.meta.dirname, '../..') + +const readWorkflow = (relativePath) => parse(readFileSync(join(projectDir, relativePath), 'utf8')) + +// Every step that mirrors this repo's whole ref namespace onto a runner disk to +// prove a commit is reachable from a branch or tag before signing it. +const REF_MIRRORS = [ + ['.github/workflows/adhoc-mac-build.yml', 'build-adhoc-mac', 'Vet the requested ref'], + ['.github/workflows/dev-channel-win-build.yml', 'build-win', 'Vet the requested inputs'] +] + +describe('ref-mirroring vet steps', () => { + it('keeps the full-history adhoc checkout on the same case-safe backend', () => { + const steps = readWorkflow('.github/workflows/adhoc-mac-build.yml').jobs['build-adhoc-mac'] + .steps + const checkout = steps.find((step) => step.name === 'Checkout the requested ref') + expect(checkout.env.GIT_DEFAULT_REF_FORMAT).toBe('reftable') + expect(checkout.with.ref).toBe('${{ steps.vetted.outputs.sha }}') + expect(checkout.with['fetch-depth']).toBe(0) + expect(checkout.with['persist-credentials']).toBe(false) + }) + + // Why: macOS and Windows runner disks are case-insensitive, and this repo has + // branches that differ only in casing. The files backend cannot store both, and + // it fails the whole fetch rather than the one ref — so the vet step dies before + // any build runs. reftable keys refs in a table instead of file paths. + it.each(REF_MIRRORS)( + '%s creates its scratch repo with the reftable backend', + (path, job, step) => { + const run = readWorkflow(path).jobs[job].steps.find( + (candidate) => candidate.name === step + ).run + + expect(run).toContain('+refs/heads/*:refs/heads/*') + expect(run).toMatch(/git init\b[^\n]*--ref-format=reftable/) + expect(run).not.toMatch(/git init -q --bare "\$scratch"/) + } + ) +}) diff --git a/config/scripts/workflow-ref-reachability.test.mjs b/config/scripts/workflow-ref-reachability.test.mjs new file mode 100644 index 00000000000..d71c3094c56 --- /dev/null +++ b/config/scripts/workflow-ref-reachability.test.mjs @@ -0,0 +1,125 @@ +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { pathToFileURL } from 'node:url' +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { parse } from 'yaml' +import { runProcess } from '../../src/shared/child-process/run-process' + +const readWorkflow = (name) => parse(readFileSync(`.github/workflows/${name}.yml`, 'utf8')) +const windowsVet = readWorkflow('dev-channel-win-build').jobs['build-win'].steps.find( + (step) => step.id === 'vetted' +) +const macSteps = readWorkflow('adhoc-mac-build').jobs['build-adhoc-mac'].steps +const macVet = macSteps.find((step) => step.id === 'vetted') +const macCheckout = macSteps.find((step) => step.name === 'Checkout the requested ref') +const directory = mkdtempSync(join(tmpdir(), 'workflow-ref-reachability-')) +const repository = join(directory, 'remote.git') +const identity = { + ...process.env, + GIT_AUTHOR_NAME: 'Ref test', + GIT_AUTHOR_EMAIL: 'ref-test@example.com', + GIT_COMMITTER_NAME: 'Ref test', + GIT_COMMITTER_EMAIL: 'ref-test@example.com' +} +let ancestor, upper, lower, untrusted + +async function git(args, env = identity) { + const result = await runProcess({ program: 'git', args, env }) + expect(result.code, result.stderr).toBe(0) + return result.stdout.trim() +} + +beforeAll(async () => { + await git(['init', '--bare', '--ref-format=reftable', repository]) + const tree = await git(['-C', repository, 'mktree']) + ancestor = await git(['-C', repository, 'commit-tree', tree, '-m', 'ancestor']) + upper = await git(['-C', repository, 'commit-tree', tree, '-p', ancestor, '-m', 'upper']) + lower = await git(['-C', repository, 'commit-tree', tree, '-p', ancestor, '-m', 'lower']) + untrusted = await git(['-C', repository, 'commit-tree', tree, '-m', 'PR only']) + for (const [ref, sha] of [ + ['refs/heads/Fix', upper], + ['refs/heads/fix', lower], + ['refs/pull/1/head', untrusted] + ]) { + await git(['-C', repository, 'update-ref', ref, sha]) + } + await git(['-C', repository, 'tag', '-a', 'Release', upper, '-m', 'upper tag']) + await git(['-C', repository, 'tag', '-a', 'release', lower, '-m', 'lower tag']) + await git(['-C', repository, 'config', 'uploadpack.allowFilter', 'true']) +}) + +afterAll(() => rmSync(directory, { recursive: true, force: true })) + +async function vet(step, ref) { + const scratch = mkdtempSync(join(directory, 'attempt-')) + const script = join(scratch, 'vet.sh') + writeFileSync(script, step.run) + return runProcess({ + program: 'bash', + args: [script], + env: { + ...identity, + REPO_URL: pathToFileURL(repository).href, + RUNNER_TEMP: scratch, + GITHUB_OUTPUT: join(scratch, 'output'), + REQUESTED_REF: ref, + REQUESTED_SHA: ref, + CHANNEL: 'hourly', + TAG: 'v1.0.0-hourly.test', + VERSION: '1.0.0-hourly.test' + } + }) +} + +describe('release ref trust with case-twin names', () => { + it('accepts both branch tips, annotated tags, and their common ancestor', async () => { + for (const sha of [upper, lower, ancestor]) { + const result = await vet(windowsVet, sha) + expect(result.code, result.stderr).toBe(0) + } + for (const ref of ['Fix', 'fix', 'Release', 'release', ancestor]) { + const result = await vet(macVet, ref) + expect(result.code, result.stderr).toBe(0) + } + }) + + it('rejects PR-only commits even when the server has their objects', async () => { + for (const step of [windowsVet, macVet]) { + const result = await vet(step, untrusted) + expect(result.code).not.toBe(0) + expect(result.stdout).toContain('not reachable from any branch or tag') + } + const result = await vet(macVet, 'refs/pull/1/head') + expect(result.code).not.toBe(0) + expect(result.stdout).toContain('Refusing to build PR ref') + }) + + it('preserves both case variants in the subsequent full-history checkout', async () => { + const checkout = join(directory, 'checkout') + const env = { ...identity, ...macCheckout.env } + await git(['init', checkout], env) + await git( + [ + '-C', + checkout, + 'fetch', + '--no-tags', + repository, + '+refs/heads/*:refs/remotes/origin/*', + '+refs/tags/*:refs/tags/*' + ], + env + ) + await git(['-C', checkout, 'checkout', '--detach', upper], env) + for (const [ref, sha] of [ + ['refs/remotes/origin/Fix', upper], + ['refs/remotes/origin/fix', lower], + ['refs/tags/Release', upper], + ['refs/tags/release', lower] + ]) { + expect(await git(['-C', checkout, 'rev-parse', `${ref}^{commit}`], env)).toBe(sha) + } + expect(await git(['-C', checkout, 'rev-parse', 'HEAD'], env)).toBe(upper) + }) +}) diff --git a/config/tsconfig.cli.json b/config/tsconfig.cli.json index b8c40c7c8aa..80cf4a511f2 100644 --- a/config/tsconfig.cli.json +++ b/config/tsconfig.cli.json @@ -16,6 +16,7 @@ "../src/main/agent-hooks/managed-hook-script-refresh.ts", "../src/main/agent-hooks/posix-hook-command.ts", "../src/main/agent-hooks/runtime-home-hook-command.ts", + "../src/main/agent-hooks/windows-direct-cmd-hook-command.ts", "../src/main/agent-hooks/windows-powershell-hook-launcher.ts", "../src/main/amp/agent-status-plugin-source.ts", "../src/main/amp/hook-service.ts", @@ -118,6 +119,7 @@ "../src/main/hermes/hermes-home-filesystem.ts", "../src/main/hermes/hermes-managed-plugin-source.ts", "../src/main/hermes/hook-service.ts", + "../src/main/git-bash.ts", "../src/main/in-flight-run-dedupe.ts", "../src/main/kimi/hook-service.ts", "../src/main/kimi/kimi-hook-config-toml.ts", diff --git a/config/vitest.performance.config.ts b/config/vitest.performance.config.ts new file mode 100644 index 00000000000..7682b7b9698 --- /dev/null +++ b/config/vitest.performance.config.ts @@ -0,0 +1,33 @@ +import { existsSync } from 'node:fs' +import { resolve } from 'node:path' +import { defineConfig } from 'vitest/config' +import baseConfig from './vitest.config' + +const contracts = [ + 'src/main/sqlite/sync-database.test.ts', + 'src/main/runtime/orchestration/db/row-column-lists.test.ts', + 'src/relay/fs-path-metadata-symlink-concurrency.test.ts', + 'src/renderer/src/components/editor/rich-markdown-list-tokenizers.test.ts', + 'src/renderer/src/components/editor/rich-markdown-lowlight-cache.test.ts', + 'src/renderer/src/components/terminal-pane/agent-completion-coordinator-queued-inspection-disposal.test.ts', + 'src/renderer/src/lib/pane-manager/pane-terminal-output-scheduler-queue-retention.test.ts', + 'config/scripts/app-store-performance-plugin.test.mjs', + 'config/scripts/quadratic-buffer-concat-plugin.test.mjs', + 'config/scripts/sort-comparator-performance-plugin.test.mjs' +] + +for (const contract of contracts) { + if (!existsSync(resolve(contract))) { + throw new Error(`Missing performance contract: ${contract}`) + } +} + +export default defineConfig({ + ...baseConfig, + test: { + ...baseConfig.test, + include: contracts, + fileParallelism: false, + retry: 0 + } +}) diff --git a/docs/assets/readme-downloads.svg b/docs/assets/readme-downloads.svg index ef8ebb61bb4..33ad276aa2d 100644 --- a/docs/assets/readme-downloads.svg +++ b/docs/assets/readme-downloads.svg @@ -1,5 +1,5 @@ - - downloads: 38m + + downloads: 41m @@ -15,7 +15,7 @@ downloads downloads - 38m - 38m + 41m + 41m diff --git a/docs/assets/wechat-qr-group9.jpg b/docs/assets/wechat-qr-group9.jpg new file mode 100644 index 00000000000..2bf46a28c3d Binary files /dev/null and b/docs/assets/wechat-qr-group9.jpg differ diff --git a/docs/readme/README.fr.md b/docs/readme/README.fr.md index 97c78d4e713..e601abc2344 100644 --- a/docs/readme/README.fr.md +++ b/docs/readme/README.fr.md @@ -243,9 +243,9 @@ Associez-la à l'app de bureau pour surveiller et piloter vos agents depuis votr - **Discord :** Rejoignez la communauté sur **[Discord](https://discord.gg/fzjDKHxv8Q)**. - **Twitter / X :** Suivez **[@orca_build](https://x.com/orca_build)** pour les news et annonces. -- **WeChat :** Scannez pour rejoindre le groupe WeChat 8 de la communauté Orca. +- **WeChat :** Scannez pour rejoindre le groupe WeChat 8 de la communauté Orca. Le groupe 8 est peut-être complet ; dans ce cas, scannez plutôt le QR code du groupe 9. - QR code WeChat groupe 8 de la communauté Orca + QR code WeChat groupe 8 de la communauté Orca  QR code WeChat groupe 9 de la communauté Orca - **Feedback & idées :** On ship vite. Il manque quelque chose ? [Demandez une feature](https://github.com/stablyai/orca/issues). - **Confidentialité :** Voir la [doc confidentialité & télémétrie](https://www.onorca.dev/docs/telemetry) pour ce qu'Orca collecte en anonyme et comment désactiver la télémétrie. diff --git a/docs/readme/README.ko.md b/docs/readme/README.ko.md index 4a75722ff8c..837ecf2133f 100644 --- a/docs/readme/README.ko.md +++ b/docs/readme/README.ko.md @@ -238,9 +238,9 @@ yay -S stably-orca-bin - **Discord:** **[Discord](https://discord.gg/fzjDKHxv8Q)** 커뮤니티에 참여하세요. - **Twitter / X:** 업데이트와 공지는 **[@orca_build](https://x.com/orca_build)** 를 팔로우하세요. -- **WeChat:** QR 코드를 스캔해 Orca 커뮤니티 WeChat 그룹 8에 참여하세요. +- **WeChat:** QR 코드를 스캔해 Orca 커뮤니티 WeChat 그룹 8에 참여하세요. 그룹 8이 가득 찼을 수 있으니, 그런 경우 그룹 9 QR 코드를 스캔하세요. - Orca 커뮤니티 WeChat 그룹 8 QR 코드 + Orca 커뮤니티 WeChat 그룹 8 QR 코드  Orca 커뮤니티 WeChat 그룹 9 QR 코드 - **피드백과 아이디어:** 우리는 빠르게 출시합니다. 필요한 기능이 있나요? [새 기능을 요청](https://github.com/stablyai/orca/issues)하세요. - **개인정보 보호:** Orca가 수집하는 익명 사용 데이터와 수집 거부 방법은 [개인정보 및 텔레메트리 문서](https://www.onorca.dev/docs/telemetry)를 참고하세요. diff --git a/docs/readme/README.zh-CN.md b/docs/readme/README.zh-CN.md index d7bae3fba9e..10f47e20fe6 100644 --- a/docs/readme/README.zh-CN.md +++ b/docs/readme/README.zh-CN.md @@ -235,9 +235,9 @@ yay -S stably-orca-bin - **Discord:** 加入 **[Discord](https://discord.gg/fzjDKHxv8Q)** 社区。 - **Twitter / X:** 关注 **[@orca_build](https://x.com/orca_build)** 获取更新和公告。 -- **微信:** 扫码加入 Orca 社区微信第 8 群。 +- **微信:** 扫码加入 Orca 社区微信第 8 群。第 8 群可能已满,如遇这种情况请扫描第 9 群二维码。 - Orca 社区微信第 8 群二维码 + Orca 社区微信第 8 群二维码  Orca 社区微信第 9 群二维码 - **反馈与想法:** 我们发布很快。缺少什么功能?[提交功能请求](https://github.com/stablyai/orca/issues)。 - **隐私:** 查看[隐私与遥测文档](https://www.onorca.dev/docs/telemetry),了解 Orca 收集哪些匿名使用数据以及如何退出。 diff --git a/docs/reference/ci-runner-efficiency.md b/docs/reference/ci-runner-efficiency.md new file mode 100644 index 00000000000..6d688598097 --- /dev/null +++ b/docs/reference/ci-runner-efficiency.md @@ -0,0 +1,199 @@ +# CI efficiency and runner capacity + +Audit date: September 5, 2026. No paid capacity or provider configuration changed. + +## Measurements and changes + +Three recent successful PR runs used 54.6–64.9 aggregate runner minutes: +[33998366568](https://github.com/stablyai/orca/actions/runs/33998366568), +[33998220287](https://github.com/stablyai/orca/actions/runs/33998220287), and +[33998181502](https://github.com/stablyai/orca/actions/runs/33998181502). +These are sums of active job durations, excluding skipped jobs; they are not +billing minutes or queue time. This small sample is not a historical average. + +- Consolidate E2E routing into the existing code-path detector. The removed + detector occupied 20–22 seconds and required another runner allocation and + full-history checkout per nondraft code PR. The same routing commands remain, + including SSH and native IME selection; actual E2E results remain advisory. + A routing-script error now fails the required code-path detector. +- Use gzip for PR-only Debian/RPM artifacts. The two sampled Linux packaging + jobs took 8m10s and 8m19s overall; one spent 3m47s in electron-builder. Its + default Debian/RPM compression is xz. PR artifacts are inspected on the same + runner, so their download size offers no benefit. Keep all AppImage, Debian, + RPM, payload, launcher, and shutdown checks. Release compression is unchanged. + Hosted validation in [33999422341](https://github.com/stablyai/orca/actions/runs/33999422341) + reduced the package-build step to 2m13s and the full Linux job to 6m17s, with + all existing checks passing. This is a small observational sample. +- Cancel superseded Mobile Checks and Skill update round-trip PR runs. The + skill matrix has 13 jobs. Preserve non-cancelling main/merge-group skill runs, + with separate concurrency groups per event. +- Reuse the existing script-free root dependency action in Mobile Checks, + including the pnpm cache keyed by both root and mobile lockfiles. The root + install remains necessary because mobile types import root dependencies. + +The repository already has eight unit shards, path-scoped platform checks, +native caches, one shared E2E build, PR cancellation, incremental TypeScript +caching, and changed-spec E2E routing. Increasing shards would increase setup +work and simultaneous runner demand. Do not adjust the count without comparing +critical-path time and aggregate job time on the same commit. + +## Follow-up savings + +- Move the hourly main/release freshness lookup to a five-minute Ubuntu + preflight without a checkout. In unchanged run + [33986205749](https://github.com/stablyai/orca/actions/runs/33986205749), + Blacksmith macOS was occupied for 40 seconds, including a 30-second checkout, + before skipping. The new job-level gate avoids that Mac allocation. Actual + builds gain an Ubuntu scheduling hop; pin the Mac checkout and downstream + Windows identity to the SHA that the preflight checked. +- Avoid global `npm install -g node-gyp` for validated Linux Node-runtime cache + hits. Use the existing native-module load/provenance check before skipping; + misses, broken addons, and Electron jobs still install the rebuild toolchain. + The action file participates in cache keys, so this rollout creates fresh + native caches once. No measured warm-cache seconds are claimed yet. + +## Runner recommendations + +The repository is **public**, verified using the GitHub API. Standard +GitHub-hosted Linux, Windows, and macOS runners have free compute minutes for +public repositories. Queue pressure and third-party provider allowances still +matter; artifact storage and larger runners have separate billing rules. +See [GitHub Actions billing](https://docs.github.com/en/billing/concepts/product-billing/github-actions). + +1. Keep standard GitHub-hosted runners as the default. Ask GitHub Support for a + higher concurrent-job limit before paying for more capacity. The documented + standard limits depend on the account plan (Free: 20 total/5 macOS; Team: + 60/5; Enterprise: 500/50), and increases are subject to approval. The actual + account entitlement was not verified. See [limits](https://docs.github.com/en/actions/reference/limits). +2. Reserve existing Blacksmith allowance for macOS if that is the priority. + Blacksmith documents 3,000 free x64 2-vCPU-equivalent minutes per organization; + a 6-vCPU Mac minute consumes 20 equivalents, or 150 actual Mac minutes if + it uses the entire free pool. Cloud workflows also use Blacksmith Linux. + Moving Linux to hosted GitHub saves shared allowance, but does not necessarily + free Mac hardware capacity. Account-specific contracts and usage were not + inspected. See [Blacksmith runners](https://docs.blacksmith.sh/blacksmith-runners/overview). +3. Treat Ubicloud as an optional small Linux overflow trial. Its documented + $2.50 monthly credit buys 1,250 premium 2-vCPU minutes at $0.002/minute, or + 2,000 standard 2-vCPU minutes at $0.00125/minute. New accounts default to + premium and require a credit card. No enforceable hard spending cap was + verified, so changing runner labels cannot guarantee the no-spend constraint. + One PR's roughly 55–65 runner minutes also makes clear how small this pool + is relative to repository activity (hardware speeds differ). + See [pricing](https://ubicloud.com/docs/about/pricing) and + [setup](https://ubicloud.com/docs/github-actions-integration/quickstart). + +### A bounded Ubicloud candidate + +The Linux leg of `performance-contracts.yml` took 48 seconds in +[33994756657](https://github.com/stablyai/orca/actions/runs/33994756657). +Its daily schedule and 20-minute timeout make it a small candidate: 31 ordinary +scheduled attempts permit at most 620 job-runtime minutes, before runner +startup/cleanup billing. Actual timings on Ubicloud's 2-vCPU hardware still need +measurement; the GitHub timing is only a sizing reference. + +If enabled later, route only the first attempt of the scheduled Linux job to +Ubicloud; keep PRs, manual dispatches, reruns, and macOS/Windows on GitHub. This +avoids spending the allowance on unpredictable PR volume. Check other account +usage and available credit before enabling; a workflow timeout is not an +account-wide billing cap. On September 5, the organization's GitHub App +installation list contained Blacksmith but no Ubicloud installation, so this +follow-up leaves runner selection on GitHub rather than queueing work against +an unprovisioned label. + +## Machines that also run coding agents + +Do not register the credentialed host directly as a public-PR runner. A PR can +execute arbitrary build/test code, and a persistent host lets it access local +credentials or affect subsequent jobs. Docker alone is not adequate isolation +when it exposes the host home, Docker socket, SSH agent, or office network. + +A possible no-new-hardware experiment is a disposable VM per job, preferably on +a dedicated spare machine, with a just-in-time single-job runner, no shared +home/keychain/SSH agent or host mounts, restricted network access, and CPU/RAM +limits that leave room for coding agents. Destroy the VM after every job; +ephemeral runner registration by itself does not clean the machine. Start with +trusted branch/manual workloads and keep public fork PRs on hosted runners. +Provisioning and ongoing patching are real operational costs even when the +machine is already owned. See GitHub's +[self-hosted runner security guidance](https://docs.github.com/en/actions/security-for-github-actions/security-guides/security-hardening-for-github-actions). + +## Release waits + +The latest successful sampled Windows release used 13m59s of a 21m56s job in +signing wait/download steps. The same release held an Ubuntu job for 11m38s +polling the isolated Mac build. These are stronger occupancy opportunities than +small checkout savings, especially when approval takes hours. + +[Windows signing without occupying a runner](windows-signing-runner-time.md) +describes a staged, same-run design, required protected environments, and +rehearsal criteria. No callback integration or protected Windows signing +environments currently exist. An environment-gated design adds a GitHub +approval after each SignPath approval and changes the current automatic inner +signing timeout fallback; those are explicit release-policy decisions, so this +PR leaves production signing behavior unchanged. + +## Second audit and hosted trials + +- Cloud Verify ran 100 times in a sampled 39-hour window (84 PR and 16 push + runs). Move its four Ubuntu 22.04 jobs from Blacksmith to standard hosted + Ubuntu 22.04, preserving Postgres, secret scanning, build, tests, and Terraform + validation. Baseline [34001538145](https://github.com/stablyai/orca/actions/runs/34001538145) + used 64/72/26/19 seconds for security/test/build/Terraform respectively. + This conserves the shared provider allowance; hosted latency must be checked. +- Keep full tag history for the 13-job skill round-trip matrix, but fetch blobs + lazily. Only two historical SKILL.md files are materialized. Baseline + [33999994876](https://github.com/stablyai/orca/actions/runs/33999994876) + spent 42–84 seconds per checkout, about 14 aggregate runner minutes. A hosted + trial must verify historical blob fetches on all three operating systems. +- Use the existing Electron/native dependency cache for native IME CI. Keep + both deterministic boundary and real IBus tests. Add pnpm store caching to + terminal perf and release golden/evidence lanes; retain their raw installs + because manually selected older refs may not contain the shared action. +- Disable ZIP recompression only for already-compressed NSIS installers sent + to SignPath. Installer contents, release compression, and signing stay intact. +- Advance existing placement and startup deadlines with scoped fake timers in + three renderer test files. All 34 tests pass in 62 ms of local test execution, + versus 65.182 seconds in the sampled hosted baseline. Imports and transforms + still dominate invocation time; this is not a claim of equal PR wall savings. + +Eight unit shards already have balanced 260–296-second sample durations. +Reducing shards or removing test isolation lacks evidence of a net gain. Real +subprocess tests intentionally cover lifecycle behavior and retain real clocks. +The 14-way E2E split retains headroom after earlier 12-way timeouts. Lowering +coverage or schedule frequency is outside this efficiency pass. Cache complexity +for a seven-second docs install is unlikely to pay back. Release build reuse +across modes risks differing telemetry identities and native platform artifacts. + +Terminal Perf's baseline [33955846492](https://github.com/stablyai/orca/actions/runs/33955846492) +failed waiting 30 seconds for workspaceSessionReady in its shared-page fixture, +before measuring terminal performance. Compare hosted trials against that known +failure rather than attributing it to dependency cache changes. + +Hosted trials for the second audit: + +- [Cloud Verify 34002295216](https://github.com/stablyai/orca/actions/runs/34002295216) + passed all four jobs on standard hosted Ubuntu: security 57s, test 102s, build + 35s, Terraform 19s. The test lane is 30s slower than the Blacksmith sample; + retain this modest latency tradeoff to conserve shared allowance. +- [Skill matrix 34002295221](https://github.com/stablyai/orca/actions/runs/34002295221) + passed all 13 legs, including historical blob materialization. Checkout took + 18–20s on Linux, 39–45s on macOS, and 49–58s on Windows, versus the earlier + 42–84s range across platforms. These are observational samples. +- [Native IME 34002299594](https://github.com/stablyai/orca/actions/runs/34002299594) + passed both deterministic and real IBus checks. Shared dependency setup took + 29s, versus 35s for the old install/toolchain steps in the sampled baseline. +- Native-IME-only source/spec changes no longer allocate the reusable E2E + build, cache, and consumer jobs just to filter out the native spec. The + separate native workflow still runs; SSH-only and mixed spec lists still + allocate the reusable workflow. Routing contracts exercise these cases. +- [Hourly 34001816449](https://github.com/stablyai/orca/actions/runs/34001816449) + exercised the new five-second preflight and successfully published macOS. + The Windows follow-up failed in its unchanged input-vetting fetch because + remote refs differ only by case on its case-insensitive filesystem. The + requested SHA was correct; this does not validate an unchanged-main skip yet. + +Moving the daily Mac freshness check has lower expected value than hourly: +only one potential idle allocation per day, and active development usually +requires that build. Defer another release-graph change until skip frequency +justifies it. The substantive remaining release occupancy opportunity is the +separately documented asynchronous signing policy decision. diff --git a/docs/reference/windows-cmd-shim-resolution.md b/docs/reference/windows-cmd-shim-resolution.md new file mode 100644 index 00000000000..c17380e800d --- /dev/null +++ b/docs/reference/windows-cmd-shim-resolution.md @@ -0,0 +1,77 @@ +# Resolving Windows `.cmd` shims past cmd.exe + +Node refuses to spawn a `.cmd`/`.bat` target without a shell (the +CVE-2024-27980 mitigation), so `resolveSpawn` has to make `cmd.exe` the program +and hand it `/d /v:off /s /c ""`. For an agent CLI that +means a long `cmd.exe /c` line whose caret-escaped payload is natural-language +prompt text — which Microsoft Defender for Endpoint's command-line model scores +as obfuscation. `codex.cmd` appeared in the spawn cluster of an MDE incident +against Orca for exactly this reason. + +`src/shared/child-process/windows-cmd-shim-resolution.ts` sidesteps it. npm's +`cmd-shim` and pnpm's `@zkochan/cmd-shim` generate files whose entire body is +"find a Node interpreter and run this script". Reading one lets `resolveSpawn` +spawn `node.exe - - - `) - }) - await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve)) - const port = (server.address() as AddressInfo).port - return { - sourceUrl: `http://127.0.0.1:${port}/source`, - close: () => closeServer(server) - } -} - async function startBrowserWindowCloseServer(): Promise<{ url: string sourceUrl: string @@ -281,8 +204,8 @@ async function clickBrowserLink( browserTabId: string, selector: string, options: { - modifiers?: ('meta' | 'control')[] - button?: 'left' | 'middle' + modifiers?: ('meta' | 'control' | 'shift')[] + button?: 'left' | 'middle' | 'right' frameSelector?: string } = {} ): Promise { @@ -317,21 +240,31 @@ async function clickBrowserLink( if (!point) { throw new Error(`Missing browser link ${targetSelector}`) } - await webview.sendInputEvent({ type: 'mouseMove', modifiers: inputModifiers, ...point }) - await webview.sendInputEvent({ - type: 'mouseDown', - button, - clickCount: 1, - modifiers: inputModifiers, - ...point - }) - await webview.sendInputEvent({ - type: 'mouseUp', - button, - clickCount: 1, - modifiers: inputModifiers, - ...point - }) + const holdShift = inputModifiers.includes('shift') + if (holdShift) { + await webview.sendInputEvent({ type: 'keyDown', keyCode: 'Shift', modifiers: ['shift'] }) + } + try { + await webview.sendInputEvent({ type: 'mouseMove', modifiers: inputModifiers, ...point }) + await webview.sendInputEvent({ + type: 'mouseDown', + button, + clickCount: 1, + modifiers: inputModifiers, + ...point + }) + await webview.sendInputEvent({ + type: 'mouseUp', + button, + clickCount: 1, + modifiers: inputModifiers, + ...point + }) + } finally { + if (holdShift) { + await webview.sendInputEvent({ type: 'keyUp', keyCode: 'Shift' }) + } + } }, { targetBrowserTabId: browserTabId, @@ -343,21 +276,43 @@ async function clickBrowserLink( ) } -async function expectBrowserTabActive( +async function waitForTabIdByExactTitle( page: Parameters[0], title: string -): Promise { +): Promise { const resolveTabId = (): Promise => page.locator('[data-tab-id]').evaluateAll((tabs, exactTitle) => { const tab = tabs.find((candidate) => candidate.textContent?.trim() === exactTitle) return tab?.getAttribute('data-tab-id') ?? null }, title) await expect.poll(resolveTabId, { timeout: 10_000 }).not.toBeNull() - const tabId = await resolveTabId() - expect(tabId).toBeTruthy() + return (await resolveTabId()) as string +} + +async function expectBrowserTabActive( + page: Parameters[0], + title: string +): Promise { + const tabId = await waitForTabIdByExactTitle(page, title) await expect(page.locator(`[data-browser-overlay-tab-id="${tabId}"]`)).toHaveCSS('opacity', '1') } +async function expectBrowserTabOpenedInBackground( + page: Parameters[0], + sourceTabId: string, + title: string +): Promise { + const openedTabId = await waitForTabIdByExactTitle(page, title) + await expect(page.locator(`[data-browser-overlay-tab-id="${sourceTabId}"]`)).toHaveCSS( + 'opacity', + '1' + ) + await expect(page.locator(`[data-browser-overlay-tab-id="${openedTabId}"]`)).toHaveCSS( + 'opacity', + '0' + ) +} + async function readBrowserInputValue( page: Parameters[0], browserTabId: string @@ -680,7 +635,7 @@ test.describe('Browser Tab', () => { } }) - test('every new-tab link gesture activates an Orca tab and never a native window', async ({ + test('new-tab link gestures follow Chrome foreground and background behavior', async ({ electronApp, orcaPage }) => { @@ -698,38 +653,51 @@ test.describe('Browser Tab', () => { const baseWindowCount = await electronApp.evaluate( ({ BaseWindow }) => BaseWindow.getAllWindows().length ) - // A plain target=_blank click is a new-tab request, in the main frame and in an iframe; - // the source tab must stay put rather than navigate away under it. + // A plain main-frame target=_blank click must not navigate the source tab away. const sourceTabLocator = orcaPage.locator(`[data-tab-id="${sourceTab!.id}"]`) - await clickBrowserLink(orcaPage, sourceTab!.id, '#external-link') - await expectBrowserTabActive(orcaPage, 'Linked destination') + await clickBrowserLink(orcaPage, sourceTab!.id, '#blank-link') + await expectBrowserTabActive(orcaPage, 'Blank target destination') await expect(sourceTabLocator).toContainText('Source page') await switchToBrowserTab(orcaPage, worktreeId, sourceTab!.id) + // Context-menu links keep the source visible until the new tab is selected. + await clickBrowserLink(orcaPage, sourceTab!.id, '#external-link', { button: 'right' }) + await orcaPage + .getByRole('menuitem', { name: 'Open Link In Orca Browser', exact: true }) + .click() + await expectBrowserTabOpenedInBackground(orcaPage, sourceTab!.id, 'Linked destination') await clickBrowserLink(orcaPage, sourceTab!.id, '#frame-link', { frameSelector: '#link-frame' }) await expectBrowserTabActive(orcaPage, 'Frame destination') - await expect(sourceTabLocator).toContainText('Source page') await switchToBrowserTab(orcaPage, worktreeId, sourceTab!.id) await clickBrowserLink(orcaPage, sourceTab!.id, '#frame-modifier-link', { frameSelector: '#link-frame', modifiers: process.platform === 'darwin' ? ['meta'] : ['control'] }) - await expectBrowserTabActive(orcaPage, 'Frame modifier destination') - await switchToBrowserTab(orcaPage, worktreeId, sourceTab!.id) + await expectBrowserTabOpenedInBackground( + orcaPage, + sourceTab!.id, + 'Frame modifier destination' + ) await clickBrowserLink(orcaPage, sourceTab!.id, '#frame-middle-link', { button: 'middle', frameSelector: '#link-frame' }) - await expectBrowserTabActive(orcaPage, 'Frame middle destination') - await switchToBrowserTab(orcaPage, worktreeId, sourceTab!.id) + await expectBrowserTabOpenedInBackground(orcaPage, sourceTab!.id, 'Frame middle destination') await clickBrowserLink(orcaPage, sourceTab!.id, '#modifier-link', { modifiers: process.platform === 'darwin' ? ['meta'] : ['control'] }) - await expectBrowserTabActive(orcaPage, 'Modifier destination') + await expectBrowserTabOpenedInBackground(orcaPage, sourceTab!.id, 'Modifier destination') + + await clickBrowserLink(orcaPage, sourceTab!.id, '#frame-shift-middle-link', { + button: 'middle', + modifiers: ['shift'], + frameSelector: '#link-frame' + }) + await expectBrowserTabActive(orcaPage, 'Frame shift middle destination') await switchToBrowserTab(orcaPage, worktreeId, sourceTab!.id) const tabCountBeforeCancelledClick = await orcaPage.locator('[data-tab-id]').count() @@ -740,7 +708,7 @@ test.describe('Browser Tab', () => { await expect(orcaPage.locator('[data-tab-id]')).toHaveCount(tabCountBeforeCancelledClick) await clickBrowserLink(orcaPage, sourceTab!.id, '#middle-link', { button: 'middle' }) - await expectBrowserTabActive(orcaPage, 'Middle-click destination') + await expectBrowserTabOpenedInBackground(orcaPage, sourceTab!.id, 'Middle-click destination') await expect .poll(() => electronApp.evaluate(({ BaseWindow }) => BaseWindow.getAllWindows().length), { timeout: 5_000 diff --git a/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts b/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts index d79c99b679d..8a407a29d15 100644 --- a/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts +++ b/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts @@ -2,9 +2,14 @@ // way the terminal wire harness is: current code against a real published release. // // Three skews matter here, and none can be checked from one build alone — an old -// client must not be shown a session it cannot render, a new client must find an -// old host's missing surface cleanly, and a client's cursor must survive the host -// process that minted it. +// client must not receive a journal-backed RPC surface it cannot read, a new client +// must find an old host's missing surface cleanly, and a client's cursor must survive +// the host process that minted it. +// +// The session-tabs projection may keep a metadata-only row for an incapable mobile client so the +// chat is not simply absent on the phone. Every `agentSession.*` method and destructive close stays +// refused, which is what the tests below pin; the row-level behaviour is pinned in +// src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts. import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' @@ -18,7 +23,10 @@ import { setStructuredAgentSessionHost } from '../../../src/main/native-chat/age import { AgentSessionRecordStore } from '../../../src/main/runtime/agent-session-record-store' import { computeAgentSessionPayloadFingerprint } from '../../../src/shared/agent-session-mutation-envelope' import type { AgentSessionSubscribeEvent } from '../../../src/shared/agent-session-wire' -import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../src/shared/protocol-version' +import { + AGENT_SESSION_STATUS_FEED_RUNTIME_CAPABILITY, + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY +} from '../../../src/shared/protocol-version' import { resolveBaselineReleaseRef } from './release-checkout' import { loadAgentSessionWireBuild, @@ -36,6 +44,7 @@ const WORKSPACE = 'workspace-1' const THREAD = '019fd532-7c11-7a90-b6de-4e1a2c3d5f60' const NOW = 1_800_000_000_000 const CLIENT_CAPABILITY_UPDATE_METHOD = 'runtime.clientCapabilities.update' +const STATUS_FEED_METHOD = 'agentSession.subscribeStatus' /** Every method the structured surface publishes: the host method it must reach, * and the result it must hand back. A gate that hides one method and leaks @@ -77,6 +86,11 @@ const STRUCTURED_CALLS: { hostMethod: 'setOption', result: { ok: true, replayed: false } }, + { + method: 'agentSession.requestHandoff', + hostMethod: 'requestHandoff', + result: { status: { owner: 'native' } } + }, { method: 'agentSession.handoffStatus', hostMethod: 'handoffStatus', @@ -87,6 +101,11 @@ const STRUCTURED_CALLS: { hostMethod: 'readOptions', result: { current: { model: 'gpt-live' } } }, + { + method: 'agentSession.reveal', + hostMethod: 'revealSession', + result: { ok: true, sessionId: SESSION, workspaceId: WORKSPACE, agent: 'codex', readable: true } + }, { method: 'agentSession.hold', hostMethod: 'hold', result: { held: true } }, { method: 'agentSession.release', hostMethod: 'release', result: { released: true } }, { @@ -97,6 +116,12 @@ const STRUCTURED_CALLS: { // A subscription that opens with nothing to say answers with no reply at all, // so reaching the host is the only signal that the gate opened. { method: 'agentSession.subscribe', hostMethod: 'subscribe' }, + // The status feed opens with a snapshot of every session, so its first reply is the contract. + { + method: STATUS_FEED_METHOD, + hostMethod: 'subscribeStatus', + result: { type: 'snapshot', sessions: [] } + }, // Teardown runs through the runtime's subscription registry rather than the // host, so its reply is the only signal that the gate opened. { method: 'agentSession.unsubscribe', hostMethod: null, result: { unsubscribed: true } } @@ -197,6 +222,14 @@ function paramsFor(method: string): unknown { const fields = { itemId: 'item-1', expectedRevision: 1, optionId: 'allow' } return { envelope: envelope({ method, fields, fence }), ...fields } } + case 'agentSession.requestHandoff': { + const fields = { + direction: 'to-tui' as const, + mode: 'now' as const, + action: 'start' as const + } + return { envelope: envelope({ method, fields, fence }), ...fields } + } case 'agentSession.setOption': { const fields = { key: 'model', value: 'gpt-5' } return { envelope: envelope({ method, fields, fence }), ...fields } @@ -286,9 +319,19 @@ async function callBuild( function structuredHostStub(): Record> { return { attach: vi.fn(async () => ({ ok: true, replayed: false, value: { sessionId: SESSION } })), + // Attach-shaped entries take a client-supplied location, so the host is asked whether it + // supports creating there. A real host always answers; leaving it unstubbed made every + // `ensure` refuse for the harness's own reason rather than the location's. + supportsCreate: vi.fn(() => true), send: vi.fn(async () => ({ ok: true, replayed: false })), cancel: vi.fn(async () => ({ ok: true, replayed: false })), close: vi.fn(async () => undefined), + revealSession: vi.fn(async () => ({ + sessionId: SESSION, + workspaceId: WORKSPACE, + agent: 'codex' as const, + readable: true + })), hold: vi.fn(async () => undefined), release: vi.fn(() => undefined), respondToPrompt: vi.fn(async () => ({ ok: true, replayed: false })), @@ -298,6 +341,10 @@ function structuredHostStub(): Record> { readOptions: vi.fn(async () => ({ models: [], current: { model: 'gpt-live' } })), history: vi.fn(() => ({ ok: true, page: { items: [] } })), subscribe: vi.fn(() => () => undefined), + subscribeStatus: vi.fn((subscriber: { emit: (event: unknown) => void }) => { + subscriber.emit({ type: 'snapshot', sessions: [] }) + return () => undefined + }), unsubscribe: vi.fn() } } @@ -421,6 +468,14 @@ describe('cross-version structured agent sessions', () => { expect(baseline.capabilities.includes(STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY)).toBe( baselineStructuredMethods().length > 0 ) + // The status feed is additive to a surface that already shipped, so it carries its own + // capability or a client cannot tell "host too old" from "the call failed" — and it + // would relay-retry a method_not_found forever instead of degrading once. + for (const build of [current, baseline]) { + expect(build.capabilities.includes(AGENT_SESSION_STATUS_FEED_RUNTIME_CAPABILITY)).toBe( + build.methodNames.includes(STATUS_FEED_METHOD) + ) + } // Additive surface: bumping the protocol number would strand every paired // device on this release rather than degrade one feature. expect(current.protocolVersion).toBe(baseline.protocolVersion) @@ -723,6 +778,10 @@ describe('cross-version structured agent sessions', () => { /** Phase 2 owns provider processes; the adapter is the only stub here. */ function adapter(): StructuredAgentSessionAdapter { return { + // Every real adapter answers this; without it adapterSupportsCreate falls through to + // `supportsLocation`, which this fake also lacks, so the client-supplied-location gate + // refused for the fake's silence rather than for the location. + supportsCreate: () => true, acquire: async ({ fence }) => ({ process: { hostId: 'local', diff --git a/tests/e2e/cross-version-wire/release-checkout.unit.test.ts b/tests/e2e/cross-version-wire/release-checkout.unit.test.ts index 7057a38babd..106e2ea778e 100644 --- a/tests/e2e/cross-version-wire/release-checkout.unit.test.ts +++ b/tests/e2e/cross-version-wire/release-checkout.unit.test.ts @@ -263,12 +263,24 @@ afterEach(() => { describe('release checkout materialization', () => { it('single-flights concurrent consumers of one release identity', async () => { const cacheRoot = temporaryCacheRoot() + let publications = 0 + const options = { + cacheRoot, + testHooks: { + populateStaging: async (context: CheckoutStagingContext) => { + publications++ + await populateMinimalStaging(context) + } + } + } const checkouts = await Promise.all([ - materializeReleaseCheckout('v1.4.190', { cacheRoot }), - materializeReleaseCheckout('v1.4.190', { cacheRoot }), - materializeReleaseCheckout('v1.4.190', { cacheRoot }) + materializeReleaseCheckout('v1.4.190', options), + materializeReleaseCheckout('v1.4.190', options), + materializeReleaseCheckout('v1.4.190', options) ]) + expect(publications).toBe(1) + expect(new Set(checkouts.map(({ root }) => root))).toHaveLength(1) expect(relative(cacheRoot, checkouts[0]!.root)).not.toMatch(/^\.\./) }) @@ -291,18 +303,29 @@ describe('release checkout materialization', () => { ) const cacheRoot = temporaryCacheRoot() - const first = await materializeReleaseCheckout(firstRef, { cacheRoot }) + const options = { cacheRoot, testHooks: { populateStaging: populateMinimalStaging } } + const first = await materializeReleaseCheckout(firstRef, options) const dependency = join(first.root, 'delayed-dependency.mjs') const entry = join(first.root, 'delayed-entry.mjs') + const importStarted = join(cacheRoot, 'import-started') + const continueImport = join(cacheRoot, 'continue-import') writeFileSync(dependency, "export const loaded = 'first-release'\n") writeFileSync( entry, - 'await new Promise((resolve) => setTimeout(resolve, 100))\n' + + "import { existsSync, writeFileSync } from 'node:fs'\n" + + `writeFileSync(${JSON.stringify(importStarted)}, '')\n` + + `while (!existsSync(${JSON.stringify(continueImport)})) await new Promise((resolve) => setTimeout(resolve, 10))\n` + "export const loaded = (await import('./delayed-dependency.mjs')).loaded\n" ) const loading = importReleaseCheckoutModule(first, '/delayed-entry.mjs') - const second = await materializeReleaseCheckout(secondRef, { cacheRoot }) + let second: ReleaseCheckout + try { + await waitForFile(importStarted, 5_000) + second = await materializeReleaseCheckout(secondRef, options) + } finally { + writeFileSync(continueImport, '') + } await expect(loading).resolves.toMatchObject({ loaded: 'first-release' }) expect(first.root).not.toBe(second.root) @@ -311,7 +334,10 @@ describe('release checkout materialization', () => { it('causally single-flights a rival process before publishing an in-use checkout', async () => { const cacheRoot = temporaryCacheRoot() const scratch = temporaryCacheRoot() - const published = await materializeReleaseCheckout('v1.4.190', { cacheRoot }) + const published = await materializeReleaseCheckout('v1.4.190', { + cacheRoot, + testHooks: { populateStaging: populateMinimalStaging } + }) await expect(runContentionPhase(published, scratch, 'locked', false)).resolves.toBe(true) // In the same causally acknowledged interleaving, a no-lock materializer diff --git a/tests/e2e/electron-home-isolation.spec.ts b/tests/e2e/electron-home-isolation.spec.ts index 65aa1b7dc10..2fae6f42eec 100644 --- a/tests/e2e/electron-home-isolation.spec.ts +++ b/tests/e2e/electron-home-isolation.spec.ts @@ -1,4 +1,5 @@ import type { ElectronApplication } from '@stablyai/playwright-test' +import { realpathSync } from 'node:fs' import path from 'node:path' import { expect, test } from './helpers/orca-app' @@ -23,7 +24,7 @@ async function readElectronHomeState(electronApp: ElectronApplication) { // HOME boundary and that real-home routing lands inside the disposable profile. test('isolates Electron and Codex from the developer home by default', async ({ electronApp }) => { const state = await readElectronHomeState(electronApp) - const expectedHome = path.join(state.userDataDir!, 'home') + const expectedHome = realpathSync.native(path.join(state.userDataDir!, 'home')) expect(state.appHome).toBe(expectedHome) expect(state.nodeHome).toBe(expectedHome) diff --git a/tests/e2e/ephemeral-vm-provisioned-root.spec.ts b/tests/e2e/ephemeral-vm-provisioned-root.spec.ts index 0dc51224bb3..394868e63be 100644 --- a/tests/e2e/ephemeral-vm-provisioned-root.spec.ts +++ b/tests/e2e/ephemeral-vm-provisioned-root.spec.ts @@ -1,6 +1,6 @@ import { execFileSync } from 'node:child_process' import { chmodSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs' -import { tmpdir } from 'node:os' +import { homedir, tmpdir } from 'node:os' import path from 'node:path' import { expect, test } from './helpers/orca-app' import { ensureDockerSshRelayImage } from './helpers/docker-ssh-relay-image' @@ -136,6 +136,8 @@ async function addRecipeRepo(page: Parameters[0], re function seedRecipeRepo(repoPath: string, target: DockerSshRelayTarget): string { const createScript = path.join(repoPath, 'create.sh') const destroyScript = path.join(repoPath, 'destroy.sh') + // The recipe's isolated HOME must still address the engine that owns the fixture container. + const docker = `docker --config ${shellQuote(process.env.DOCKER_CONFIG ?? path.join(homedir(), '.docker'))}` writeFileSync( createScript, `#!/usr/bin/env bash @@ -145,8 +147,8 @@ set -euo pipefail [ -n "\${ORCA_REPO_REF:-}" ] [ -n "\${ORCA_REPO_REF_HEAD:-}" ] [ -n "\${ORCA_REPO_BRANCH:-}" ] -docker exec ${shellQuote(target.containerName)} git -C ${shellQuote(DOCKER_SSH_RELAY_REMOTE_REPO_PATH)} cat-file -e "$ORCA_REPO_REF_HEAD^{commit}" -docker exec ${shellQuote(target.containerName)} git -C ${shellQuote(DOCKER_SSH_RELAY_REMOTE_REPO_PATH)} checkout -B "$ORCA_REPO_BRANCH" "$ORCA_REPO_REF_HEAD" >&2 +${docker} exec ${shellQuote(target.containerName)} git -C ${shellQuote(DOCKER_SSH_RELAY_REMOTE_REPO_PATH)} cat-file -e "$ORCA_REPO_REF_HEAD^{commit}" +${docker} exec ${shellQuote(target.containerName)} git -C ${shellQuote(DOCKER_SSH_RELAY_REMOTE_REPO_PATH)} checkout -B "$ORCA_REPO_BRANCH" "$ORCA_REPO_REF_HEAD" >&2 node -e 'console.log(JSON.stringify({schemaVersion:2,checkoutMode:"provisioned-root",connection:{type:"ssh",projectRoot:process.argv[1],target:{label:"Docker provisioned root",host:process.argv[2],port:Number(process.argv[3]),username:"root",identityFile:process.argv[4],identitiesOnly:true}}}))' ${shellQuote(DOCKER_SSH_RELAY_REMOTE_REPO_PATH)} ${shellQuote(target.host)} ${target.port} ${shellQuote(target.identityFile)} ` ) @@ -155,7 +157,7 @@ node -e 'console.log(JSON.stringify({schemaVersion:2,checkoutMode:"provisioned-r `#!/usr/bin/env bash set -euo pipefail cat >/dev/null -docker rm -f ${shellQuote(target.containerName)} >/dev/null +${docker} rm -f ${shellQuote(target.containerName)} >/dev/null ` ) chmodSync(createScript, 0o755) diff --git a/tests/e2e/feature-wall.spec.ts b/tests/e2e/feature-wall.spec.ts index 428ce5d7996..fb422ec8bf8 100644 --- a/tests/e2e/feature-wall.spec.ts +++ b/tests/e2e/feature-wall.spec.ts @@ -179,9 +179,40 @@ test.describe('Feature tour modal', () => { }) test('does not pre-check configured workflows until the user visits them', async ({ - orcaPage + orcaPage, + electronApp }) => { - await orcaPage.evaluate(() => { + await electronApp.evaluate( + ({ ipcMain }, preflightStatus) => { + ipcMain.removeHandler('preflight:check') + ipcMain.handle('preflight:check', () => preflightStatus) + ipcMain.removeHandler('linear:status') + ipcMain.handle('linear:status', () => ({ connected: false, viewer: null })) + ipcMain.removeHandler('jira:status') + ipcMain.handle('jira:status', () => ({ connected: false, viewer: null })) + }, + { + git: { installed: true }, + gh: { installed: true, authenticated: true }, + glab: { installed: false, authenticated: false }, + bitbucket: { configured: false, authenticated: false, account: null }, + azureDevOps: { + configured: false, + authenticated: false, + account: null, + baseUrl: null, + tokenConfigured: false + }, + gitea: { + configured: false, + authenticated: false, + account: null, + baseUrl: null, + tokenConfigured: false + } + } + ) + await orcaPage.evaluate(async () => { for (const key of [ 'orca.featureWall.visitedWorkflows.v1', 'orca.featureWall.visitedAgentSteps.v1', @@ -198,32 +229,12 @@ test.describe('Feature tour modal', () => { if (!store) { throw new Error('window.__store is not available') } - store.setState({ - preflightStatus: { - git: { installed: true }, - gh: { installed: true, authenticated: true }, - glab: { installed: false, authenticated: false }, - bitbucket: { configured: false, authenticated: false, account: null }, - azureDevOps: { - configured: false, - authenticated: false, - account: null, - baseUrl: null, - tokenConfigured: false - }, - gitea: { - configured: false, - authenticated: false, - account: null, - baseUrl: null, - tokenConfigured: false - } - }, - preflightStatusChecked: true, - preflightStatusLoading: false, - linearStatus: { connected: false, viewer: null }, - linearStatusChecked: true - }) + // Seed through the status actions so each result gets the current execution context. + await Promise.all([ + store.getState().refreshPreflightStatus({ force: true }), + store.getState().checkLinearConnection(true), + store.getState().checkJiraConnection() + ]) store.getState().openModal('feature-wall', { source: 'help_menu' }) }) diff --git a/tests/e2e/file-explorer-watch-refresh.spec.ts b/tests/e2e/file-explorer-watch-refresh.spec.ts index d8a1bc1e4e7..85fbd66409e 100644 --- a/tests/e2e/file-explorer-watch-refresh.spec.ts +++ b/tests/e2e/file-explorer-watch-refresh.spec.ts @@ -35,7 +35,7 @@ test('refreshes the visible tree after external Windows file changes', async ({ const row = (name: string) => orcaPage .locator('[data-file-explorer-row]') - .filter({ hasText: new RegExp(`^${name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}$`) }) + .filter({ has: orcaPage.getByText(name, { exact: true }) }) rmSync(originalPath, { force: true }) rmSync(renamedPath, { force: true }) diff --git a/tests/e2e/folder-setup-shallow-priority.spec.ts b/tests/e2e/folder-setup-shallow-priority.spec.ts index 15bccef63ee..e8cb72d64e8 100644 --- a/tests/e2e/folder-setup-shallow-priority.spec.ts +++ b/tests/e2e/folder-setup-shallow-priority.spec.ts @@ -1,3 +1,4 @@ +import { openSidebarProjectDialog } from './helpers/sidebar-project-dialog' import { execFileSync } from 'node:child_process' import { mkdirSync, realpathSync, rmSync, writeFileSync } from 'node:fs' import { mkdtemp } from 'node:fs/promises' @@ -166,10 +167,7 @@ test('prioritizes shallow sibling repositories in a bounded nested scan', async const fixture = await createShallowPriorityTruncationFixture() await chooseFolderInNativeDialog(electronApp, fixture.parentPath) - await orcaPage - .getByRole('button', { name: /Add Project/i }) - .first() - .click() + await openSidebarProjectDialog(orcaPage) const dialog = orcaPage.getByRole('dialog', { name: /Add a project/i }) await expect(dialog).toBeVisible() await dialog.getByRole('button', { name: /Browse folder/i }).click() @@ -256,10 +254,7 @@ test('can stop a nested repo scan and import repositories found so far', async ( }) await chooseFolderInNativeDialog(electronApp, fixture.parentPath) - await orcaPage - .getByRole('button', { name: /Add Project/i }) - .first() - .click() + await openSidebarProjectDialog(orcaPage) const dialog = orcaPage.getByRole('dialog', { name: /Add a project/i }) await dialog.getByRole('button', { name: /Browse folder/i }).click() diff --git a/tests/e2e/folder-setup.spec.ts b/tests/e2e/folder-setup.spec.ts index fc0f27824cc..5c736ec7efe 100644 --- a/tests/e2e/folder-setup.spec.ts +++ b/tests/e2e/folder-setup.spec.ts @@ -1,3 +1,4 @@ +import { openSidebarProjectDialog } from './helpers/sidebar-project-dialog' import { execFileSync } from 'node:child_process' import { mkdirSync, realpathSync, rmSync, writeFileSync } from 'node:fs' import { mkdtemp } from 'node:fs/promises' @@ -122,10 +123,7 @@ test.describe('Folder setup', () => { const fixture = await createNestedRepoFixture() await chooseFolderInNativeDialog(electronApp, fixture.parentPath) - await orcaPage - .getByRole('button', { name: /Add Project/i }) - .first() - .click() + await openSidebarProjectDialog(orcaPage) const dialog = orcaPage.getByRole('dialog', { name: /Add a project/i }) await expect(dialog).toBeVisible() await dialog.getByRole('button', { name: /Browse folder/i }).click() @@ -190,10 +188,7 @@ test.describe('Folder setup', () => { const fixture = await createLargeNestedRepoFixture() await chooseFolderInNativeDialog(electronApp, fixture.parentPath) - await orcaPage - .getByRole('button', { name: /Add Project/i }) - .first() - .click() + await openSidebarProjectDialog(orcaPage) const dialog = orcaPage.getByRole('dialog', { name: /Add a project/i }) await expect(dialog).toBeVisible() await dialog.getByRole('button', { name: /Browse folder/i }).click() diff --git a/tests/e2e/git-no-upstream-polling-churn.spec.ts b/tests/e2e/git-no-upstream-polling-churn.spec.ts index 2cdd1d129f6..6b3ca765fc7 100644 --- a/tests/e2e/git-no-upstream-polling-churn.spec.ts +++ b/tests/e2e/git-no-upstream-polling-churn.spec.ts @@ -3,6 +3,23 @@ import { existsSync, readFileSync, realpathSync, unlinkSync, writeFileSync } fro import type { Page, TestInfo } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' +// This isolated app needs local trace files; network telemetry remains disabled. +test.use({ + orcaAppExtraEnv: { + CI: '', + GITHUB_ACTIONS: '', + GITLAB_CI: '', + CIRCLECI: '', + TRAVIS: '', + BUILDKITE: '', + JENKINS_URL: '', + TEAMCITY_VERSION: '', + ORCA_DIAGNOSTICS_DISABLED: '', + DO_NOT_TRACK: '1', + ORCA_TELEMETRY_DISABLED: '1' + } +}) + // Repro command: // SKIP_BUILD=1 pnpm exec playwright test tests/e2e/git-no-upstream-polling-churn.spec.ts --config tests/playwright.config.ts --project electron-headless --reporter=json // Trigger: active worktree branch "Initi-Project" has no configured upstream @@ -21,6 +38,7 @@ type RendererTimerMeasurement = { } type GitProbeFailureCounts = { + observedGitCommands: number noConfiguredUpstreamFailures: number missingSameNameOriginFailures: number } @@ -138,11 +156,11 @@ async function measureRendererDuringPolling(page: Page): Promise { await selectRepoForActivePolling(orcaPage, testRepoPath, repoPath) const diagnostics = await readDiagnosticsStatus(orcaPage) - test.skip(!diagnostics.localFileEnabled, 'local diagnostic traces are disabled') + expect(diagnostics.localFileEnabled).toBe(true) + expect(diagnostics.bundleEnabled).toBe(false) clearTraceFile(diagnostics) const measurement = await measureRendererDuringPolling(orcaPage) @@ -209,6 +229,10 @@ test.describe('Git no-upstream polling churn repro', () => { const counts = readGitProbeFailureCounts(diagnostics.traceFilePath, repoPath) annotatePolling(testInfo, measurement, counts) + expect( + counts.observedGitCommands, + 'No Git activity was recorded for the measured repo' + ).toBeGreaterThan(0) expect(measurement.maxTimerDriftMs).toBeLessThan(MAX_RENDERER_TIMER_DRIFT_MS) // Why: the #4559 trace showed these stable negative upstream probes being // retried every poll. Under parallel e2e load one in-flight refresh can diff --git a/tests/e2e/github-url-smart-input-transition.spec.ts b/tests/e2e/github-url-smart-input-transition.spec.ts index e078007bd78..f042c4ef676 100644 --- a/tests/e2e/github-url-smart-input-transition.spec.ts +++ b/tests/e2e/github-url-smart-input-transition.spec.ts @@ -203,6 +203,12 @@ async function installHeldGitLabLookup( __releaseGitLabUrlLookup?: () => void } fixture.__gitlabUrlLookupStarted = false + ipcMain.removeHandler('preflight:check') + ipcMain.handle('preflight:check', () => ({ + git: { installed: true }, + gh: { installed: true, authenticated: true }, + glab: { installed: true, authenticated: true } + })) ipcMain.removeHandler('gitlab:listMRs') ipcMain.handle('gitlab:listMRs', () => ({ items: [wrongItem], @@ -221,23 +227,12 @@ async function installHeldGitLabLookup( }, { wrongItem: GITLAB_WRONG_ITEM, targetItem: GITLAB_TARGET_ITEM } ) - await page.evaluate(() => { + await page.evaluate(async () => { const store = window.__store if (!store) { throw new Error('window.__store is not available') } - const state = store.getState() - if (!state.preflightStatusContextKey) { - throw new Error('preflight context is not ready') - } - store.setState({ - preflightStatus: { - git: state.preflightStatus?.git ?? { installed: true }, - gh: state.preflightStatus?.gh ?? { installed: true, authenticated: true }, - glab: { installed: true, authenticated: true } - }, - preflightStatusChecked: true - }) + await store.getState().refreshPreflightStatus({ force: true }) }) } diff --git a/tests/e2e/golden-core-flows.spec.ts b/tests/e2e/golden-core-flows.spec.ts index 96863251081..7d56b0805b4 100644 --- a/tests/e2e/golden-core-flows.spec.ts +++ b/tests/e2e/golden-core-flows.spec.ts @@ -1,3 +1,4 @@ +import { openSidebarProjectDialog } from './helpers/sidebar-project-dialog' import { execFileSync } from 'node:child_process' import { mkdirSync, realpathSync, rmSync, writeFileSync } from 'node:fs' import { mkdtemp } from 'node:fs/promises' @@ -211,10 +212,7 @@ async function addProjectFromSidebar( repoPath: string ): Promise { await chooseFolderInNativeDialog(electronApp, repoPath) - await page - .getByRole('button', { name: /Add Project/i }) - .first() - .click() + await openSidebarProjectDialog(page) const addDialog = page.getByRole('dialog', { name: /Add a project/i }) await expect(addDialog).toBeVisible() await addDialog.getByRole('button', { name: /Browse folder/i }).click() diff --git a/tests/e2e/helpers/browser-link-server.ts b/tests/e2e/helpers/browser-link-server.ts new file mode 100644 index 00000000000..81debdc5859 --- /dev/null +++ b/tests/e2e/helpers/browser-link-server.ts @@ -0,0 +1,100 @@ +import { createServer, type Server } from 'node:http' +import type { AddressInfo } from 'node:net' + +async function closeServer(server: Server): Promise { + await new Promise((resolve, reject) => + server.close((error) => { + if (error) { + reject(error) + return + } + resolve() + }) + ) +} + +export async function startBrowserLinkServer(): Promise<{ + sourceUrl: string + close: () => Promise +}> { + const server = createServer((request, response) => { + const origin = `http://127.0.0.1:${(server.address() as AddressInfo).port}` + const pathname = new URL(request.url ?? '/', origin).pathname + response.writeHead(200, { 'Content-Type': 'text/html; charset=utf-8' }) + if (pathname === '/destination') { + response.end( + `Linked destinationDestination Return` + ) + return + } + if (pathname === '/blank-destination') { + response.end( + 'Blank target destinationBlank target destination' + ) + return + } + if (pathname === '/frame-destination') { + response.end( + `Frame destinationFrame destination Return` + ) + return + } + if (pathname === '/frame-modifier-destination') { + response.end( + 'Frame modifier destinationFrame modifier destination' + ) + return + } + if (pathname === '/frame-middle-destination') { + response.end( + 'Frame middle destinationFrame middle destination' + ) + return + } + if (pathname === '/frame') { + response.end( + `${request.url?.includes('shift-middle') ? 'Frame shift middle destination' : ''}Open frame destinationOpen frame modifier destinationOpen frame middle destinationOpen foreground frame tab` + ) + return + } + if (pathname === '/modifier-destination') { + response.end( + 'Modifier destinationModifier destination' + ) + return + } + if (pathname === '/middle-destination') { + response.end( + 'Middle-click destinationMiddle-click destination' + ) + return + } + response.end(` + + + ${request.url?.includes('shift-middle') ? 'Shift middle destination' : 'Source page'} + + Open destination + Open blank target destination + Open with modifier + Open with middle click + Open foreground tab + Handle in page + + + + + `) + }) + await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve)) + const port = (server.address() as AddressInfo).port + return { + sourceUrl: `http://127.0.0.1:${port}/source`, + close: () => closeServer(server) + } +} diff --git a/tests/e2e/helpers/docker-ssh-relay-connection.ts b/tests/e2e/helpers/docker-ssh-relay-connection.ts index a17ad916e66..3e0c35f3c53 100644 --- a/tests/e2e/helpers/docker-ssh-relay-connection.ts +++ b/tests/e2e/helpers/docker-ssh-relay-connection.ts @@ -1,4 +1,4 @@ -import type { Page } from '@stablyai/playwright-test' +import { expect, type Page } from '@stablyai/playwright-test' import { DOCKER_SSH_PROXY_JUMP_REMOTE_REPO_PATH, @@ -227,3 +227,33 @@ export async function reconnectDisconnectedDockerSshRelayTarget( ): Promise { return performDockerSshRelayReconnect(page, targetId, false) } + +export async function recoverDockerSshRelayAfterFault( + page: Page, + targetId: string, + injectFault: () => void | Promise +): Promise { + const readAuthority = () => + page.evaluate((id) => window.__store?.getState().sshConnectionStates.get(id), targetId) + const before = await readAuthority() + expect(before).toMatchObject({ + status: 'connected', + providerEpoch: expect.any(String), + connectionGeneration: expect.any(Number) + }) + await injectFault() + // The pre-fault connected publication can remain visible until the next IPC event. + await expect + .poll( + async () => { + const after = await readAuthority() + return ( + after?.status === 'connected' && + (after.providerEpoch !== before?.providerEpoch || + after.connectionGeneration !== before?.connectionGeneration) + ) + }, + { timeout: 120_000, message: 'SSH authority did not recover after the injected fault' } + ) + .toBe(true) +} diff --git a/tests/e2e/helpers/electron-crashpad-cleanup.ts b/tests/e2e/helpers/electron-crashpad-cleanup.ts new file mode 100644 index 00000000000..8ec1a98b26b --- /dev/null +++ b/tests/e2e/helpers/electron-crashpad-cleanup.ts @@ -0,0 +1,46 @@ +import { execFileSync } from 'node:child_process' +import path from 'node:path' + +function ownsCrashpad(command: string, userDataDir: string): boolean { + return ( + command.includes('/chrome_crashpad_handler ') && + command.includes(` --database=${path.join(userDataDir, 'Crashpad')} `) + ) +} + +export function cleanupE2ECrashpad(userDataDir: string): void { + if (process.platform !== 'darwin') { + return + } + + // macOS reparents Crashpad before app exit; its inherited stderr can keep Playwright open. + try { + const table = execFileSync('ps', ['-axo', 'pid=,command='], { + encoding: 'utf8', + timeout: 5_000 + }) + for (const row of table.split('\n')) { + const match = row.match(/^\s*(\d+)\s+(.+)$/) + if (!match || !ownsCrashpad(match[2], userDataDir)) { + continue + } + const pid = Number(match[1]) + if (!Number.isSafeInteger(pid) || pid <= 1) { + continue + } + try { + const command = execFileSync('ps', ['-p', String(pid), '-o', 'command='], { + encoding: 'utf8', + timeout: 5_000 + }) + if (ownsCrashpad(command, userDataDir)) { + process.kill(pid, 'SIGTERM') + } + } catch { + // The test-owned reporter may already have exited. + } + } + } catch { + // Cleanup remains best-effort when process enumeration is unavailable. + } +} diff --git a/tests/e2e/helpers/electron-crashpad-cleanup.unit.test.ts b/tests/e2e/helpers/electron-crashpad-cleanup.unit.test.ts new file mode 100644 index 00000000000..cdf552e2548 --- /dev/null +++ b/tests/e2e/helpers/electron-crashpad-cleanup.unit.test.ts @@ -0,0 +1,45 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { execFileSync } from 'node:child_process' +import path from 'node:path' +import { cleanupE2ECrashpad } from './electron-crashpad-cleanup' + +vi.mock('node:child_process', () => ({ execFileSync: vi.fn() })) + +const profile = '/tmp/test profile' +const database = path.join(profile, 'Crashpad') +const reporter = `/Electron Framework/Helpers/chrome_crashpad_handler --database=${database} --annotation=prod=Electron` + +afterEach(() => vi.restoreAllMocks()) + +describe('test-owned macOS Crashpad cleanup', () => { + it('terminates only the reporter for the exact temporary profile after rechecking ownership', () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + const kill = vi.spyOn(process, 'kill').mockReturnValue(true) + vi.mocked(execFileSync) + .mockReturnValueOnce( + `111 ${reporter}\n222 ${reporter.replace('Crashpad ', 'Crashpad-old ')}\n333 ${reporter.replace('test profile', 'another profile')}\n444 /bin/echo --database=${database} \n` + ) + .mockReturnValueOnce(reporter) + cleanupE2ECrashpad(profile) + expect(kill).toHaveBeenCalledExactlyOnceWith(111, 'SIGTERM') + expect(execFileSync).toHaveBeenLastCalledWith('ps', ['-p', '111', '-o', 'command='], { + encoding: 'utf8', + timeout: 5_000 + }) + }) + + it('does not signal a PID whose ownership changed after enumeration', () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + const kill = vi.spyOn(process, 'kill').mockReturnValue(true) + vi.mocked(execFileSync).mockReturnValueOnce(`111 ${reporter}`).mockReturnValueOnce('/bin/sh') + cleanupE2ECrashpad(profile) + expect(kill).not.toHaveBeenCalled() + }) + + it.each(['win32', 'linux'] as const)('does not enumerate processes on %s', (platform) => { + vi.spyOn(process, 'platform', 'get').mockReturnValue(platform) + vi.mocked(execFileSync).mockClear() + cleanupE2ECrashpad(profile) + expect(execFileSync).not.toHaveBeenCalled() + }) +}) diff --git a/tests/e2e/helpers/electron-launch-args.ts b/tests/e2e/helpers/electron-launch-args.ts index fc2ff1e81aa..9128a5fb551 100644 --- a/tests/e2e/helpers/electron-launch-args.ts +++ b/tests/e2e/helpers/electron-launch-args.ts @@ -7,6 +7,10 @@ export function getOrcaElectronLaunchArgs(mainPath: string, headful: boolean): s // these Chromium switches startup can block before the first renderer target. const keychainArgs = process.platform === 'darwin' ? ['--password-store=basic', '--use-mock-keychain'] : [] + if (process.platform === 'darwin') { + // Crash tests must not block later launches on AppKit's saved-window recovery dialog. + return [...keychainArgs, appPath, '-ApplePersistenceIgnoreState', 'YES'] + } if (headful || process.platform !== 'linux') { return [...keychainArgs, appPath] } diff --git a/tests/e2e/helpers/electron-launch-args.unit.test.ts b/tests/e2e/helpers/electron-launch-args.unit.test.ts index ed981951f14..636299fbc4e 100644 --- a/tests/e2e/helpers/electron-launch-args.unit.test.ts +++ b/tests/e2e/helpers/electron-launch-args.unit.test.ts @@ -8,10 +8,17 @@ describe('getOrcaElectronLaunchArgs', () => { const mainPath = join(root, 'out', 'main', 'index.js') const args = getOrcaElectronLaunchArgs(mainPath, true) - expect(args.at(-1)).toBe(root) if (process.platform === 'darwin') { - expect(args.slice(0, -1)).toEqual(['--password-store=basic', '--use-mock-keychain']) + expect(args).toEqual([ + '--password-store=basic', + '--use-mock-keychain', + root, + '-ApplePersistenceIgnoreState', + 'YES' + ]) + } else { + expect(args.at(-1)).toBe(root) } - expect(getOrcaElectronLaunchArgs(mainPath, false).at(-1)).toBe(root) + expect(getOrcaElectronLaunchArgs(mainPath, false)).toContain(root) }) }) diff --git a/tests/e2e/helpers/electron-process-shutdown.ts b/tests/e2e/helpers/electron-process-shutdown.ts index 5180575f1a6..f9b642a676e 100644 --- a/tests/e2e/helpers/electron-process-shutdown.ts +++ b/tests/e2e/helpers/electron-process-shutdown.ts @@ -2,6 +2,7 @@ import type { ChildProcess } from 'node:child_process' import { execFileSync } from 'node:child_process' import { existsSync, readFileSync, readdirSync } from 'node:fs' import path from 'node:path' +import { cleanupE2ECrashpad } from './electron-crashpad-cleanup' import type { ElectronApplication } from '@stablyai/playwright-test' const GRACEFUL_CLOSE_TIMEOUT_MS = 10_000 @@ -19,6 +20,16 @@ function hasExited(proc: ChildProcess): boolean { return proc.exitCode !== null || proc.signalCode !== null } +function releaseExitedProcessPipes(proc: ChildProcess): void { + if (!hasExited(proc)) { + return + } + // Detached SSH helpers can retain inherited pipes after Electron itself exits. + for (const stream of proc.stdio) { + stream?.destroy() + } +} + function waitForExit(proc: ChildProcess, timeoutMs: number): Promise { if (hasExited(proc)) { return Promise.resolve(true) @@ -166,12 +177,16 @@ export async function forceQuitElectronAppForE2E(app: ElectronApplication): Prom } } await waitForExit(proc, PROCESS_EXIT_TIMEOUT_MS) + releaseExitedProcessPipes(proc) // Hands the dead app back to Playwright so worker teardown has nothing left to wait on. await app.close().catch(() => undefined) } export async function closeElectronAppForE2E(app: ElectronApplication): Promise { const proc = app.process() + const releasePipes = (): void => releaseExitedProcessPipes(proc) + proc.once('exit', releasePipes) + releasePipes() try { await withTimeout(app.close(), GRACEFUL_CLOSE_TIMEOUT_MS, 'Timed out closing Electron app') if (proc) { @@ -184,6 +199,9 @@ export async function closeElectronAppForE2E(app: ElectronApplication): Promise< if (proc) { await forceKillProcessTree(proc) } + } finally { + proc.off('exit', releasePipes) + releasePipes() } } @@ -221,4 +239,5 @@ export async function cleanupE2EDaemons(userDataDir: string): Promise { for (const pid of readDaemonPidFiles(userDataDir)) { await forceKillPidTree(pid) } + cleanupE2ECrashpad(userDataDir) } diff --git a/tests/e2e/helpers/electron-process-shutdown.unit.test.ts b/tests/e2e/helpers/electron-process-shutdown.unit.test.ts new file mode 100644 index 00000000000..316aec32ed5 --- /dev/null +++ b/tests/e2e/helpers/electron-process-shutdown.unit.test.ts @@ -0,0 +1,59 @@ +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import type { ChildProcess } from 'node:child_process' +import type { ElectronApplication } from '@stablyai/playwright-test' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { closeElectronAppForE2E } from './electron-process-shutdown' + +function exitedAppFixture() { + const proc = Object.assign(new EventEmitter(), { + exitCode: null as number | null, + signalCode: null, + stdio: [new PassThrough(), new PassThrough(), new PassThrough()] + }) + const pipesClosed = Promise.all( + proc.stdio.map((stream) => new Promise((resolve) => stream.once('close', resolve))) + ) + const close = vi.fn(() => pipesClosed) + const app = { + process: () => proc as unknown as ChildProcess, + close + } as unknown as ElectronApplication + return { proc, app, close } +} + +afterEach(() => vi.useRealTimers()) + +describe('Electron shutdown with inherited pipes', () => { + it('releases retained pipes only after Electron exits, settling Playwright cleanup', async () => { + const { proc, app, close } = exitedAppFixture() + const closing = closeElectronAppForE2E(app) + expect(close).toHaveBeenCalledOnce() + expect(proc.stdio.every((stream) => !stream.destroyed)).toBe(true) + proc.exitCode = 0 + proc.emit('exit', 0, null) + await closing + expect(proc.stdio.every((stream) => stream.destroyed)).toBe(true) + expect(proc.listenerCount('exit')).toBe(0) + }) + + it('releases pipes when Electron already exited before cleanup starts', async () => { + const { proc, app } = exitedAppFixture() + proc.exitCode = 0 + await closeElectronAppForE2E(app) + expect(proc.stdio.every((stream) => stream.destroyed)).toBe(true) + }) + + it('does not release pipes if shutdown times out without confirmed process exit', async () => { + vi.useFakeTimers() + const { proc, app } = exitedAppFixture() + const closing = closeElectronAppForE2E(app) + await vi.advanceTimersByTimeAsync(10_000) + await closing + expect(proc.stdio.every((stream) => !stream.destroyed)).toBe(true) + expect(proc.listenerCount('exit')).toBe(0) + for (const stream of proc.stdio) { + stream.destroy() + } + }) +}) diff --git a/tests/e2e/helpers/git-status-retry-barrier.ts b/tests/e2e/helpers/git-status-retry-barrier.ts new file mode 100644 index 00000000000..e166313d003 --- /dev/null +++ b/tests/e2e/helpers/git-status-retry-barrier.ts @@ -0,0 +1,61 @@ +import type { ElectronApplication } from '@stablyai/playwright-test' + +type StatusArgs = { worktreePath?: string; admissionTier?: string } +type StatusHandler = (event: unknown, args?: StatusArgs) => unknown +type RetryBarrier = { + captured: boolean + release: () => void + original: StatusHandler +} +type BarrierScope = typeof globalThis & { __gitStatusRetryBarrier?: RetryBarrier } + +export async function installGitStatusRetryBarrier( + app: ElectronApplication, + repoPath: string +): Promise { + await app.evaluate(({ ipcMain }, repoPath) => { + const scope = globalThis as BarrierScope + const handlers = (ipcMain as unknown as { _invokeHandlers: Map }) + ._invokeHandlers + const original = handlers.get('git:status') + if (!original || scope.__gitStatusRetryBarrier) { + throw new Error('Git status handler unavailable or retry barrier already installed') + } + let release!: () => void + const pending = new Promise((resolve) => { + release = resolve + }) + const state: RetryBarrier = { captured: false, release, original } + scope.__gitStatusRetryBarrier = state + handlers.set('git:status', async (event, args) => { + if ( + !state.captured && + args?.worktreePath === repoPath && + args.admissionTier === 'interactive' + ) { + state.captured = true + await pending + } + return original(event, args) + }) + }, repoPath) +} + +export async function hasCapturedGitStatusRetry(app: ElectronApplication): Promise { + return app.evaluate(() => (globalThis as BarrierScope).__gitStatusRetryBarrier?.captured ?? false) +} + +export async function restoreGitStatusRetryHandler(app: ElectronApplication): Promise { + await app.evaluate(({ ipcMain }) => { + const scope = globalThis as BarrierScope + const state = scope.__gitStatusRetryBarrier + if (!state) { + return + } + const handlers = (ipcMain as unknown as { _invokeHandlers: Map }) + ._invokeHandlers + handlers.set('git:status', state.original) + state.release() + delete scope.__gitStatusRetryBarrier + }) +} diff --git a/tests/e2e/helpers/git-status-retry-barrier.unit.test.ts b/tests/e2e/helpers/git-status-retry-barrier.unit.test.ts new file mode 100644 index 00000000000..74bdd8d8159 --- /dev/null +++ b/tests/e2e/helpers/git-status-retry-barrier.unit.test.ts @@ -0,0 +1,39 @@ +import type { ElectronApplication } from '@stablyai/playwright-test' +import { describe, expect, it, vi } from 'vitest' +import { + hasCapturedGitStatusRetry, + installGitStatusRetryBarrier, + restoreGitStatusRetryHandler +} from './git-status-retry-barrier' + +describe('Git status retry barrier', () => { + it('holds the target interactive request and restores the real handler on cleanup', async () => { + const original = vi.fn(async (_event: unknown, args: unknown) => args) + const handlers = new Map([['git:status', original]]) + const app = { + evaluate: (callback: (electron: unknown, arg?: unknown) => unknown, arg?: unknown) => + Promise.resolve(callback({ ipcMain: { _invokeHandlers: handlers } }, arg)) + } as unknown as ElectronApplication + await installGitStatusRetryBarrier(app, 'target-repo') + try { + const handler = handlers.get('git:status')! + const background = { worktreePath: 'target-repo', admissionTier: 'background' } + const otherRepo = { worktreePath: 'another-repo', admissionTier: 'interactive' } + await expect(handler({}, background)).resolves.toEqual(background) + await expect(handler({}, otherRepo)).resolves.toEqual(otherRepo) + expect(await hasCapturedGitStatusRetry(app)).toBe(false) + + const retry = { worktreePath: 'target-repo', admissionTier: 'interactive' } + const event = {} + const pending = handler(event, retry) + expect(await hasCapturedGitStatusRetry(app)).toBe(true) + expect(original).toHaveBeenCalledTimes(2) + await restoreGitStatusRetryHandler(app) + await expect(pending).resolves.toEqual(retry) + expect(original).toHaveBeenLastCalledWith(event, retry) + expect(handlers.get('git:status')).toBe(original) + } finally { + await restoreGitStatusRetryHandler(app) + } + }) +}) diff --git a/tests/e2e/helpers/paired-client-window-reveal.ts b/tests/e2e/helpers/paired-client-window-reveal.ts index 302d573c3d1..1ec5b635d77 100644 --- a/tests/e2e/helpers/paired-client-window-reveal.ts +++ b/tests/e2e/helpers/paired-client-window-reveal.ts @@ -29,22 +29,24 @@ export function assertPairedClientWindowRevealed(report: PairedClientWindowRevea export type PairedClientWindowFocusReport = PairedClientWindowRevealReport & { isFocused: boolean } /** - * Brings a paired client to the front, which a launched-but-background window never is. Main-side - * policies that ask whether the reader is looking at a WebContents read the OS focus state, so a - * spec driving real presses through such a policy has to put the window there first. + * Native-focus coverage must run on an isolated display or CI, never in background mode. */ export async function focusPairedClientWindow( client: RevealablePairedClient, { timeoutMs = 15_000 }: { timeoutMs?: number } = {} ): Promise { + await client.app.evaluate(() => { + if (process.env.ORCA_BACKGROUND_LAUNCH === '1') { + throw new Error('Native focus is forbidden by ORCA_BACKGROUND_LAUNCH') + } + }) const revealed = await revealPairedClientWindow(client) const deadline = Date.now() + timeoutMs let isFocused = false while (!isFocused) { isFocused = await client.app.evaluate(({ app, BrowserWindow }) => { const window = BrowserWindow.getAllWindows()[0] - // Why steal: nothing else in the run is asking for the front, and the window manager keeps - // the launching terminal there otherwise. + // Native-focus coverage requires a dedicated foreground session. app.focus({ steal: true }) window?.focus() return window?.isFocused() ?? false @@ -61,6 +63,9 @@ export async function revealPairedClientWindow( client: RevealablePairedClient ): Promise { const report = await client.app.evaluate(({ BrowserWindow }) => { + if (process.env.ORCA_BACKGROUND_LAUNCH === '1') { + throw new Error('Window reveal is forbidden by ORCA_BACKGROUND_LAUNCH') + } const windows = BrowserWindow.getAllWindows() const window = windows[0] const wasVisible = window?.isVisible() ?? false diff --git a/tests/e2e/helpers/paired-client-window-reveal.unit.test.ts b/tests/e2e/helpers/paired-client-window-reveal.unit.test.ts index dfb83e4c441..706088e6762 100644 --- a/tests/e2e/helpers/paired-client-window-reveal.unit.test.ts +++ b/tests/e2e/helpers/paired-client-window-reveal.unit.test.ts @@ -1,5 +1,10 @@ -import { describe, expect, it } from 'vitest' -import { assertPairedClientWindowRevealed } from './paired-client-window-reveal' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + assertPairedClientWindowRevealed, + focusPairedClientWindow, + revealPairedClientWindow, + type RevealablePairedClient +} from './paired-client-window-reveal' describe('assertPairedClientWindowRevealed', () => { it('accepts a window that the reveal made visible', () => { @@ -42,3 +47,41 @@ describe('assertPairedClientWindowRevealed', () => { ).toThrow(/stayed hidden after showInactive\(\)/) }) }) + +describe('paired client background safety', () => { + afterEach(() => vi.unstubAllEnvs()) + + function makeClient() { + const showInactive = vi.fn() + const focus = vi.fn() + const getAllWindows = vi.fn(() => [{ isVisible: () => false, showInactive, focus }]) + const evaluate = vi.fn(async (callback) => + callback({ + app: { focus }, + BrowserWindow: { getAllWindows } + }) + ) + const client = { + app: { evaluate }, + page: { waitForFunction: vi.fn() } + } as unknown as RevealablePairedClient + return { client, showInactive, focus, getAllWindows } + } + + it('rejects an explicit reveal before touching native windows', async () => { + vi.stubEnv('ORCA_BACKGROUND_LAUNCH', '1') + const { client, getAllWindows, showInactive } = makeClient() + await expect(revealPairedClientWindow(client)).rejects.toThrow('Window reveal is forbidden') + expect(getAllWindows).not.toHaveBeenCalled() + expect(showInactive).not.toHaveBeenCalled() + }) + + it.each(['0', '1'])('rejects focus in background mode with foreground=%s', async (foreground) => { + vi.stubEnv('ORCA_BACKGROUND_LAUNCH', '1') + vi.stubEnv('ORCA_E2E_FOREGROUND', foreground) + const { client, focus, getAllWindows } = makeClient() + await expect(focusPairedClientWindow(client)).rejects.toThrow('Native focus is forbidden') + expect(getAllWindows).not.toHaveBeenCalled() + expect(focus).not.toHaveBeenCalled() + }) +}) diff --git a/tests/e2e/helpers/remote-skill-cloud-fixture.ts b/tests/e2e/helpers/remote-skill-cloud-fixture.ts index f1d28d926b7..8e1650947dd 100644 --- a/tests/e2e/helpers/remote-skill-cloud-fixture.ts +++ b/tests/e2e/helpers/remote-skill-cloud-fixture.ts @@ -8,13 +8,12 @@ import { } from '../../../src/main/skills/skill-package-creation' import { SKILL_PACKAGE_CONTENT_TYPE } from '../../../src/shared/skill-package-manifest' -export const REMOTE_SKILL_CLOUD_PORT = Number(process.env.ORCA_E2E_SKILL_CLOUD_PORT ?? '43961') -export const REMOTE_SKILL_CLOUD_ORIGIN = `http://127.0.0.1:${REMOTE_SKILL_CLOUD_PORT}` export const REMOTE_SKILL_PACKAGE_ID = 'package_remote_e2e' export const REMOTE_SKILL_VERSION_ID = 'version_remote_e2e' export const REMOTE_SKILL_NAME = 'remote-e2e-skill' export type RemoteSkillCloudFixture = { + origin: string archive: CreatedSkillPackage bytes: Buffer requests: { method: string; path: string; body: unknown }[] @@ -39,19 +38,30 @@ export async function startRemoteSkillCloudFixture(): Promise { - void handleRemoteSkillCloudRequest({ request, response, archive, bytes, requests }).catch( - (error) => { - response.writeHead(500, { 'content-type': 'application/json' }) - response.end(JSON.stringify({ code: 'fixture_failed', message: String(error) })) - } - ) + void handleRemoteSkillCloudRequest({ + request, + response, + archive, + bytes, + requests, + origin + }).catch((error) => { + response.writeHead(500, { 'content-type': 'application/json' }) + response.end(JSON.stringify({ code: 'fixture_failed', message: String(error) })) + }) }) await new Promise((resolve, reject) => { server.once('error', reject) - server.listen(REMOTE_SKILL_CLOUD_PORT, '127.0.0.1', resolve) + server.listen(Number(process.env.ORCA_E2E_SKILL_CLOUD_PORT ?? 0), '127.0.0.1', resolve) }) - return { archive, bytes, requests, root, server } + const address = server.address() + if (!address || typeof address === 'string') { + throw new Error('Skill fixture has no TCP address') + } + origin = `http://127.0.0.1:${address.port}` + return { archive, bytes, requests, root, server, origin } } export async function stopRemoteSkillCloudFixture(fixture: RemoteSkillCloudFixture): Promise { @@ -60,13 +70,14 @@ export async function stopRemoteSkillCloudFixture(fixture: RemoteSkillCloudFixtu } async function handleRemoteSkillCloudRequest(input: { + origin: string request: IncomingMessage response: ServerResponse archive: CreatedSkillPackage bytes: Buffer requests: RemoteSkillCloudFixture['requests'] }): Promise { - const path = new URL(input.request.url ?? '/', REMOTE_SKILL_CLOUD_ORIGIN).pathname + const path = new URL(input.request.url ?? '/', input.origin).pathname if (input.request.method === 'GET' && path === '/package.tar.gz') { input.requests.push({ method: 'GET', path, body: null }) input.response.writeHead(200, { @@ -84,17 +95,19 @@ async function handleRemoteSkillCloudRequest(input: { const body = JSON.parse(await readRequestBody(input.request)) as unknown input.requests.push({ method: 'POST', path, body }) input.response.writeHead(200, { 'content-type': 'application/json' }) - input.response.end(JSON.stringify(downloadGrant(input.archive, input.bytes.length))) + input.response.end( + JSON.stringify(downloadGrant(input.archive, input.bytes.length, input.origin)) + ) return } input.response.writeHead(404, { 'content-type': 'application/json' }) input.response.end(JSON.stringify({ code: 'not_found', message: 'Not found' })) } -function downloadGrant(archive: CreatedSkillPackage, compressedBytes: number) { +function downloadGrant(archive: CreatedSkillPackage, compressedBytes: number, origin: string) { return { grant: { - url: `${REMOTE_SKILL_CLOUD_ORIGIN}/package.tar.gz`, + url: `${origin}/package.tar.gz`, expiresAt: '2099-01-01T00:00:00.000Z' }, version: { diff --git a/tests/e2e/helpers/remote-skill-cloud-fixture.unit.test.ts b/tests/e2e/helpers/remote-skill-cloud-fixture.unit.test.ts new file mode 100644 index 00000000000..70291cf0e8f --- /dev/null +++ b/tests/e2e/helpers/remote-skill-cloud-fixture.unit.test.ts @@ -0,0 +1,41 @@ +import { expect, it, vi } from 'vitest' +import { + REMOTE_SKILL_PACKAGE_ID, + REMOTE_SKILL_VERSION_ID, + startRemoteSkillCloudFixture, + stopRemoteSkillCloudFixture +} from './remote-skill-cloud-fixture' + +it('serves concurrent skill fixtures from independent bound origins', async () => { + vi.stubEnv('ORCA_E2E_SKILL_CLOUD_PORT', undefined) + const results = await Promise.allSettled([ + startRemoteSkillCloudFixture(), + startRemoteSkillCloudFixture() + ]) + const fixtures = results.flatMap((result) => + result.status === 'fulfilled' ? [result.value] : [] + ) + try { + expect(results.every((result) => result.status === 'fulfilled')).toBe(true) + expect(new Set(fixtures.map((fixture) => fixture.origin)).size).toBe(2) + for (const fixture of fixtures) { + const response = await fetch( + `${fixture.origin}/v1/skill-packages/${REMOTE_SKILL_PACKAGE_ID}/versions/${REMOTE_SKILL_VERSION_ID}/download-grants`, + { + method: 'POST', + body: '{}', + headers: { 'content-type': 'application/json' } + } + ) + expect(response.status).toBe(200) + const result = (await response.json()) as { grant: { url: string } } + expect(result.grant.url).toBe(`${fixture.origin}/package.tar.gz`) + const archive = await fetch(result.grant.url) + expect(Buffer.from(await archive.arrayBuffer())).toEqual(fixture.bytes) + expect(fixture.requests).toHaveLength(2) + } + } finally { + await Promise.all(fixtures.map(stopRemoteSkillCloudFixture)) + vi.unstubAllEnvs() + } +}) diff --git a/tests/e2e/helpers/seeded-test-repo.ts b/tests/e2e/helpers/seeded-test-repo.ts index dd88351b282..34c4f346714 100644 --- a/tests/e2e/helpers/seeded-test-repo.ts +++ b/tests/e2e/helpers/seeded-test-repo.ts @@ -28,7 +28,7 @@ export function isValidGitRepo(repoPath: string): boolean { } } -export function createSeededTestRepo(): string { +export function createSeededTestRepo(options: { publishPath?: boolean } = {}): string { // Why: realpathSync so the seeded path matches the store's repo.path on // macOS, where os.tmpdir() (/var/...) symlinks to /private/var/... and the // app canonicalizes repo.path via `git rev-parse --show-toplevel` on add. @@ -63,6 +63,8 @@ export function createSeededTestRepo(): string { stdio: 'pipe' }) - writeFileSync(TEST_REPO_PATH_FILE, testRepoDir) + if (options.publishPath !== false) { + writeFileSync(TEST_REPO_PATH_FILE, testRepoDir) + } return testRepoDir } diff --git a/tests/e2e/helpers/sidebar-project-dialog.ts b/tests/e2e/helpers/sidebar-project-dialog.ts new file mode 100644 index 00000000000..0cd4c453b5d --- /dev/null +++ b/tests/e2e/helpers/sidebar-project-dialog.ts @@ -0,0 +1,9 @@ +import { expect, type Page } from '@stablyai/playwright-test' + +export async function openSidebarProjectDialog(page: Page): Promise { + // The compact overflow retains standalone project import; the composer hosts a different flow. + await page.evaluate(() => window.__store!.getState().setSidebarWidth(220)) + await page.getByRole('button', { name: 'More workspace actions', exact: true }).click() + await page.getByRole('menuitem', { name: 'Add Project', exact: true }).click() + await expect(page.getByRole('dialog', { name: /Add a project/i })).toBeVisible() +} diff --git a/tests/e2e/helpers/source-control-ai-generation.ts b/tests/e2e/helpers/source-control-ai-generation.ts index f6be16d7342..c93a20e37a1 100644 --- a/tests/e2e/helpers/source-control-ai-generation.ts +++ b/tests/e2e/helpers/source-control-ai-generation.ts @@ -67,7 +67,7 @@ export async function seedCreatePrComposer(page: Page): Promise<{ prWorktreePath: string primaryBranch: string }> { - return page.evaluate(async () => { + const seeded = await page.evaluate(async () => { const store = window.__store ?? (() => { @@ -101,6 +101,7 @@ export async function seedCreatePrComposer(page: Page): Promise<{ const eligibility = { provider: 'github' as const, review: null, + reviewLookupOutcome: 'not_found' as const, canCreate: true, blockedReason: null, nextAction: null, @@ -121,7 +122,7 @@ export async function seedCreatePrComposer(page: Page): Promise<{ ...current.remoteStatusesByWorktree, [prWorktree.id]: { hasUpstream: true, - upstreamName: `origin/${branch}`, + upstreamName: primaryBranch, ahead: 0, behind: 0 } @@ -130,6 +131,10 @@ export async function seedCreatePrComposer(page: Page): Promise<{ args.branch === branch ? eligibility : { ...eligibility, canCreate: false }, fetchHostedReviewForBranch: async () => null, fetchPRForBranch: async () => null, + enqueueGitHubPRRefresh: () => undefined, + // Ignore provider work queued before this generation-only fixture was installed. + getEffectiveGitHubPRRefreshState: () => undefined, + prRefreshStates: {}, fetchUpstreamStatus: async () => undefined, setUpstreamStatus: () => undefined })) @@ -141,6 +146,12 @@ export async function seedCreatePrComposer(page: Page): Promise<{ primaryBranch } }) + // Checks reads fresh Git state instead of the seeded store cache. + execFileSync('git', ['branch', '--set-upstream-to', seeded.primaryBranch], { + cwd: seeded.prWorktreePath, + stdio: 'pipe' + }) + return seeded } export async function seedCommitMessageComposer(page: Page): Promise<{ diff --git a/tests/e2e/helpers/source-control-ai-generators.ts b/tests/e2e/helpers/source-control-ai-generators.ts index be3f1b43247..8c09bb8b556 100644 --- a/tests/e2e/helpers/source-control-ai-generators.ts +++ b/tests/e2e/helpers/source-control-ai-generators.ts @@ -14,13 +14,13 @@ async function setCustomGenerator(page: Page, scriptPath: string): Promise } await store.getState().updateSettings({ activeRuntimeEnvironmentId: null, - commitMessageAi: { - ...currentSettings.commitMessageAi, + sourceControlAi: { enabled: true, agentId: 'custom' as const, selectedModelByAgent: {}, selectedThinkingByModel: {}, - customPrompt: '', + instructionsByOperation: {}, + actions: {}, customAgentCommand: `node ${JSON.stringify(scriptPath)}` } }) diff --git a/tests/e2e/helpers/source-control-generation-app.ts b/tests/e2e/helpers/source-control-generation-app.ts new file mode 100644 index 00000000000..2a1d00ca390 --- /dev/null +++ b/tests/e2e/helpers/source-control-generation-app.ts @@ -0,0 +1,21 @@ +import { test as base, expect } from './orca-app' +import { createSeededTestRepo } from './seeded-test-repo' +import { cleanupTestRepository } from '../global-teardown' + +export { expect } + +export const test = base.extend({ + testRepoPath: [ + // oxlint-disable-next-line no-empty-pattern -- Playwright requires destructured fixture arguments. + async ({}, provideFixture) => { + // Generation must not fetch external remotes installed by unrelated specs. + const repoPath = createSeededTestRepo({ publishPath: false }) + try { + await provideFixture(repoPath) + } finally { + cleanupTestRepository(repoPath) + } + }, + { scope: 'worker' } + ] +}) diff --git a/tests/e2e/helpers/ssh-config-host-picker.ts b/tests/e2e/helpers/ssh-config-host-picker.ts index 482eb9ed84e..bad31c818d9 100644 --- a/tests/e2e/helpers/ssh-config-host-picker.ts +++ b/tests/e2e/helpers/ssh-config-host-picker.ts @@ -1,3 +1,4 @@ +import { openSidebarProjectDialog } from './sidebar-project-dialog' /** * Shared helpers for SSH config host picker / import E2E specs. * Prefer role/label locators and user-visible copy over ids / data-*. @@ -72,23 +73,36 @@ export async function closeSettingsPage(page: Page): Promise { export async function closeOpenDialogs(page: Page): Promise { for (let attempt = 0; attempt < 5; attempt += 1) { + // Nested dialogs can finish their exit animations in different frames. + await expect(page.locator('[role="dialog"][data-state="closed"]')).toHaveCount(0, { + timeout: 3_000 + }) const dialogCount = await page.getByRole('dialog').count() if (dialogCount === 0) { return } - const dialog = page.getByRole('dialog').last() - const cancelOrBack = dialog.getByRole('button', { name: /^(Cancel|Back)$/ }) - await ((await cancelOrBack - .first() - .isVisible() - .catch(() => false)) - ? cancelOrBack.first().click() - : page.keyboard.press('Escape')) - await expect - .poll(async () => page.getByRole('dialog').count(), { timeout: 3_000 }) - .toBeLessThan(dialogCount) - .catch(() => undefined) + const dialogId = await page.getByRole('dialog').last().getAttribute('id') + if (!dialogId) { + throw new Error('Open dialog is missing its Radix identity') + } + const dialog = page.locator(`[role="dialog"][id=${JSON.stringify(dialogId)}]`) + const back = dialog.getByRole('button', { name: 'Back', exact: true }) + if (await back.isVisible()) { + await back.click() + // The picker and host form reuse the same Radix dialog. + await expect(back).toBeHidden({ timeout: 3_000 }) + await expect(dialog.getByRole('button', { name: 'Cancel', exact: true })).toBeVisible({ + timeout: 3_000 + }) + continue + } + const cancel = dialog.getByRole('button', { name: 'Cancel', exact: true }) + await ((await cancel.isVisible()) ? cancel.click() : page.keyboard.press('Escape')) + // Hidden Electron windows can park CSS exits before their first compositor frame. + await page.screenshot({ animations: 'disabled' }) + await expect(dialog).toBeHidden({ timeout: 3_000 }) } + await expect(page.getByRole('dialog')).toHaveCount(0, { timeout: 3_000 }) } /** Leave settings / overlays so the main shell (Add Project) is reachable. */ @@ -101,10 +115,7 @@ export async function returnToAppShell(page: Page): Promise { /** Add Project → Host → Add remote host → Add SSH host → form dialog. */ export async function openAddSshHostDialog(page: Page): Promise { await returnToAppShell(page) - await page - .getByRole('button', { name: /Add Project/i }) - .first() - .click() + await openSidebarProjectDialog(page) const addProjectDialog = page.getByRole('dialog', { name: /Add a project/i }) await expect(addProjectDialog).toBeVisible({ timeout: 10_000 }) diff --git a/tests/e2e/helpers/ssh-recovery-input-observation.ts b/tests/e2e/helpers/ssh-recovery-input-observation.ts new file mode 100644 index 00000000000..06b4bd7de3e --- /dev/null +++ b/tests/e2e/helpers/ssh-recovery-input-observation.ts @@ -0,0 +1,53 @@ +import type { Page, TestInfo } from '@playwright/test' +import type { RuntimeTerminalListResult } from '../../../src/shared/runtime-types' + +export async function attachSshRecoveryInputObservation( + page: Page, + testInfo: TestInfo, + targetId: string, + originalPtyId: string, + label: string +): Promise { + const observation = await page.evaluate( + async ({ targetId, originalPtyId }) => { + const state = window.__store?.getState() + const panes = [...(window.__paneManagers?.entries() ?? [])].flatMap(([tabId, manager]) => + manager.getPanes().map((pane) => ({ + tabId, + leafId: pane.leafId, + ptyId: pane.container.dataset.ptyId, + active: manager.getActivePane()?.id === pane.id + })) + ) + let timer: ReturnType | undefined + try { + const runtime = await Promise.race([ + window.api.runtime + .call({ method: 'terminal.list', params: { limit: 50, includeVisualLayouts: false } }) + .then((response) => + response.ok + ? { terminals: (response.result as RuntimeTerminalListResult).terminals } + : { error: response.error } + ), + new Promise<{ error: string }>((resolve) => { + timer = setTimeout(() => resolve({ error: 'Observation timed out' }), 1000) + }) + ]) + return { + originalPtyId, + authority: state?.sshConnectionStates.get(targetId), + activeWorktreeId: state?.activeWorktreeId, + panes, + runtime + } + } finally { + clearTimeout(timer) + } + }, + { targetId, originalPtyId } + ) + await testInfo.attach(`ssh-input-${label}.json`, { + body: JSON.stringify(observation, null, 2), + contentType: 'application/json' + }) +} diff --git a/tests/e2e/helpers/startup-exec-readiness-oracle.ts b/tests/e2e/helpers/startup-exec-readiness-oracle.ts index 577702c8ea0..7aa56e38863 100644 --- a/tests/e2e/helpers/startup-exec-readiness-oracle.ts +++ b/tests/e2e/helpers/startup-exec-readiness-oracle.ts @@ -9,6 +9,7 @@ import type { import { toWebTerminalSurfaceTabId } from '../../../src/shared/terminal-surface-id' import { expect } from './orca-app' import { getTerminalContent, waitForActivePanePtyId } from './terminal' +import { readFreshTerminalInventory } from './terminal-inventory-observation' const RECOVERY_DEADLINE_MS = 8_000 @@ -76,10 +77,6 @@ function count(text: string, marker: string): number { return text.split(marker).length - 1 } -function isTransientPtyLivenessError(error: unknown): boolean { - return error instanceof Error && error.message.includes('terminal_liveness_unavailable') -} - async function expectSingleOwningPty( page: Page, worktreeId: string, @@ -90,24 +87,15 @@ async function expectSingleOwningPty( await expect .poll( async () => { - try { - const listed = await callStartupExecRuntime( - page, - 'terminal.list', - { - worktree: `id:${worktreeId}`, - requireFreshPtyLiveness: true - } - ) - return listed.terminals - .filter((candidate) => candidate.tabId === tabId) - .map((candidate) => ({ handle: candidate.handle, ptyId: candidate.ptyId })) - } catch (error) { - if (isTransientPtyLivenessError(error)) { - return [] - } - throw error - } + const listed = await readFreshTerminalInventory(() => + callStartupExecRuntime(page, 'terminal.list', { + worktree: `id:${worktreeId}`, + requireFreshPtyLiveness: true + }) + ) + return (listed?.terminals ?? []) + .filter((candidate) => candidate.tabId === tabId) + .map((candidate) => ({ handle: candidate.handle, ptyId: candidate.ptyId })) }, { timeout: 30_000 } ) diff --git a/tests/e2e/helpers/terminal-inventory-observation.ts b/tests/e2e/helpers/terminal-inventory-observation.ts new file mode 100644 index 00000000000..0fcefdb23c8 --- /dev/null +++ b/tests/e2e/helpers/terminal-inventory-observation.ts @@ -0,0 +1,14 @@ +import type { RuntimeTerminalListResult } from '../../../src/shared/runtime-types' + +export async function readFreshTerminalInventory( + read: () => Promise +): Promise { + try { + return await read() + } catch (error) { + if (error instanceof Error && error.message.includes('terminal_liveness_unavailable')) { + return null + } + throw error + } +} diff --git a/tests/e2e/live-background-terminal-mount-authority.spec.ts b/tests/e2e/live-background-terminal-mount-authority.spec.ts index ea6ffd1871a..a785454f695 100644 --- a/tests/e2e/live-background-terminal-mount-authority.spec.ts +++ b/tests/e2e/live-background-terminal-mount-authority.spec.ts @@ -23,6 +23,10 @@ import type { } from '../../src/shared/runtime-types' import { PROTOCOL_VERSION } from '../../src/main/daemon/types' import { makePaneKey } from '../../src/shared/stable-pane-id' +import { + buildFakeAgentCommandOverride, + FAKE_AGENT_WINDOWS_SHELL +} from './helpers/fake-agent-command-override' type SpawnEvent = { args: string[]; pid: number } type TerminalIdentity = Pick< @@ -70,6 +74,10 @@ if (process.platform === 'win32') { chmodSync(executable, 0o755) } +const fakeCodexCommand = buildFakeAgentCommandOverride( + path.join(fakeCliDir, process.platform === 'win32' ? 'codex.cmd' : 'codex') +) + const test = base.extend({ launchEnv: [ { @@ -535,23 +543,28 @@ test('adopts runtime-owned agent and Setup PTYs on first mount', async ({ const repoId = added.result.repo.id await expect .poll(() => - orcaPage.evaluate(async (repoId) => { - const state = window.__store?.getState() - await state?.fetchRepos() - const repo = window.__store?.getState().repos.find((candidate) => candidate.id === repoId) - if (!repo) { - return false - } - await window.__store?.getState().updateRepo(repoId, { - hookSettings: { ...repo.hookSettings, setupAgentStartupPolicy: 'start-immediately' } - }) - await window.__store?.getState().updateSettings({ - disabledTuiAgents: [], - setupScriptLaunchMode: 'new-tab', - terminalHiddenViewParking: false - }) - return true - }, repoId) + orcaPage.evaluate( + async ({ repoId, command, windowsShell }) => { + const state = window.__store?.getState() + await state?.fetchRepos() + const repo = window.__store?.getState().repos.find((candidate) => candidate.id === repoId) + if (!repo) { + return false + } + await window.__store?.getState().updateRepo(repoId, { + hookSettings: { ...repo.hookSettings, setupAgentStartupPolicy: 'start-immediately' } + }) + await window.__store?.getState().updateSettings({ + agentCmdOverrides: { codex: command }, + terminalWindowsShell: windowsShell, + disabledTuiAgents: [], + setupScriptLaunchMode: 'new-tab', + terminalHiddenViewParking: false + }) + return true + }, + { repoId, command: fakeCodexCommand, windowsShell: FAKE_AGENT_WINDOWS_SHELL } + ) ) .toBe(true) diff --git a/tests/e2e/multi-client-navigation-isolation.spec.ts b/tests/e2e/multi-client-navigation-isolation.spec.ts index c8a01eff636..1f281a9c264 100644 --- a/tests/e2e/multi-client-navigation-isolation.spec.ts +++ b/tests/e2e/multi-client-navigation-isolation.spec.ts @@ -1,3 +1,4 @@ +import { openSidebarProjectDialog } from './helpers/sidebar-project-dialog' import { execFileSync } from 'node:child_process' import { randomUUID } from 'node:crypto' import { mkdtempSync, rmSync } from 'node:fs' @@ -343,10 +344,7 @@ test('routes Add Project folder browsing through the paired host', async ({ const offer = await createPairingOffer(orcaPage) const client = await openPairedClient(electronApp, offer, visibleWorktreeId) try { - await client - .getByRole('button', { name: /Add Project/i }) - .first() - .click() + await openSidebarProjectDialog(client) const addDialog = client.getByRole('dialog', { name: /Add a project/i }) await expect(addDialog).toBeVisible() await expect(addDialog).not.toContainText('Local Mac') diff --git a/tests/e2e/new-workspace-create-more.spec.ts b/tests/e2e/new-workspace-create-more.spec.ts new file mode 100644 index 00000000000..b166d61dfe5 --- /dev/null +++ b/tests/e2e/new-workspace-create-more.spec.ts @@ -0,0 +1,106 @@ +import { writeFileSync } from 'node:fs' +import { execFileSync } from 'node:child_process' +import { test, expect } from './helpers/orca-app' +import { waitForActiveWorktree, waitForSessionReady } from './helpers/store' + +test.use({ orcaAppExtraEnv: { ORCA_BACKGROUND_LAUNCH: '1' } }) + +test('Create more clears the GitHub PR source before the next worktree', async ({ + electronApp, + orcaPage, + testRepoPath +}, testInfo) => { + await waitForSessionReady(orcaPage) + await waitForActiveWorktree(orcaPage) + const sha = execFileSync('git', ['rev-parse', 'HEAD'], { + cwd: testRepoPath, + encoding: 'utf8' + }).trim() + await electronApp.evaluate(({ ipcMain }, baseBranch) => { + ipcMain.removeHandler('worktrees:resolvePrBase') + ipcMain.handle('worktrees:resolvePrBase', () => ({ baseBranch })) + }, sha) + await orcaPage.evaluate(() => { + const store = window.__store! + const state = store.getState() + store.setState({ settings: { ...state.settings!, defaultTuiAgent: 'blank' } }) + }) + await orcaPage.getByRole('button', { name: 'New workspace', exact: true }).click() + await orcaPage.evaluate(() => { + const store = window.__store! + const repoId = store.getState().repos[0].id + const item = { + id: 'pr-4242', + provider: 'github' as const, + type: 'pr' as const, + number: 4242, + title: 'Fix workspace task reset', + state: 'open' as const, + url: 'https://github.com/acme/app/pull/4242', + labels: [], + updatedAt: '2026-09-01T00:00:00Z', + author: 'e2e', + repoId + } + store.setState({ + getCachedWorkItems: () => [item], + fetchWorkItems: async () => [item], + fetchWorkItemsAcrossRepos: async () => ({ + items: [item], + failedCount: 0, + githubUnavailable: false + }) + }) + }) + const dialog = orcaPage.getByRole('dialog', { name: /Create (Workspace|Worktree)/i }) + const input = dialog.locator('[data-workspace-name-input="true"]') + await input.click() + await orcaPage + .getByRole('option', { name: '#4242 Fix workspace task reset', exact: true }) + .click() + const pill = dialog.locator('[data-workspace-source-pill="true"]') + await expect(pill).toContainText('Fix workspace task reset') + await dialog.getByRole('switch', { name: 'Create more' }).click() + await dialog.getByRole('button', { name: /^Create/ }).click() + await expect(dialog).toBeVisible() + await expect(input).toHaveValue('') + await expect + .poll(() => + orcaPage.evaluate(() => + window + .__store!.getState() + .allWorktrees() + .some((worktree) => worktree.linkedPR === 4242) + ) + ) + .toBe(true) + const cdp = await orcaPage.context().newCDPSession(orcaPage) + const screenshot = await cdp.send('Page.captureScreenshot') + const proofPath = testInfo.outputPath('create-more-result.png') + writeFileSync(proofPath, Buffer.from(screenshot.data, 'base64')) + await testInfo.attach('create-more-result.png', { + path: proofPath, + contentType: 'image/png' + }) + await cdp.detach() + await expect(pill).toHaveCount(0) + await expect(dialog.getByRole('switch', { name: 'Create more' })).toHaveAttribute( + 'aria-checked', + 'true' + ) + await input.fill('next-independent-worktree') + await dialog.getByRole('button', { name: /^Create/ }).click() + await expect + .poll(() => + orcaPage.evaluate(() => { + const worktree = window + .__store!.getState() + .allWorktrees() + .find((entry) => entry.displayName === 'next-independent-worktree') + return worktree ? { linkedPR: worktree.linkedPR, linkedIssue: worktree.linkedIssue } : null + }) + ) + .toEqual({ linkedPR: null, linkedIssue: null }) + await expect(input).toHaveValue('') + await expect(pill).toHaveCount(0) +}) diff --git a/tests/e2e/orchestration-worker-terminal-visibility.spec.ts b/tests/e2e/orchestration-worker-terminal-visibility.spec.ts index af603c8ca1f..32b0013ff5a 100644 --- a/tests/e2e/orchestration-worker-terminal-visibility.spec.ts +++ b/tests/e2e/orchestration-worker-terminal-visibility.spec.ts @@ -11,6 +11,10 @@ import { waitForSessionReady } from './helpers/store' import { waitForActivePaneHookDescriptor, waitForActivePanePtyId } from './helpers/terminal' +import { + buildFakeAgentCommandOverride, + FAKE_AGENT_WINDOWS_SHELL +} from './helpers/fake-agent-command-override' import { RuntimeClient } from '../../src/cli/runtime-client' import type { RuntimeTerminalListResult, RuntimeTerminalRead } from '../../src/shared/runtime-types' @@ -111,6 +115,22 @@ test('worker-start preserves one live inactive worker across workspace re-entry' electronApp }) => { await waitForSessionReady(orcaPage) + await orcaPage.evaluate( + async ({ command, windowsShell }) => { + const state = window.__store!.getState() + await state.updateSettings({ + agentCmdOverrides: { ...state.settings?.agentCmdOverrides, codex: command }, + terminalWindowsShell: windowsShell + }) + }, + { + command: buildFakeAgentCommandOverride( + path.join(fakeCliDir, process.platform === 'win32' ? 'codex.cmd' : 'codex') + ), + windowsShell: FAKE_AGENT_WINDOWS_SHELL + } + ) + const worktreeId = await waitForActiveWorktree(orcaPage) await ensureTerminalVisible(orcaPage) const coordinatorTabId = await getActiveTabId(orcaPage) @@ -160,7 +180,7 @@ test('worker-start preserves one live inactive worker across workspace re-entry' const terminals = await client.call('terminal.list') const workerTerminal = terminals.result.terminals.find( - (terminal) => terminal.title === 'Codex Ready' + (terminal) => terminal.handle === workerHandle ) expect(workerTerminal?.tabId).toBeTruthy() expect(workerTerminal?.leafId).toBeTruthy() diff --git a/tests/e2e/paired-remote-terminal-host-restart-background-sync.spec.ts b/tests/e2e/paired-remote-terminal-host-restart-background-sync.spec.ts index 9855d8d92b0..cf2f96e9c84 100644 --- a/tests/e2e/paired-remote-terminal-host-restart-background-sync.spec.ts +++ b/tests/e2e/paired-remote-terminal-host-restart-background-sync.spec.ts @@ -283,6 +283,24 @@ async function expectTerminalInteractive( } async function moveHostAwayFromWorktree(page: Page, targetWorktreeId: string): Promise { + await expect + .poll( + () => + page.evaluate(async (targetId) => { + const state = window.__store?.getState() + const target = state?.allWorktrees().find((worktree) => worktree.id === targetId) + if (!state || !target) { + return false + } + await state.fetchWorktrees(target.repoId) + return window + .__store!.getState() + .allWorktrees() + .some((worktree) => worktree.repoId === target.repoId && worktree.id !== targetId) + }, targetWorktreeId), + { message: 'Seeded alternate host worktree never loaded' } + ) + .toBe(true) const alternateWorktreeId = await page.evaluate((targetId) => { const state = window.__store?.getState() const alternate = state?.allWorktrees().find((worktree) => worktree.id !== targetId) @@ -423,6 +441,9 @@ test('foregrounds a preserved daemon PTY after the paired host relaunches', asyn expect(reconnectControl.ptyId).not.toBe(target.ptyId) await openClientTab(client.page, worktreeId, reconnectControl.webTabId) await waitForPaneConnected(client.page, reconnectControl.webTabId) + await expect + .poll(() => readPaneContent(client!.page, reconnectControl.webTabId), { timeout: 30_000 }) + .toContain('READY') await expectTerminalInteractive(client, reconnectControl, 'y') } finally { if (client) { diff --git a/tests/e2e/paired-remote-terminal-materialization-reconnect.spec.ts b/tests/e2e/paired-remote-terminal-materialization-reconnect.spec.ts index ef331ee6277..fda18fc74c6 100644 --- a/tests/e2e/paired-remote-terminal-materialization-reconnect.spec.ts +++ b/tests/e2e/paired-remote-terminal-materialization-reconnect.spec.ts @@ -16,6 +16,7 @@ import { launchPairedElectronClient } from './helpers/paired-electron-client' import { getTerminalContent, waitForActivePanePtyId } from './helpers/terminal' +import { readFreshTerminalInventory } from './helpers/terminal-inventory-observation' const scratch = mkdtempSync(path.join(os.tmpdir(), 'orca-paired-materialize-')) const fixturePath = path.join(scratch, 'materialize-terminal.mjs') @@ -316,18 +317,17 @@ async function runMaterializationJourney( await tab.click() await expect.poll(() => getTerminalContent(page), { timeout: 10_000 }).toContain(marker) - const listed = await callRuntime( - page, - environmentId, - 'terminal.list', - { - worktree: `id:${worktreeId}`, - requireFreshPtyLiveness: true - } - ) - expect( - listed.terminals.filter((terminal) => terminal.tabId === created.tab.parentTabId) - ).toHaveLength(1) + await expect + .poll(async () => { + const listed = await readFreshTerminalInventory(() => + callRuntime(page, environmentId, 'terminal.list', { + worktree: `id:${worktreeId}`, + requireFreshPtyLiveness: true + }) + ) + return listed?.terminals.filter((terminal) => terminal.tabId === created.tab.parentTabId) + }) + .toHaveLength(1) await callRuntime(page, environmentId, 'terminal.closeTab', { terminal: replacementHandle }) } @@ -354,15 +354,7 @@ test('materializes a stopped terminal on reconnect from a headed paired host', a } }) -// Why fixme: this journey's fault injection cannot be set up on a headless `orca serve` host. -// `terminal.stopExact` keeps returning terminal_exact_stop_failed because stopAndWait's -// keep-history verification window expires before the parked PTY is observed gone, so the pane -// never reaches pending-handle and the reconnect behavior is never exercised. That precondition -// fails identically on this PR's base, so it is a pre-existing exact-stop defect rather than a -// reconnect-activation one. The recovery behavior itself was confirmed by hand in this topology -// (the host materializes the pending surface and the client rebinds to the replacement PTY); -// re-enable once exact stop settles deterministically against a serve host. -test.fixme('materializes a stopped terminal on reconnect from a headless folder host', async ({ +test('materializes a stopped terminal on reconnect from a headless folder host', async ({ testRepoPath }, testInfo) => { test.setTimeout(150_000) diff --git a/tests/e2e/paired-remote-terminal-stall-recovery.spec.ts b/tests/e2e/paired-remote-terminal-stall-recovery.spec.ts index 4406ebd2f18..2c81b76f077 100644 --- a/tests/e2e/paired-remote-terminal-stall-recovery.spec.ts +++ b/tests/e2e/paired-remote-terminal-stall-recovery.spec.ts @@ -1,3 +1,4 @@ +import { runProcess } from '../../src/shared/child-process/run-process' import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' import os from 'node:os' import path from 'node:path' @@ -101,11 +102,26 @@ async function minimizeHeadedHost(electronApp: ElectronApplication, page: Page): .poll(() => host.evaluate((window) => ({ backgroundThrottling: window.webContents.getBackgroundThrottling(), - minimized: window.isMinimized(), - visible: window.isVisible() + minimized: window.isMinimized() })) ) - .toEqual({ backgroundThrottling: true, minimized: true, visible: false }) + .toEqual({ backgroundThrottling: true, minimized: true }) + // Linux reports isVisible/document visibility differently; the window manager owns iconification. + if (process.platform === 'linux') { + const nativeId = await host.evaluate((window) => window.getNativeWindowHandle().readUInt32LE(0)) + await expect + .poll(async () => { + const result = await runProcess({ + program: 'xprop', + args: ['-id', String(nativeId), '_NET_WM_STATE'], + timeoutMs: 5_000 + }) + return result.stdout + }) + .toContain('_NET_WM_STATE_HIDDEN') + } else { + await expect.poll(() => page.evaluate(() => document.visibilityState)).toBe('hidden') + } } async function restoreHeadedHost(electronApp: ElectronApplication, page: Page): Promise { diff --git a/tests/e2e/paired-skill-installation.spec.ts b/tests/e2e/paired-skill-installation.spec.ts index ac31a81e2f1..0268dbb84a3 100644 --- a/tests/e2e/paired-skill-installation.spec.ts +++ b/tests/e2e/paired-skill-installation.spec.ts @@ -14,7 +14,6 @@ import { type HeadlessPairedRuntimeHost } from './helpers/headless-paired-runtime-host' import { - REMOTE_SKILL_CLOUD_ORIGIN, REMOTE_SKILL_NAME, REMOTE_SKILL_PACKAGE_ID, REMOTE_SKILL_VERSION_ID, @@ -119,13 +118,14 @@ test('installs on a headless serve runtime through the same contract', async ({ }) function cloudClientEnvironment(): Record { + const { origin } = requireCloudFixture() return { - ORCA_ARTIFACTS_API_URL: REMOTE_SKILL_CLOUD_ORIGIN, - ORCA_CLOUD_API_URL: REMOTE_SKILL_CLOUD_ORIGIN, + ORCA_ARTIFACTS_API_URL: origin, + ORCA_CLOUD_API_URL: origin, ORCA_CLOUD_CLIENT_ID: 'skills-e2e-client', ORCA_CLOUD_DEV_AUTH: '1', ORCA_CLOUD_ALLOW_PLAINTEXT_SESSION: '1', - ORCA_SKILL_PACKAGE_DOWNLOAD_ORIGINS: REMOTE_SKILL_CLOUD_ORIGIN + ORCA_SKILL_PACKAGE_DOWNLOAD_ORIGINS: origin } } diff --git a/tests/e2e/paired-web-add-project-unavailable-host.spec.ts b/tests/e2e/paired-web-add-project-unavailable-host.spec.ts index 771feffdde7..a4f7ecc8d05 100644 --- a/tests/e2e/paired-web-add-project-unavailable-host.spec.ts +++ b/tests/e2e/paired-web-add-project-unavailable-host.spec.ts @@ -1,3 +1,4 @@ +import { openSidebarProjectDialog } from './helpers/sidebar-project-dialog' import type { ElectronApplication, Page, TestInfo } from '@stablyai/playwright-test' import { expect, test } from './helpers/orca-app' import { @@ -61,10 +62,7 @@ async function assertCreationActionsDisabled(args: { testInfo: TestInfo topology: 'headed' | 'headless' }): Promise { - await args.page - .getByRole('button', { name: /Add Project/i }) - .first() - .click() + await openSidebarProjectDialog(args.page) const dialog = args.page.getByRole('dialog', { name: /Add a project/i }) await expect(dialog).toBeVisible() const hostPicker = dialog.getByRole('combobox') diff --git a/tests/e2e/pr11346-selected-runtime-add.spec.ts b/tests/e2e/pr11346-selected-runtime-add.spec.ts index 93e49d4c4f2..6225f13c623 100644 --- a/tests/e2e/pr11346-selected-runtime-add.spec.ts +++ b/tests/e2e/pr11346-selected-runtime-add.spec.ts @@ -1,3 +1,4 @@ +import { openSidebarProjectDialog } from './helpers/sidebar-project-dialog' import { rmSync } from 'node:fs' import path from 'node:path' import type { ElectronApplication, Locator, Page, TestInfo } from '@stablyai/playwright-test' @@ -22,10 +23,7 @@ import { } from './pr11346-selected-runtime-identity-oracle' async function selectRuntimeHost(page: Page, runtimeName: string): Promise { - await page - .getByRole('button', { name: /Add Project/i }) - .first() - .click() + await openSidebarProjectDialog(page) const dialog = page.getByRole('dialog', { name: /Add a project/i }) await expect(dialog).toBeVisible() const hostPicker = dialog.getByRole('combobox') diff --git a/tests/e2e/restart-restore-terminal-input.spec.ts b/tests/e2e/restart-restore-terminal-input.spec.ts index 1ceed4254c5..79528fabffe 100644 --- a/tests/e2e/restart-restore-terminal-input.spec.ts +++ b/tests/e2e/restart-restore-terminal-input.spec.ts @@ -239,7 +239,6 @@ test('restored pane recovers input after the daemon un-wedges', async (// oxlint const second = await session.launch() secondApp = second.app - await settleRestoredLaunch(second.page) // Field-fidelity check, not a hard gate: does the pane paint restored // content while its PTY attach cannot complete? That visible-but-dead @@ -258,6 +257,8 @@ test('restored pane recovers input after the daemon un-wedges', async (// oxlint } stoppedDaemonPid = null + // Session readiness requires a daemon response; resume it before waiting for restoration. + await settleRestoredLaunch(second.page) await expectRestoredPaneAcceptsInput( second.page, `daemon wedged during relaunch (painted while wedged: ${paintedWhileWedged}, ` + diff --git a/tests/e2e/right-sidebar-windows-titlebar.spec.ts b/tests/e2e/right-sidebar-windows-titlebar.spec.ts index 1d6d4b8981f..ce39d7de28d 100644 --- a/tests/e2e/right-sidebar-windows-titlebar.spec.ts +++ b/tests/e2e/right-sidebar-windows-titlebar.spec.ts @@ -6,41 +6,19 @@ type RightSidebarHeaderGeometry = { stripTop: number closeTop: number titlebarActivityButtonCount: number + activityButtonCount: number firstButtonCenterHitsFirst: boolean lastButtonCenterHitsLast: boolean } -test.describe('Right sidebar Windows titlebar spacing', () => { - test('top activity buttons render inside the sidebar instead of the titlebar', async ({ - orcaPage - }) => { - await orcaPage.addInitScript(() => { - const userAgent = - 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/146 Safari/537.36' - Object.defineProperty(navigator, 'userAgent', { - get: () => userAgent, - configurable: true - }) - }) - await orcaPage.reload({ waitUntil: 'domcontentloaded' }) - await orcaPage.waitForFunction(() => Boolean(window.__store), null, { timeout: 30_000 }) +test.describe('Right sidebar native titlebar spacing', () => { + test('top activity buttons follow the native desktop chrome layout', async ({ orcaPage }) => { await waitForSessionReady(orcaPage) await waitForActiveWorktree(orcaPage) await ensureTerminalVisible(orcaPage) - await expect - .poll( - async () => - orcaPage.evaluate(() => ({ - hasWindowsUserAgent: navigator.userAgent.includes('Windows'), - hasWindowsTitlebarChrome: Boolean(document.querySelector('.window-controls')) - })), - { - timeout: 5_000, - message: 'Renderer did not switch to the Windows titlebar branch' - } - ) - .toEqual({ hasWindowsUserAgent: true, hasWindowsTitlebarChrome: true }) + const hasDesktopWindowChrome = process.platform !== 'darwin' + expect(await orcaPage.evaluate(() => window.api.platform.get().platform)).toBe(process.platform) await orcaPage.evaluate(() => { const store = window.__store @@ -95,6 +73,7 @@ test.describe('Right sidebar Windows titlebar spacing', () => { stripTop: stripRect.top, closeTop: closeRect.top, titlebarActivityButtonCount, + activityButtonCount: activityButtons.length, firstButtonCenterHitsFirst: elementAtFirstCenter !== null && firstButton.contains(elementAtFirstCenter), lastButtonCenterHitsLast: @@ -117,8 +96,13 @@ test.describe('Right sidebar Windows titlebar spacing', () => { .toBe(true) expect(headerGeometry).not.toBeNull() - expect(headerGeometry!.titlebarActivityButtonCount).toBe(0) - expect(headerGeometry!.stripTop).toBeGreaterThanOrEqual(headerGeometry!.headerBottom) + if (hasDesktopWindowChrome) { + expect(headerGeometry!.titlebarActivityButtonCount).toBe(0) + expect(headerGeometry!.stripTop).toBeGreaterThanOrEqual(headerGeometry!.headerBottom) + } else { + expect(headerGeometry!.titlebarActivityButtonCount).toBe(headerGeometry!.activityButtonCount) + expect(headerGeometry!.stripTop).toBeLessThan(headerGeometry!.headerBottom) + } expect(headerGeometry!.closeTop).toBeLessThan(headerGeometry!.headerBottom) expect(headerGeometry!.firstButtonCenterHitsFirst).toBe(true) expect(headerGeometry!.lastButtonCenterHitsLast).toBe(true) diff --git a/tests/e2e/settings-agent-awake.spec.ts b/tests/e2e/settings-agent-awake.spec.ts index 8a2ad840a14..ebea82a1241 100644 --- a/tests/e2e/settings-agent-awake.spec.ts +++ b/tests/e2e/settings-agent-awake.spec.ts @@ -1,4 +1,5 @@ import { randomUUID } from 'node:crypto' +import { runProcess } from '../../src/shared/child-process/run-process' import type { ElectronApplication, Page } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' import { waitForSessionReady } from './helpers/store' @@ -104,6 +105,19 @@ async function readPowerSaveBlockerProbe( }) } +async function readMacosSleepAssertionPids(electronApp: ElectronApplication): Promise { + const result = await runProcess({ + program: '/usr/bin/pgrep', + args: ['-P', String(electronApp.process().pid), '-f', '^/usr/bin/caffeinate -i -s$'], + maxOutputBytes: 4_096 + }) + if (result.code === 1) { + return [] + } + expect(result.code, result.stderr).toBe(0) + return result.stdout.trim().split(/\s+/).filter(Boolean).map(Number) +} + async function postCodexHookEvent( electronApp: ElectronApplication, options: { @@ -176,7 +190,9 @@ test.describe('Agent awake setting', () => { electronApp, orcaPage }) => { - await installPowerSaveBlockerProbe(electronApp) + if (process.platform !== 'darwin') { + await installPowerSaveBlockerProbe(electronApp) + } await setKeepAwake(orcaPage, true) const tabId = 'e2e-awake-tab' @@ -187,24 +203,33 @@ test.describe('Agent awake setting', () => { eventName: 'UserPromptSubmit' }) - await expect - .poll(async () => await readPowerSaveBlockerProbe(electronApp), { - timeout: 5_000, - message: 'powerSaveBlocker did not start for the working agent' - }) - .toEqual( - expect.objectContaining({ - activeIds: expect.arrayContaining([expect.any(Number)]), - starts: expect.arrayContaining([ - expect.objectContaining({ type: 'prevent-display-sleep' }) - ]) + await expect( + orcaPage.getByRole('button', { name: 'Keep computer awake, Agent · Active' }) + ).toBeVisible() + let startedIds: number[] = [] + if (process.platform === 'darwin') { + // macOS uses an app-owned caffeinate assertion instead of Electron's display blocker. + await expect + .poll(() => readMacosSleepAssertionPids(electronApp), { timeout: 5_000 }) + .not.toEqual([]) + } else { + await expect + .poll(async () => await readPowerSaveBlockerProbe(electronApp), { + timeout: 5_000, + message: 'powerSaveBlocker did not start for the working agent' }) - ) + .toEqual( + expect.objectContaining({ + activeIds: expect.arrayContaining([expect.any(Number)]), + starts: expect.arrayContaining([ + expect.objectContaining({ type: 'prevent-display-sleep' }) + ]) + }) + ) - const startedIds = (await readPowerSaveBlockerProbe(electronApp)).starts.map( - (start) => start.id - ) - expect(startedIds.length).toBeGreaterThan(0) + startedIds = (await readPowerSaveBlockerProbe(electronApp)).starts.map((start) => start.id) + expect(startedIds.length).toBeGreaterThan(0) + } await postCodexHookEvent(electronApp, { paneKey, @@ -212,6 +237,15 @@ test.describe('Agent awake setting', () => { eventName: 'Stop' }) + await expect( + orcaPage.getByRole('button', { name: 'Keep computer awake, Agent · Inactive' }) + ).toBeVisible() + if (process.platform === 'darwin') { + await expect + .poll(() => readMacosSleepAssertionPids(electronApp), { timeout: 5_000 }) + .toEqual([]) + return + } await expect .poll(async () => await readPowerSaveBlockerProbe(electronApp), { timeout: 5_000, diff --git a/tests/e2e/setup-script-import.spec.ts b/tests/e2e/setup-script-import.spec.ts index 180b34f85aa..340258c884b 100644 --- a/tests/e2e/setup-script-import.spec.ts +++ b/tests/e2e/setup-script-import.spec.ts @@ -1,5 +1,5 @@ import { execFileSync } from 'node:child_process' -import { mkdirSync, rmSync, writeFileSync } from 'node:fs' +import { mkdirSync, realpathSync, rmSync, writeFileSync } from 'node:fs' import path from 'node:path' import type { Locator, Page } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' @@ -108,7 +108,7 @@ async function addAndActivateRepo(page: Page, repoPath: string): Promise state.setActiveWorktree(worktree.id) state.setSidebarOpen(true) return addedRepo.id - }, repoPath) + }, realpathSync.native(repoPath)) } async function openRepoSettings(page: Page, repoId: string): Promise { diff --git a/tests/e2e/setup-script-prompt-unreadable-orca-yaml.spec.ts b/tests/e2e/setup-script-prompt-unreadable-orca-yaml.spec.ts index 151a4b02bbc..199694b961a 100644 --- a/tests/e2e/setup-script-prompt-unreadable-orca-yaml.spec.ts +++ b/tests/e2e/setup-script-prompt-unreadable-orca-yaml.spec.ts @@ -1,5 +1,5 @@ import { execFileSync } from 'node:child_process' -import { mkdirSync, rmSync, writeFileSync } from 'node:fs' +import { mkdirSync, realpathSync, rmSync, writeFileSync } from 'node:fs' import path from 'node:path' import type { ElectronApplication, Page } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' @@ -112,17 +112,10 @@ async function addRepoAndActivateMainWorktree( if (!store) { throw new Error('window.__store is not available') } - const normalize = (value: string): string => - value.startsWith('/private/var/') ? value.slice('/private'.length) : value - const state = store.getState() const worktrees = state.worktreesByRepo[targetRepoId] ?? [] - const mainWorktree = worktrees.find( - (entry) => normalize(entry.path) === normalize(targetRepoPath) - ) - const featureWorktree = worktrees.find( - (entry) => normalize(entry.path) === normalize(targetFeaturePath) - ) + const mainWorktree = worktrees.find((entry) => entry.path === targetRepoPath) + const featureWorktree = worktrees.find((entry) => entry.path === targetFeaturePath) if (!mainWorktree || !featureWorktree) { throw new Error( `Missing worktrees for ${targetRepoPath}: ${worktrees.map((entry) => entry.path).join(', ')}` @@ -145,7 +138,11 @@ async function addRepoAndActivateMainWorktree( featureWorktreeId: featureWorktree.id } }, - { targetRepoId: repoId, targetRepoPath: repoPath, targetFeaturePath: featureWorktreePath } + { + targetRepoId: repoId, + targetRepoPath: realpathSync.native(repoPath), + targetFeaturePath: realpathSync.native(featureWorktreePath) + } ) } diff --git a/tests/e2e/source-control-large-file-count.spec.ts b/tests/e2e/source-control-large-file-count.spec.ts index 28a5e7758f2..c8c699b2bd8 100644 --- a/tests/e2e/source-control-large-file-count.spec.ts +++ b/tests/e2e/source-control-large-file-count.spec.ts @@ -20,12 +20,18 @@ import type { ElectronApplication, Page } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' import { waitForSessionReady } from './helpers/store' +import { + hasCapturedGitStatusRetry, + installGitStatusRetryBarrier, + restoreGitStatusRetryHandler +} from './helpers/git-status-retry-barrier' import { createLargeFileCountRepo, removeLargeFileCountRepo, removeLargeFileCountUntrackedTree } from './large-file-count-fixtures' import { DEFAULT_GIT_STATUS_LIMIT } from '../../src/shared/git-status-limit' +import { RIGHT_SIDEBAR_MIN_WIDTH } from '../../src/renderer/src/components/right-sidebar/right-sidebar-width' // Matches the large-diff freeze budget: a blocking stall past 1s is the // "UI becomes unresponsive" symptom reported in #8013. @@ -411,12 +417,17 @@ test.describe('Source Control large file count (#8013)', () => { rendererWorkingSetMb: { before: workingSetBeforeMb, after: workingSetAfterMb } }) - const tooManyChangesBanner = orcaPage.getByText('Too many changes detected.', { - exact: false - }) + const tooManyChangesBanner = orcaPage.getByTestId('too-many-changes-banner') await expect(tooManyChangesBanner).toBeVisible() if (process.env.ORCA_LARGE_FILE_SCREENSHOT_PATH) { - await orcaPage.screenshot({ path: process.env.ORCA_LARGE_FILE_SCREENSHOT_PATH }) + // Narrowest supported sidebar is where the banner layout is worst. + await orcaPage.evaluate((minWidth) => { + window.__store?.getState().setRightSidebarWidth(minWidth) + document.documentElement.classList.add('dark') + }, RIGHT_SIDEBAR_MIN_WIDTH) + await tooManyChangesBanner.screenshot({ + path: process.env.ORCA_LARGE_FILE_SCREENSHOT_PATH + }) } expect(measurement.didHitLimit).toBe(true) @@ -434,13 +445,17 @@ test.describe('Source Control large file count (#8013)', () => { ) expect(hugeState).not.toBeNull() - // Why: watcher refreshes stay parked while huge; the visible Retry is the - // explicit recovery path after the underlying change count drops. - removeLargeFileCountUntrackedTree(fixture.repoPath) - await expect(tooManyChangesBanner).toBeVisible() - const retryButton = tooManyChangesBanner.locator('..').getByRole('button', { name: 'Retry' }) + const retryButton = tooManyChangesBanner.getByRole('button', { name: 'Retry' }) await expect(retryButton).toBeVisible() - await retryButton.click() + // Keep automatic refreshes from removing Retry before its real request starts. + await installGitStatusRetryBarrier(electronApp, fixture.repoPath) + try { + await retryButton.click() + await expect.poll(() => hasCapturedGitStatusRetry(electronApp)).toBe(true) + removeLargeFileCountUntrackedTree(fixture.repoPath) + } finally { + await restoreGitStatusRetryHandler(electronApp) + } await expect(tooManyChangesBanner).not.toBeVisible() await expect .poll(() => diff --git a/tests/e2e/source-control-pr-generation-switch.spec.ts b/tests/e2e/source-control-pr-generation-switch.spec.ts index 58091cd3cf8..47b4acb1b5d 100644 --- a/tests/e2e/source-control-pr-generation-switch.spec.ts +++ b/tests/e2e/source-control-pr-generation-switch.spec.ts @@ -1,7 +1,7 @@ import type { Page, TestInfo } from '@stablyai/playwright-test' import { mkdirSync, readFileSync, writeFileSync } from 'node:fs' import path from 'node:path' -import { test, expect } from './helpers/orca-app' +import { test, expect } from './helpers/source-control-generation-app' import { waitForActiveWorktree, waitForSessionReady } from './helpers/store' import { createBranchCommit, diff --git a/tests/e2e/source-control-pr-linked-issue-ai.spec.ts b/tests/e2e/source-control-pr-linked-issue-ai.spec.ts index 625d19d58b8..be6966b7f3a 100644 --- a/tests/e2e/source-control-pr-linked-issue-ai.spec.ts +++ b/tests/e2e/source-control-pr-linked-issue-ai.spec.ts @@ -1,7 +1,7 @@ import { rmSync } from 'node:fs' import os from 'node:os' import path from 'node:path' -import { test, expect } from './helpers/orca-app' +import { test, expect } from './helpers/source-control-generation-app' import { createBranchCommit, openSourceControl, diff --git a/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts b/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts index e19a090df4d..f4c02d04c94 100644 --- a/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts +++ b/tests/e2e/ssh-codex-display-artifacts-repro.spec.ts @@ -48,7 +48,7 @@ import { resetWebglAndCaptureGraySlabAnalysis } from './terminal-webgl-reset-cap const RUN_DOCKER_SSH = process.env.ORCA_E2E_SSH_DOCKER === '1' const RUN_REAL_REMOTE_CODEX = process.env.ORCA_E2E_REAL_REMOTE_CODEX === '1' -const EXPECT_NO_ARTIFACTS = process.env.ORCA_E2E_EXPECT_NO_CODEX_ARTIFACTS === '1' +const EXPECT_NO_ARTIFACTS = process.env.ORCA_E2E_EXPECT_NO_CODEX_ARTIFACTS !== '0' const CAPTURE_WHILE_REMOTE_TUI_RUNNING = process.env.ORCA_E2E_CAPTURE_WHILE_REMOTE_TUI_RUNNING === '1' const HIDE_UNTIL_REMOTE_TUI_DONE = process.env.ORCA_E2E_HIDE_UNTIL_REMOTE_TUI_DONE === '1' diff --git a/tests/e2e/ssh-codex-repro-remote-fixtures.ts b/tests/e2e/ssh-codex-repro-remote-fixtures.ts index 3ee48187e7c..14597bc084a 100644 --- a/tests/e2e/ssh-codex-repro-remote-fixtures.ts +++ b/tests/e2e/ssh-codex-repro-remote-fixtures.ts @@ -135,7 +135,7 @@ async function insertCodexHistory(frame) { const phase = String(frame).padStart(4, '0') + '.' + index await write('\\r\\n') await write(\`\\x1b[48;2;72;72;72m\\x1b[K\`) - await write(\`\\x1b[38;2;220;220;220;48;2;72;72;72m\${pad('gpt-5.5 high · ~/code/pr-12250-migration-compare-move-baseprice-claim · /ps to view · /stop to close ' + phase, width)}\\x1b[0m\`) + await write(\`\\x1b[38;2;220;220;220;48;2;72;72;72m\${pad('gpt-5.5 high · ' + phase + ' · ~/code/pr-12250-migration-compare-move-baseprice-claim · /ps to view · /stop to close', width)}\\x1b[0m\`) } await write('\\x1b[r') await write(\`\\x1b[\${viewportBottom};1H\`) @@ -176,7 +176,7 @@ for (let frame = 0; frame < ${REMOTE_CODEX_FIXTURE_FRAMES}; frame += 1) { await reverseIndexCodexHistory(frame) } if (frame % 9 === 0) { - await grayScrollLine(\`gpt-5.5 high · ~/code/pr-12250-migration-compare-move-baseprice-claim · /ps to view · /stop to close \${frame}\`) + await grayScrollLine(\`gpt-5.5 high · \${frame} · ~/code/pr-12250-migration-compare-move-baseprice-claim · /ps to view · /stop to close\`) } await sleep(${REMOTE_CODEX_FIXTURE_FRAME_DELAY_MS}) } diff --git a/tests/e2e/ssh-docker-half-open-link.spec.ts b/tests/e2e/ssh-docker-half-open-link.spec.ts index c5b6f715dc9..d77eba3a264 100644 --- a/tests/e2e/ssh-docker-half-open-link.spec.ts +++ b/tests/e2e/ssh-docker-half-open-link.spec.ts @@ -68,7 +68,7 @@ test.describe('Docker SSH half-open link', () => { const ptyId = await waitForActivePanePtyId(orcaPage, 60_000) const runId = String(Date.now()) - await execInTerminal(orcaPage, ptyId, `echo LIVE_${runId}`) + await execInTerminal(orcaPage, ptyId, `printf 'LIVE_%s\\n' ${runId}`) await waitForTerminalOutput(orcaPage, `LIVE_${runId}`, 60_000) expect(await readSshStatus(orcaPage, remote.targetId)).toBe('connected') @@ -78,13 +78,15 @@ test.describe('Docker SSH half-open link', () => { const frozenAt = Date.now() let verdict: string | null = 'connected' - while (Date.now() - frozenAt < LOST_VERDICT_BUDGET_MS) { - verdict = await readSshStatus(orcaPage, remote.targetId) - if (verdict !== 'connected') { - break - } - await orcaPage.waitForTimeout(1_000) - } + await expect + .poll( + async () => { + verdict = await readSshStatus(orcaPage, remote.targetId) + return verdict + }, + { timeout: LOST_VERDICT_BUDGET_MS, message: 'frozen host remained connected' } + ) + .not.toBe('connected') const verdictMs = Date.now() - frozenAt console.log( `[half-open] ${JSON.stringify({ verdict, verdictMs, budgetMs: LOST_VERDICT_BUDGET_MS })}` @@ -105,7 +107,7 @@ test.describe('Docker SSH half-open link', () => { .poll(() => readSshStatus(orcaPage, remote.targetId), { timeout: 120_000 }) .toBe('connected') const recoveredPtyId = await waitForActivePanePtyId(orcaPage, 60_000) - await execInTerminal(orcaPage, recoveredPtyId, `echo RECOVERED_${runId}`) + await execInTerminal(orcaPage, recoveredPtyId, `printf 'RECOVERED_%s\\n' ${runId}`) await waitForTerminalOutput(orcaPage, `RECOVERED_${runId}`, 90_000) } finally { if (target && paused) { diff --git a/tests/e2e/ssh-docker-transport-drop-recovery.spec.ts b/tests/e2e/ssh-docker-transport-drop-recovery.spec.ts index e18336d12c5..c64761ede80 100644 --- a/tests/e2e/ssh-docker-transport-drop-recovery.spec.ts +++ b/tests/e2e/ssh-docker-transport-drop-recovery.spec.ts @@ -1,9 +1,10 @@ import path from 'node:path' import { readFileSync } from 'node:fs' -import type { ElectronApplication, Page } from '@playwright/test' +import type { ElectronApplication } from '@playwright/test' import { test, expect } from './helpers/orca-app' import { DEFAULT_LOCAL_ORCA_PROFILE_ID } from '../../src/shared/orca-profiles' import { sshRemotePtyLeaseAllowsReattach, type SshRemotePtyLease } from '../../src/shared/ssh-types' +import { toRelaySshPtyId } from '../../src/shared/ssh-pty-id' import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' import { execInTerminal, @@ -18,7 +19,10 @@ import { startDockerSshRelayTarget, type DockerSshRelayTarget } from './helpers/docker-ssh-relay-target' -import { connectDockerSshRelayTarget } from './helpers/docker-ssh-relay-connection' +import { + connectDockerSshRelayTarget, + recoverDockerSshRelayAfterFault +} from './helpers/docker-ssh-relay-connection' import { clearDockerSshRelayFaults, dropDockerSshRelayTransport, @@ -26,6 +30,8 @@ import { withStalledDockerSshRelayTarget } from './helpers/docker-ssh-relay-faults' +import { attachSshRecoveryInputObservation } from './helpers/ssh-recovery-input-observation' + const RUN_DOCKER_SSH = process.env.ORCA_E2E_SSH_DOCKER === '1' /** @@ -46,13 +52,6 @@ const RUN_DOCKER_SSH = process.env.ORCA_E2E_SSH_DOCKER === '1' * with only the first cannot tell a resume from a silent cold start * (docs/reference/ssh-execution-boundary.md). */ -async function readSshStatus(orcaPage: Page, targetId: string) { - return orcaPage.evaluate( - (targetId) => window.__store?.getState().sshConnectionStates.get(targetId)?.status ?? null, - targetId - ) -} - /** * Every lease `reattachKnownPtys` would feed to `pty.attach` on the next connect, read from the * durable store rather than from the renderer — leases are main-owned and never published. @@ -122,7 +121,9 @@ test.describe('SSH transport drop recovery', () => { enableDockerSshRelayTargetShellTitle(target) await waitForSessionReady(orcaPage) await waitForActiveWorktree(orcaPage) - const remote = await connectDockerSshRelayTarget(orcaPage, target) + const remote = await connectDockerSshRelayTarget(orcaPage, target, { + relayGracePeriodSeconds: 0 + }) await ensureTerminalVisible(orcaPage, 45_000) await waitForActiveTerminalManager(orcaPage, 60_000) const ptyId = await waitForActivePanePtyId(orcaPage, 60_000) @@ -134,20 +135,12 @@ test.describe('SSH transport drop recovery', () => { await execInTerminal(orcaPage, ptyId, `printf 'DROP_MARKER_%s\\n' ${markerSuffix}`) await waitForTerminalOutput(orcaPage, marker, 30_000) - const dropped = dropDockerSshRelayTransport(target) - expect(dropped, 'no live SSH connection was found to drop').toBeGreaterThan(0) - - // Nothing below calls ssh.connect(). Recovery has to come from the client's own ladder, - // which is the behaviour users depend on and the thing a scripted reconnect never exercised. - await expect - .poll(() => readSshStatus(orcaPage, remote.targetId), { - timeout: 120_000, - message: 'SSH target never returned to connected after the transport was dropped' - }) - .toBe('connected') + await recoverDockerSshRelayAfterFault(orcaPage, remote.targetId, () => { + expect(dropDockerSshRelayTransport(target!)).toBeGreaterThan(0) + }) await waitForActiveTerminalManager(orcaPage, 60_000) - await waitForActivePanePtyId(orcaPage, 60_000) + expect(await waitForActivePanePtyId(orcaPage, 60_000)).toBe(ptyId) // The pane must still show what it had. A blank pane here is the reported bug. await waitForTerminalOutput(orcaPage, marker, 60_000) @@ -170,10 +163,7 @@ test.describe('SSH transport drop recovery', () => { } }) - // Fixme: fails in CI on its first real run — the pane keeps its PTY and repaints, but a command - // run after the flood produces no output within the poll budget. Same shape as #18018 (deaf pane - // after a stalled host resumes), and not caused by this spec. Tracked there; the three verdict - // assertions around it stay enforced. + // #18018: local authority-aware recovery still loses the flooded pane's relay channel. test.fixme('stays bounded when a disconnected shell floods its pty', async ({ orcaPage }, testInfo) => { @@ -195,7 +185,9 @@ test.describe('SSH transport drop recovery', () => { enableDockerSshRelayTargetShellTitle(target) await waitForSessionReady(orcaPage) await waitForActiveWorktree(orcaPage) - const remote = await connectDockerSshRelayTarget(orcaPage, target) + const remote = await connectDockerSshRelayTarget(orcaPage, target, { + relayGracePeriodSeconds: 0 + }) await ensureTerminalVisible(orcaPage, 45_000) await waitForActiveTerminalManager(orcaPage, 240_000) const ptyId = await waitForActivePanePtyId(orcaPage, 240_000) @@ -214,18 +206,12 @@ test.describe('SSH transport drop recovery', () => { await execInTerminal( orcaPage, ptyId, - `yes "$(printf 'ORCA_%s' FLOOD_LINE)" | head -c 48000000; echo FLOODED` + `yes "$(printf 'ORCA_%s' FLOOD_LINE)" | head -c 48000000; printf 'FLOO%s\\n' DED` ) await waitForTerminalOutput(orcaPage, 'ORCA_FLOOD_LINE', 30_000, 20_000) - const dropped = dropDockerSshRelayTransport(target) - expect(dropped).toBeGreaterThan(0) - - await expect - .poll(() => readSshStatus(orcaPage, remote.targetId), { - timeout: 120_000, - message: 'SSH target never returned to connected' - }) - .toBe('connected') + await recoverDockerSshRelayAfterFault(orcaPage, remote.targetId, () => { + expect(dropDockerSshRelayTransport(target!)).toBeGreaterThan(0) + }) await waitForActiveTerminalManager(orcaPage, 240_000) // Why a generous ceiling: this is an OOM guard, not a memory budget. Unbounded retention of @@ -236,6 +222,9 @@ test.describe('SSH transport drop recovery', () => { `relay grew ${afterRssKb - baselineRssKb}KB after 48MB of undeliverable output` ).toBeLessThan(200_000) + // Wait for the finite producer to finish before sending a shell command behind it. + await waitForTerminalOutput(orcaPage, 'FLOODED', 120_000, 20_000) + // And the session must still be usable, not merely alive. const markerSuffix = Date.now() const marker = `FLOOD_AFTER_${markerSuffix}` @@ -273,7 +262,9 @@ test.describe('SSH transport drop recovery', () => { enableDockerSshRelayTargetShellTitle(target) await waitForSessionReady(orcaPage) await waitForActiveWorktree(orcaPage) - const remote = await connectDockerSshRelayTarget(orcaPage, target) + const remote = await connectDockerSshRelayTarget(orcaPage, target, { + relayGracePeriodSeconds: 0 + }) await ensureTerminalVisible(orcaPage, 45_000) await waitForActiveTerminalManager(orcaPage, 60_000) const ptyId = await waitForActivePanePtyId(orcaPage, 60_000) @@ -283,15 +274,9 @@ test.describe('SSH transport drop recovery', () => { await execInTerminal(orcaPage, ptyId, `printf 'KILL_MARKER_%s\\n' ${markerSuffix}`) await waitForTerminalOutput(orcaPage, marker, 30_000) - const killed = killDockerSshRelayDaemon(target) - expect(killed, 'no relay process was found to kill').toBeGreaterThan(0) - - await expect - .poll(() => readSshStatus(orcaPage, remote.targetId), { - timeout: 120_000, - message: 'SSH target never returned to connected after the relay was killed' - }) - .toBe('connected') + await recoverDockerSshRelayAfterFault(orcaPage, remote.targetId, () => { + expect(killDockerSshRelayDaemon(target!)).toBeGreaterThan(0) + }) await waitForActiveTerminalManager(orcaPage, 60_000) // The verdict, expressed as the only thing a user can observe: the pane is now backed by a @@ -345,7 +330,9 @@ test.describe('SSH transport drop recovery', () => { enableDockerSshRelayTargetShellTitle(target) await waitForSessionReady(orcaPage) await waitForActiveWorktree(orcaPage) - const remote = await connectDockerSshRelayTarget(orcaPage, target) + const remote = await connectDockerSshRelayTarget(orcaPage, target, { + relayGracePeriodSeconds: 0 + }) await ensureTerminalVisible(orcaPage, 45_000) await waitForActiveTerminalManager(orcaPage, 60_000) await waitForActivePanePtyId(orcaPage, 60_000) @@ -354,35 +341,37 @@ test.describe('SSH transport drop recovery', () => { const generations: string[][] = [] for (let generation = 1; generation <= 5; generation++) { - expect( - killDockerSshRelayDaemon(target), - 'no relay process was found to kill' - ).toBeGreaterThan(0) - await expect - .poll(() => readSshStatus(orcaPage, remote.targetId), { - timeout: 120_000, - message: `SSH target never reconnected after relay kill ${generation}` - }) - .toBe('connected') + const previousPtyId = await waitForActivePanePtyId(orcaPage, 60_000) + await recoverDockerSshRelayAfterFault(orcaPage, remote.targetId, () => { + expect( + killDockerSshRelayDaemon(target!), + 'no relay process was found to kill' + ).toBeGreaterThan(0) + }) await waitForActiveTerminalManager(orcaPage, 120_000) - // The pane must be usable again before the count is meaningful: recovery is what mints the - // successor lease that retires the generation before it. + // Transport status can still be connected while the pane retains its old binding. + await expect + .poll(() => waitForActivePanePtyId(orcaPage, 60_000).catch(() => previousPtyId), { + timeout: 120_000, + message: `pane kept its old PTY binding after relay kill ${generation}` + }) + .not.toBe(previousPtyId) const ptyId = await waitForActivePanePtyId(orcaPage, 120_000) - const marker = `LEASE_GEN_${generation}_${Date.now()}` - await execInTerminal(orcaPage, ptyId, `printf '%s\\n' ${marker}`) + const markerSuffix = `${generation}_${Date.now()}` + const marker = `LEASE_GEN_${markerSuffix}` + await execInTerminal(orcaPage, ptyId, `printf 'LEASE_GEN_%s\\n' ${markerSuffix}`) await waitForTerminalOutput(orcaPage, marker, 60_000) try { await expect - .poll(() => readReattachablePtyIds(userDataDir, remote.targetId).length, { + .poll(() => readReattachablePtyIds(userDataDir, remote.targetId), { timeout: 60_000 }) - .toBe(1) + .toEqual([toRelaySshPtyId(remote.targetId, ptyId)]) } catch (error) { - // Why re-thrown with the rows: the count alone cannot say WHICH predecessor stayed - // reattachable, and the user-data dir is torn down before the report is read. + // Preserve lease ownership diagnostics before the user-data directory is removed. throw new Error( - `reattachable lease count never settled at 1 in generation ${generation}; leases: ${describeSshLeases(userDataDir, remote.targetId)}`, + `reattachable leases never settled at the active PTY ${ptyId} in generation ${generation}; leases: ${describeSshLeases(userDataDir, remote.targetId)}`, { cause: error } ) } @@ -418,7 +407,7 @@ test.describe('SSH transport drop recovery', () => { enableDockerSshRelayTargetShellTitle(target) await waitForSessionReady(orcaPage) await waitForActiveWorktree(orcaPage) - await connectDockerSshRelayTarget(orcaPage, target) + await connectDockerSshRelayTarget(orcaPage, target, { relayGracePeriodSeconds: 0 }) await ensureTerminalVisible(orcaPage, 45_000) await waitForActiveTerminalManager(orcaPage, 60_000) const ptyId = await waitForActivePanePtyId(orcaPage, 60_000) @@ -446,38 +435,64 @@ test.describe('SSH transport drop recovery', () => { } }) - /** - * Known broken on main, kept as the reproduction. The verdict test above passes: after a 30s - * freeze the pane keeps its PTY and repaints its scrollback. What does not come back is the - * shell — a command run afterwards produces no output within 60s, so the pane is live-looking and - * deaf. Measured twice at `waitForTerminalOutput(STALL_AFTER_…)`, and it reproduces unchanged - * with the reattach-token/delivery-ownership fix applied, so that is not the cause. - * - * Split out rather than folded into the test above so the `unverifiable` verdict stays enforced - * in CI instead of being masked by this failure. - */ - test.fixme('accepts input again after a frozen host resumes', async ({ orcaPage }, testInfo) => { + // #18018: wait for the recovered authority before input; a retained manager can still be disconnected. + test('accepts input again after a frozen host resumes', async ({ orcaPage }, testInfo) => { test.slow() let target: DockerSshRelayTarget | null = null + let observationTarget: { targetId: string; ptyId: string } | undefined try { target = startDockerSshRelayTarget(testInfo) enableDockerSshRelayTargetShellTitle(target) await waitForSessionReady(orcaPage) await waitForActiveWorktree(orcaPage) - await connectDockerSshRelayTarget(orcaPage, target) + const remote = await connectDockerSshRelayTarget(orcaPage, target, { + relayGracePeriodSeconds: 0 + }) await ensureTerminalVisible(orcaPage, 45_000) await waitForActiveTerminalManager(orcaPage, 60_000) const ptyId = await waitForActivePanePtyId(orcaPage, 60_000) - await withStalledDockerSshRelayTarget(target, async () => { - await orcaPage.waitForTimeout(30_000) + observationTarget = { targetId: remote.targetId, ptyId } + const beforeSuffix = Date.now() + await execInTerminal(orcaPage, ptyId, `printf 'STALL_BEFORE_%s\\n' ${beforeSuffix}`) + await waitForTerminalOutput(orcaPage, `STALL_BEFORE_${beforeSuffix}`, 60_000) + await attachSshRecoveryInputObservation( + orcaPage, + testInfo, + remote.targetId, + ptyId, + 'before-freeze' + ) + + await recoverDockerSshRelayAfterFault(orcaPage, remote.targetId, async () => { + await withStalledDockerSshRelayTarget(target!, async () => { + await orcaPage.waitForTimeout(30_000) + }) }) await waitForActiveTerminalManager(orcaPage, 60_000) const afterSuffix = Date.now() const afterMarker = `STALL_AFTER_${afterSuffix}` await execInTerminal(orcaPage, ptyId, `printf 'STALL_AFTER_%s\\n' ${afterSuffix}`) + await attachSshRecoveryInputObservation( + orcaPage, + testInfo, + remote.targetId, + ptyId, + 'after-write' + ) await waitForTerminalOutput(orcaPage, afterMarker, 60_000) + } catch (error) { + if (observationTarget) { + await attachSshRecoveryInputObservation( + orcaPage, + testInfo, + observationTarget.targetId, + observationTarget.ptyId, + 'failure-before-cleanup' + ).catch(() => undefined) + } + throw error } finally { if (target) { clearDockerSshRelayFaults(target) diff --git a/tests/e2e/ssh-skill-installation.spec.ts b/tests/e2e/ssh-skill-installation.spec.ts index a102478fb62..794883b2cd1 100644 --- a/tests/e2e/ssh-skill-installation.spec.ts +++ b/tests/e2e/ssh-skill-installation.spec.ts @@ -10,7 +10,6 @@ import { import { connectDockerSshRelayTarget } from './helpers/docker-ssh-relay-connection' import { waitForActiveWorktree, waitForSessionReady } from './helpers/store' import { - REMOTE_SKILL_CLOUD_ORIGIN, REMOTE_SKILL_NAME, REMOTE_SKILL_PACKAGE_ID, REMOTE_SKILL_VERSION_ID, @@ -25,13 +24,19 @@ const REMOTE_FOLDER = '/tmp/orca-skill-folder-workspace' let cloud: RemoteSkillCloudFixture | null = null test.use({ - orcaAppExtraEnv: { - ORCA_ARTIFACTS_API_URL: REMOTE_SKILL_CLOUD_ORIGIN, - ORCA_CLOUD_API_URL: REMOTE_SKILL_CLOUD_ORIGIN, - ORCA_CLOUD_CLIENT_ID: 'skills-e2e-client', - ORCA_CLOUD_DEV_AUTH: '1', - ORCA_CLOUD_ALLOW_PLAINTEXT_SESSION: '1', - ORCA_SKILL_PACKAGE_DOWNLOAD_ORIGINS: REMOTE_SKILL_CLOUD_ORIGIN + // oxlint-disable-next-line no-empty-pattern -- The server starts in beforeAll before this test fixture runs. + orcaAppExtraEnv: async ({}, provideEnv) => { + if (!cloud) { + throw new Error('Skill cloud fixture unavailable') + } + await provideEnv({ + ORCA_ARTIFACTS_API_URL: cloud.origin, + ORCA_CLOUD_API_URL: cloud.origin, + ORCA_CLOUD_CLIENT_ID: 'skills-e2e-client', + ORCA_CLOUD_DEV_AUTH: '1', + ORCA_CLOUD_ALLOW_PLAINTEXT_SESSION: '1', + ORCA_SKILL_PACKAGE_DOWNLOAD_ORIGINS: cloud.origin + }) } }) diff --git a/tests/e2e/tab-rename.spec.ts b/tests/e2e/tab-rename.spec.ts index 6e7f0a7fdc1..30cdb9175a9 100644 --- a/tests/e2e/tab-rename.spec.ts +++ b/tests/e2e/tab-rename.spec.ts @@ -126,7 +126,7 @@ test.describe('Tab Rename (Inline)', () => { expect(originalTitle.length).toBeGreaterThan(0) await tabLocatorByTitle(orcaPage, originalTitle).click({ button: 'right' }) - await orcaPage.getByRole('menuitem', { name: 'Change Title', exact: true }).click() + await orcaPage.getByRole('menuitem', { name: /^Change Title(?:\s|$)/ }).click() const renameInput = orcaPage.getByRole('textbox', { name: `Rename tab ${originalTitle}`, diff --git a/tests/e2e/terminal-codex-home.spec.ts b/tests/e2e/terminal-codex-home.spec.ts index 1f85a4f4c9b..3364152d38c 100644 --- a/tests/e2e/terminal-codex-home.spec.ts +++ b/tests/e2e/terminal-codex-home.spec.ts @@ -1,3 +1,5 @@ +import { mkdirSync, writeFileSync } from 'node:fs' +import path from 'node:path' import { test, expect } from './helpers/orca-app' import { execInTerminal, @@ -27,7 +29,42 @@ test.describe('Terminal Codex runtime home', () => { await ensureTerminalVisible(orcaPage) }) - test('terminal process receives the Orca-managed Codex home', async ({ orcaPage }) => { + test('terminal process receives the selected account Codex home', async ({ + electronApp, + orcaPage + }) => { + const userData = await electronApp.evaluate(({ app }) => app.getPath('userData')) + const accountId = 'e2e-terminal-home' + const managedHomePath = path.join(userData, 'codex-accounts', accountId, 'home') + mkdirSync(managedHomePath, { recursive: true }) + writeFileSync(path.join(managedHomePath, '.orca-managed-home'), `${accountId}\n`) + writeFileSync( + path.join(managedHomePath, 'auth.json'), + JSON.stringify({ OPENAI_API_KEY: 'e2e-placeholder' }) + ) + await orcaPage.evaluate( + async ({ accountId, managedHomePath }) => { + const state = window.__store!.getState() + await state.updateSettings({ + codexManagedAccounts: [ + { + id: accountId, + email: 'terminal-home@example.invalid', + managedHomePath, + createdAt: 1, + updatedAt: 1, + lastAuthenticatedAt: 1 + } + ], + activeCodexManagedAccountId: accountId, + activeCodexManagedAccountIdsByRuntime: { host: accountId, wsl: {} } + }) + const tab = state.createTab(state.activeWorktreeId!) + state.setActiveTab(tab.id) + state.setActiveTabType('terminal') + }, + { accountId, managedHomePath } + ) await waitForActiveTerminalManager(orcaPage) const ptyId = await waitForActivePanePtyId(orcaPage) const marker = `__ORCA_CODEX_HOME_E2E_${Date.now()}__` @@ -43,17 +80,10 @@ test.describe('Terminal Codex runtime home', () => { .poll( async () => { probe = readCodexHomeProbe(await getTerminalContent(orcaPage), marker) - return Boolean( - probe?.codexHome && - probe.orcaCodexHome && - probe.codexHome === probe.orcaCodexHome && - /[\\/]codex-runtime-home[\\/]home$/.test(probe.codexHome) - ) + return probe }, - { timeout: 15_000, message: 'Terminal did not expose Orca-managed Codex home env' } + { timeout: 15_000, message: 'Terminal did not expose the selected Codex account home' } ) - .toBe(true) - - expect(probe?.codexHome).toBe(probe?.orcaCodexHome) + .toEqual({ codexHome: managedHomePath, orcaCodexHome: managedHomePath }) }) }) diff --git a/tests/e2e/terminal-hangul-terminating-digit-native.spec.ts b/tests/e2e/terminal-hangul-terminating-digit-native.spec.ts index 2fb90a9c992..4344f94adaf 100644 --- a/tests/e2e/terminal-hangul-terminating-digit-native.spec.ts +++ b/tests/e2e/terminal-hangul-terminating-digit-native.spec.ts @@ -3,20 +3,18 @@ * the pty. Written to reproduce #15299, where a digit typed straight after a Hangul syllable was * dropped under Wayland but not under X11. * - * THIS DOES NOT RUN IN CI. It is gated on ORCA_E2E_NATIVE_IBUS_HANGUL=1 and needs a compositor - * session that CI does not have, so it is a manual reproduction harness rather than coverage. - * That is stated plainly because this repo already carries native IME specs that are skipped - * everywhere and were mistaken for coverage they never provided. + * CI runs the default xdotool injector under X11, checking exact Hangul-plus-digit PTY bytes. + * That path passed even before the Wayland fix; it does not prove #15299 is fixed. + * Reproducing #15299 still requires the nested Wayland session below. * - * To run it, on a machine with gnome-shell and ibus-hangul: + * To run the Wayland reproduction on a machine with gnome-shell and ibus-hangul: * * Xvfb :65 -extension GLX & * DISPLAY=:65 gnome-shell --nested --wayland # nested, NOT --headless * ORCA_E2E_NATIVE_IBUS_HANGUL=1 ORCA_E2E_IME_INJECTOR=nested npx playwright test \ * tests/e2e/terminal-hangul-terminating-digit-native.spec.ts * - * Eight things that decide whether a run is real or a silent false negative, each of which cost a - * failed attempt: + * Nested Wayland prerequisites: * * - Nested, not headless. A headless mutter never answers RemoteDesktop.CreateSession, so there * is no way to inject input; nested makes the whole compositor an X window that xdotool can @@ -45,6 +43,7 @@ import { mkdirSync, writeFileSync } from 'node:fs' import path from 'node:path' import type { Page, TestInfo } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' +import { appendImeEngagementReceipt } from './terminal-ime-engagement-receipt' import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' import { focusActiveTerminalInput, @@ -232,6 +231,11 @@ test.describe('Hangul terminating digit @headful', () => { } receivedBytes = await waitForTerminalImeBytes(page, reader, 20_000) + expect(receivedBytes.map((hex) => Buffer.from(hex, 'hex').toString('utf8'))).toEqual( + Array.from({ length: REPETITIONS }, () => `${EXPECTED_LINE}\n`) + ) + const trace = await readTerminalImeBoundaryTrace(page) + appendImeEngagementReceipt(testInfo.title, trace) } finally { await writeEvidence(page, testInfo, 'hangul-terminating-digit', { expectedHex, @@ -243,8 +247,5 @@ test.describe('Hangul terminating digit @headful', () => { await sendToTerminal(page, ptyId, '\x03').catch(() => undefined) removeTerminalImeByteReader(reader) } - expect(receivedBytes.map((hex) => Buffer.from(hex, 'hex').toString('utf8'))).toEqual( - Array.from({ length: REPETITIONS }, () => `${EXPECTED_LINE}\n`) - ) }) }) diff --git a/tests/e2e/terminal-pane-divider-capture-loss.spec.ts b/tests/e2e/terminal-pane-divider-capture-loss.spec.ts index 5f3bfca177b..8120c2a60e7 100644 --- a/tests/e2e/terminal-pane-divider-capture-loss.spec.ts +++ b/tests/e2e/terminal-pane-divider-capture-loss.spec.ts @@ -1,4 +1,4 @@ -import type { ElectronApplication, Page } from '@stablyai/playwright-test' +import type { Page } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' import { splitActiveTerminalPane, @@ -22,32 +22,6 @@ type DividerGeometry = { test.use({ seedTestRepo: false }) -async function setFullscreen(electronApp: ElectronApplication, page: Page): Promise { - await expect - .poll(async () => { - try { - return await electronApp.evaluate(({ BrowserWindow }) => { - const window = BrowserWindow.getAllWindows()[0] - if (!window) { - return false - } - if (window.isMinimized()) { - window.restore() - } - window.show() - window.focus() - window.setFullScreen(true) - return window.isFullScreen() - }) - } catch { - return false - } - }) - .toBe(true) - await expect.poll(() => page.evaluate(() => innerWidth >= 1000 && innerHeight >= 700)).toBe(true) - await page.waitForTimeout(1200) -} - async function addTestRepo(page: Page, repoPath: string): Promise { const repoId = await page.evaluate(async (path) => { const result = await window.api.repos.add({ path }) @@ -122,17 +96,21 @@ function gridsMatch(geometry: DividerGeometry): boolean { } test('@headful keeps resizing after the divider loses pointer capture', async ({ - electronApp, orcaPage, testRepoPath }, testInfo) => { - await setFullscreen(electronApp, orcaPage) + // Keep the 260px drag above the fit floor regardless of the CI display resolution. + await orcaPage.setViewportSize({ width: 1600, height: 1000 }) await addTestRepo(orcaPage, testRepoPath) await ensureTerminalVisible(orcaPage, 30_000) await waitForActiveTerminalManager(orcaPage, 30_000) await splitActiveTerminalPane(orcaPage, 'vertical') await waitForPaneCount(orcaPage, 2, 30_000) + await expect + .poll(async () => (await readDividerGeometry(orcaPage)).second.width) + .toBeGreaterThan(400) + const divider = orcaPage.locator('.pane-divider.is-vertical').first() await expect(divider).toBeVisible() const box = await divider.boundingBox() @@ -170,11 +148,11 @@ test('@headful keeps resizing after the divider loses pointer capture', async ({ } element.releasePointerCapture(pointerId) }) + // Pending capture changes are dispatched with the next pointer event. + await orcaPage.mouse.move(startX + 260, startY, { steps: 10 }) await expect .poll(() => divider.evaluate((element) => Number(element.dataset.captureLossCount ?? '0'))) .toBe(1) - - await orcaPage.mouse.move(startX + 260, startY, { steps: 10 }) await orcaPage.mouse.up() await expect.poll(async () => gridsMatch(await readDividerGeometry(orcaPage))).toBe(true) const after = await readDividerGeometry(orcaPage) diff --git a/tests/e2e/terminal-reattach-mouse-mode-leak.spec.ts b/tests/e2e/terminal-reattach-mouse-mode-leak.spec.ts index cbdae675a49..8afca0fb7b5 100644 --- a/tests/e2e/terminal-reattach-mouse-mode-leak.spec.ts +++ b/tests/e2e/terminal-reattach-mouse-mode-leak.spec.ts @@ -35,6 +35,7 @@ import { discoverActivePtyId, execInTerminal, waitForActiveTerminalManager, + waitForActivePanePtyId, waitForPaneCount, waitForTerminalOutput } from './helpers/terminal' @@ -146,6 +147,10 @@ test.describe('reattach mouse-mode leak', () => { await ensureTerminalVisible(secondLaunch.page) await waitForActiveTerminalManager(secondLaunch.page, 30_000) await waitForPaneCount(secondLaunch.page, 1, 30_000) + // Live output is released only after reattach replay has finished. + const reattachedPtyId = await waitForActivePanePtyId(secondLaunch.page) + await execInTerminal(secondLaunch.page, reattachedPtyId, 'echo ORCA_REATTACHED_$((21+21))') + await waitForTerminalOutput(secondLaunch.page, 'ORCA_REATTACHED_42', 15_000) // The reattach replay re-arms mouse via rehydrate, then the reset must // clear it. Poll until it settles to 'none' (times out if the reset @@ -247,10 +252,7 @@ test.describe('reattach mouse-mode leak', () => { return { afterReattach, classAfterArm, - armedReports, - // Whether the reattached pane dynamically bound xterm mouse reporting - // at all — class and listener attach together, so either signal proves it. - armedMouseReporting: classAfterArm || armedReports > 0 + armedReports } } finally { disposable.dispose() @@ -261,16 +263,6 @@ test.describe('reattach mouse-mode leak', () => { expect(probe.afterReattach.mode).toBe('none') expect(probe.afterReattach.hasEnableMouseClass).toBe(false) expect(probe.afterReattach.reports).toBe(0) - // Why: the positive control needs the reattached pane to dynamically bind - // xterm's browser MouseService. Some headless CI renderers never do on a warm - // reattach — the core mouseTrackingMode still flips but no DOM class/listener - // attaches — so arming is impossible and the probe can't run. Skip there, - // matching the pane-manager/shell guards above; the reset invariant stays - // covered by repro-7329 + pty-connection unit tests and this suite on macOS. - test.skip( - !probe.armedMouseReporting, - 'Reattached pane does not dynamically bind xterm mouse reporting in this environment' - ) // Positive control proves the motion probe genuinely detects reports. expect(probe.classAfterArm).toBe(true) expect(probe.armedReports).toBeGreaterThan(0) diff --git a/tests/e2e/terminal-scroll-intent-follow.spec.ts b/tests/e2e/terminal-scroll-intent-follow.spec.ts index c4d8b171fed..ae2dfa15466 100644 --- a/tests/e2e/terminal-scroll-intent-follow.spec.ts +++ b/tests/e2e/terminal-scroll-intent-follow.spec.ts @@ -168,6 +168,7 @@ async function injectQueuedWriteThenType(page: Page, paneKey: string): Promise { const injectionTarget = window as Window & { __terminalPtyDataInjection?: { inject: (paneKey: string, data: string) => boolean } + __releaseScrollIntentTestWrite?: () => void } const state = window.__store?.getState() const worktreeId = state?.activeWorktreeId @@ -184,38 +185,44 @@ async function injectQueuedWriteThenType(page: Page, paneKey: string): Promise void } | null } = { write: null } + const heldWrites: { data: string; callback?: () => void }[] = [] terminal.write = ((data: string, callback?: () => void) => { - holder.write = { data, callback } + heldWrites.push({ data, callback }) }) as typeof terminal.write + injectionTarget.__releaseScrollIntentTestWrite = () => { + terminal.write = originalWrite + delete injectionTarget.__releaseScrollIntentTestWrite + for (const held of heldWrites) { + originalWrite.call(terminal, held.data, held.callback) + } + } try { const payload = '\x1b[?2026h\r\x1b[2KWorking in-flight\x1b[?2026l' if (!injectionTarget.__terminalPtyDataInjection?.inject(targetPaneKey, payload)) { throw new Error('PTY injector unavailable') } + if (heldWrites.length === 0) { + throw new Error('Foreground terminal write was not captured') + } const textarea = pane.container.querySelector('.xterm-helper-textarea') if (!textarea) { throw new Error('xterm helper textarea unavailable') } textarea.focus() - const event = new KeyboardEvent('keydown', { - bubbles: true, - cancelable: true, - key: 'x', - code: 'KeyX' - }) - Object.defineProperty(event, 'keyCode', { configurable: true, value: 88 }) - Object.defineProperty(event, 'which', { configurable: true, value: 88 }) - textarea.dispatchEvent(event) - } finally { - terminal.write = originalWrite + } catch (error) { + injectionTarget.__releaseScrollIntentTestWrite() + throw error } - const heldWrite = holder.write - if (!heldWrite) { - throw new Error('Foreground terminal write was not captured') - } - originalWrite.call(terminal, heldWrite.data, heldWrite.callback) }, paneKey) + try { + await page.keyboard.press('x') + } finally { + await page.evaluate(() => { + ;( + window as Window & { __releaseScrollIntentTestWrite?: () => void } + ).__releaseScrollIntentTestWrite?.() + }) + } } async function startStreamingFixturePhase1(page: Page): Promise { @@ -308,5 +315,6 @@ test.describe('terminal scroll intent keeps following output', () => { { timeout: 5_000, intervals: [25] } ) .toBe(0) + await waitForMarkerAtBottom(orcaPage, 'STREAM_PHASE2_DONE') }) }) diff --git a/tests/e2e/terminal-send-agent-prompt-submit.spec.ts b/tests/e2e/terminal-send-agent-prompt-submit.spec.ts index 3567cb1d8a2..c1a602dd9d1 100644 --- a/tests/e2e/terminal-send-agent-prompt-submit.spec.ts +++ b/tests/e2e/terminal-send-agent-prompt-submit.spec.ts @@ -58,6 +58,7 @@ async function createFakeCodexTerminal( if (!worktree) { throw new Error(`runtime did not register ${testRepoPath}`) } + rmSync(fixtureReport, { force: true }) const created = await client.call<{ terminal: { handle: string } }>('terminal.create', { worktree: `id:${worktree.id}`, command: [fakeCodexCommand, ...args].join(' '), diff --git a/tests/e2e/workspace-board-lane-virtualization.spec.ts b/tests/e2e/workspace-board-lane-virtualization.spec.ts index 69a18e06937..36da44dd357 100644 --- a/tests/e2e/workspace-board-lane-virtualization.spec.ts +++ b/tests/e2e/workspace-board-lane-virtualization.spec.ts @@ -307,7 +307,6 @@ test.describe('Workspace board lane virtualization', () => { }) test('selects the full lane across a single large marquee scroll jump', async ({ orcaPage }) => { - test.skip(true, 'Quarantined by https://github.com/stablyai/orca/issues/12415') const statusId = 'virtual-marquee' const emptyStatusId = 'virtual-marquee-start' await orcaPage.evaluate( @@ -380,32 +379,36 @@ test.describe('Workspace board lane virtualization', () => { } // Why: CI can overlay individual lane pixels, so choose a live board-owned point. - const startPoint = await emptyLaneScroll.evaluate((element) => { - const ignored = [ - '[data-workspace-board-card-id]', - 'a', - 'button', - 'input', - 'select', - 'textarea', - '[role="button"]', - '[role="menu"]', - '[role="menuitem"]' - ].join(',') - const rect = element.getBoundingClientRect() - for (let y = Math.ceil(rect.top) + 6; y <= Math.floor(rect.top) + 40; y += 6) { - for (let x = Math.ceil(rect.left) + 8; x <= Math.floor(rect.right) - 8; x += 8) { - const target = document.elementFromPoint(x, y) - if ( - target?.closest('[data-workspace-board-selection-surface]') && - !target.closest(ignored) - ) { - return { x, y } + const findStartPoint = () => + emptyLaneScroll.evaluate((element) => { + const ignored = [ + '[data-workspace-board-card-id]', + 'a', + 'button', + 'input', + 'select', + 'textarea', + '[role="button"]', + '[role="menu"]', + '[role="menuitem"]' + ].join(',') + const rect = element.getBoundingClientRect() + for (let y = Math.ceil(rect.top) + 6; y <= Math.floor(rect.top) + 40; y += 6) { + for (let x = Math.ceil(rect.left) + 8; x <= Math.floor(rect.right) - 8; x += 8) { + const target = document.elementFromPoint(x, y) + if ( + target?.closest('[data-workspace-board-selection-surface]') && + !target.closest(ignored) + ) { + return { x, y } + } } } - } - return null - }) + return null + }) + // The board's clip animation can expose cards before the empty lane accepts pointer hits. + await expect.poll(findStartPoint).not.toBeNull() + const startPoint = await findStartPoint() expect(startPoint, 'the empty start lane must expose board-owned space').not.toBeNull() if (!startPoint) { throw new Error('Expected empty board space for the marquee start') diff --git a/tests/e2e/worktree-scroll-to-current.spec.ts b/tests/e2e/worktree-scroll-to-current.spec.ts index 19c51005cfe..07f9f87d05c 100644 --- a/tests/e2e/worktree-scroll-to-current.spec.ts +++ b/tests/e2e/worktree-scroll-to-current.spec.ts @@ -1,3 +1,5 @@ +import { mkdirSync } from 'node:fs' +import { runProcess } from '../../src/shared/child-process/run-process' import type { Page } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' import { waitForActiveWorktree, waitForSessionReady } from './helpers/store' @@ -39,22 +41,65 @@ test.describe('Reveal active workspace button', () => { // the "outside the virtualized window" test below. test('clears sidebar filters before revealing a hidden current workspace', async ({ - orcaPage - }) => { + orcaPage, + testRepoPath + }, testInfo) => { + const filterRepoPath = testInfo.outputPath('filter-repo') + mkdirSync(filterRepoPath, { recursive: true }) + for (const args of [ + ['init', filterRepoPath], + [ + '-C', + filterRepoPath, + '-c', + 'user.name=E2E', + '-c', + 'user.email=e2e@test.local', + 'commit', + '--allow-empty', + '-m', + 'Filter fixture' + ] + ]) { + const result = await runProcess({ program: 'git', args }) + expect(result.code, result.stderr).toBe(0) + } + const filterRepoId = await orcaPage.evaluate(async (repoPath) => { + const result = await window.api.repos.add({ path: repoPath }) + if ('error' in result) { + throw new Error(result.error) + } + return result.repo.id + }, filterRepoPath) + await expect + .poll(() => + orcaPage.evaluate(async (id) => { + await window.__store!.getState().fetchRepos() + return window.__store!.getState().repos.some((repo) => repo.id === id) + }, filterRepoId) + ) + .toBe(true) await prepareSidebarForScrollTest(orcaPage) - const renderedOptions = orcaPage.locator('[data-worktree-sidebar] [role="option"]') - await expect(renderedOptions).toHaveCount(2) - - const targetId = await renderedOptions.last().getAttribute('data-worktree-id') + // Other specs can add worktrees to the shared repository before this test runs. + const targetId = await orcaPage.evaluate((repoPath) => { + const state = window.__store!.getState() + const repo = state.repos.find((candidate) => candidate.path === repoPath) + return repo + ? state.worktreesByRepo[repo.id]?.find( + (worktree) => worktree.branch === 'refs/heads/e2e-secondary' + )?.id + : undefined + }, testRepoPath) if (!targetId) { - throw new Error('Bottom workspace row did not expose a data-worktree-id') + throw new Error('Seeded secondary worktree is missing') } const targetRows = orcaPage.locator( `[data-worktree-sidebar] [data-worktree-id=${JSON.stringify(targetId)}]` ) const targetRow = targetRows.first() + await expect(targetRows.and(orcaPage.getByRole('option'))).toHaveCount(1) const revealButton = orcaPage.getByRole('button', { name: 'Reveal active workspace' }) await orcaPage.evaluate((targetId) => { @@ -78,20 +123,17 @@ test.describe('Reveal active workspace button', () => { }, targetId) await expect(targetRow).toHaveAttribute('aria-current', 'page') - await orcaPage.evaluate(() => { - const store = window.__store - if (!store) { - throw new Error('window.__store is not available') - } - store.getState().setFilterRepoIds(['__filtered_repo__']) - }) - - // Why: the filter's row-hiding side effect is covered deterministically by - // visible-worktrees.test.ts. Asserting an empty DOM here over-specifies an - // incidental render-settle state that flakes under the shared page; the - // contract under test is that reveal clears the filter (asserted below). + // Catalog refreshes prune nonexistent IDs, so use a real repo to keep the filter applied. + await orcaPage.evaluate((repoId) => { + window.__store!.getState().setFilterRepoIds([repoId]) + }, filterRepoId) + await expect(targetRows).toHaveCount(0) await revealButton.click() + await orcaPage + .getByRole('dialog', { name: 'Reveal hidden workspace?' }) + .getByRole('button', { name: 'Clear filters and reveal' }) + .click() await expect(targetRow).toBeVisible() await expect(targetRow).toHaveAttribute('data-scroll-reveal-highlight', 'true') diff --git a/tests/tools/google-signin-ua-probe.cjs b/tests/tools/google-signin-ua-probe.cjs index 7adf06bb7a0..70c834b3d9d 100644 --- a/tests/tools/google-signin-ua-probe.cjs +++ b/tests/tools/google-signin-ua-probe.cjs @@ -10,8 +10,8 @@ const MODES = new Set([ 'electron-fixed', 'firefox-auth', 'firefox-fixed', - // Replicates the SHIPPED app exactly (setupClientHintsOverride + - // applyGoogleAuthUserAgent): Firefox UA is written to the WebContents on auth + // Replicates the app as it shipped before the UA rewrite was removed + // (cleaned Chrome-shaped session UA + the Google auth Firefox switch): Firefox UA is written to the WebContents on auth // navs and to the request header only for auth-host URLs; every other request // keeps whatever UA the WebContents carries. Logs incoming vs outgoing // identity for ALL requests to expose cross-host mismatches during the flow. @@ -185,7 +185,7 @@ app.whenReady().then(async () => { if (mode === 'app-fixed' && currentUa === identities.firefox) { removeClientHints(headers) } else { - // Real setupClientHintsOverride builds Chrome hints once from the + // The retired client-hints rewrite built Chrome hints once from the // session's cleaned UA (a closure), never from the per-request UA. applyChromeClientHints(headers, identities.cleaned) } diff --git a/tests/tools/repro-terminal-send-submit.mjs b/tests/tools/repro-terminal-send-submit.mjs index 7e1c0152012..41d2464cbbc 100644 --- a/tests/tools/repro-terminal-send-submit.mjs +++ b/tests/tools/repro-terminal-send-submit.mjs @@ -186,7 +186,9 @@ async function parentMain() { const expectBlocked = hasFlag('expect-blocked') const providedHandle = argValue('terminal') await mkdir(tempDir, { recursive: true }) - await rm(reportPath, { force: true }) + if (!providedHandle) { + await rm(reportPath, { force: true }) + } let handle = providedHandle if (!handle) { diff --git a/tests/tools/win-crash-survival-e2e/README.md b/tests/tools/win-crash-survival-e2e/README.md index beb56cf8c67..e0e773767c4 100644 --- a/tests/tools/win-crash-survival-e2e/README.md +++ b/tests/tools/win-crash-survival-e2e/README.md @@ -13,8 +13,8 @@ orphaned and PowerShell hard-crashed with a `0xE9` "No process is on the other end of the pipe" `FailFast`. Root cause: the terminal **daemon** (which hosts the ConPTYs) died together with the main process, severing the console pipe. -The fix re-architected the daemon into a standalone, relocated -`orca-terminal-daemon.exe` (see +The fix re-architected the daemon into a standalone daemon host relocated out of +the install dir (see [`src/main/daemon/daemon-host-relocation.ts`](../../src/main/daemon/daemon-host-relocation.ts)) that is spawned **detached** and **survives main-process death**. diff --git a/tests/tools/win-crash-survival-e2e/crash-step.mjs b/tests/tools/win-crash-survival-e2e/crash-step.mjs index 43dc18f04cd..3c7d80f51e1 100644 --- a/tests/tools/win-crash-survival-e2e/crash-step.mjs +++ b/tests/tools/win-crash-survival-e2e/crash-step.mjs @@ -4,7 +4,7 @@ // daemon (which hosts the ConPTYs) died with it, severing the console pipe, and // PowerShell hard-crashed with a 0xE9 "No process is on the other end of the // pipe" FailFast. The fix relocates the daemon into a standalone, detached -// orca-terminal-daemon.exe that SURVIVES main death (src/main/daemon/ +// host process outside the install dir that SURVIVES main death (src/main/daemon/ // daemon-host-relocation.ts). This module reproduces the crash and scans for the // pwsh FailFast that must no longer occur. diff --git a/tests/tools/win-crash-survival-e2e/run.mjs b/tests/tools/win-crash-survival-e2e/run.mjs index 4f9d8b242b0..68d8a7a3bd6 100644 --- a/tests/tools/win-crash-survival-e2e/run.mjs +++ b/tests/tools/win-crash-survival-e2e/run.mjs @@ -5,7 +5,7 @@ // process is on the other end of the pipe" FailFast, because the terminal daemon // (hosting the ConPTYs) died together with the main process and severed the // console pipe. The fix relocates the daemon into a standalone, detached -// orca-terminal-daemon.exe that survives main death (src/main/daemon/ +// host process outside the install dir that survives main death (src/main/daemon/ // daemon-host-relocation.ts). win-update-e2e proves the daemon survives a // Windows UPDATE; this harness proves it survives a CRASH of the main process. //