Merge remote-tracking branch 'origin/main' into HEAD

# Conflicts:
#	src/main/codex/codex-app-server-session.ts
This commit is contained in:
Merge Sim
2026-09-06 01:33:40 -07:00
1597 changed files with 102100 additions and 14576 deletions
+13 -1
View File
@@ -8,7 +8,19 @@
/src/cli/bundled-skill-guides.ts text eol=lf
# Bundled plugin trees are byte-hashed; CRLF checkout would break the pinned hash.
/resources/plugins/** text eol=lf
# pnpm hashes every patch byte-for-byte, so a CRLF checkout breaks the install.
# Relay assets are copied verbatim into the bundle and hashed byte-for-byte into
# .version, which names the immutable remote install dir. A CRLF checkout makes a
# Windows-built client disagree with a mac/Linux-built one on the same release,
# so one host ends up with two relay trees (#17886 review).
/config/relay-assets/** text eol=lf
# Pin the bytes so a patch reads and diffs identically on every host. It is NOT
# what makes the hash right: pnpm hashes a patch LF-normalized, so a CRLF checkout
# cannot change it. Believing otherwise put a hand-computed raw digest in the
# lockfile twice and broke every install (#17886).
# These files are stored LF, which is not always the encoding they were written
# against -- @vscode/windows-process-tree ships CRLF sources -- so any code that
# runs `git apply` on one must force `-c core.autocrlf=input` rather than trust
# the host's setting. See config/scripts/windows-process-tree-gyp-rebuild.mjs.
/config/patches/*.patch -text
# The xterm bundle hunks also make a diff nobody can read; review the hand-written
# source patch under xterm-src/ instead. The sibling patches stay diffable.
@@ -77,14 +77,6 @@ runs:
;;
esac
# pnpm's bundled gyp_main.py is not executable on fresh Linux runners.
- name: Use external node-gyp
if: runner.os == 'Linux' && inputs.native-runtime != 'none'
shell: bash
run: |
npm install -g node-gyp@11.5.0
echo "npm_config_node_gyp=$(npm root -g)/node-gyp/bin/node-gyp.js" >> "$GITHUB_ENV"
- name: Prepare dependency install
shell: bash
run: |
@@ -175,6 +167,22 @@ runs:
node_modules/.pnpm/@vscode+windows-process-tree@*/node_modules/@vscode/windows-process-tree/build
key: native-modules-${{ runner.os }}-${{ steps.native-cache-scope.outputs.scope }}-${{ runner.arch }}-${{ inputs.native-runtime }}-node${{ steps.requested-node.outputs.node-version || steps.default-node.outputs.node-version }}-${{ hashFiles('pnpm-lock.yaml', '.github/actions/install-node-dependencies/action.yml', 'config/scripts/ensure-native-runtime.mjs', 'config/scripts/rebuild-native-deps.mjs', 'config/patches/node-pty@1.1.0.patch', 'config/patches/@vscode__windows-process-tree@0.8.0.patch') }}
# pnpm's bundled gyp_main.py is not executable on fresh Linux runners.
- name: Use external node-gyp
if: runner.os == 'Linux' && inputs.native-runtime != 'none'
shell: bash
env:
NATIVE_RUNTIME: ${{ inputs.native-runtime }}
NATIVE_CACHE_HIT: ${{ steps.native-cache-restore.outputs.cache-hit || steps.native-cache-restore-only.outputs.cache-hit }}
run: |
# A cache hit can contain unusable addons; probe before skipping the rebuild toolchain.
if [ "$NATIVE_RUNTIME" = node ] && [ "$NATIVE_CACHE_HIT" = true ] &&
node config/scripts/ensure-native-runtime.mjs --check-only; then
exit 0
fi
npm install -g node-gyp@11.5.0
echo "npm_config_node_gyp=$(npm root -g)/node-gyp/bin/node-gyp.js" >> "$GITHUB_ENV"
- name: Prepare native runtime
if: inputs.native-runtime != 'none'
shell: bash
@@ -0,0 +1,26 @@
#!/usr/bin/env bash
set -euo pipefail
openbox --sm-disable > /tmp/orca-e2e-window-manager.log 2>&1 &
wm_pid=$!
cleanup() {
kill "$wm_pid" 2>/dev/null || true
wait "$wm_pid" 2>/dev/null || true
}
trap cleanup EXIT
ready=false
for attempt in {1..100}; do
if xprop -root _NET_SUPPORTING_WM_CHECK 2>/dev/null | rg -q 'window id # 0x[1-9a-fA-F]'; then
ready=true
break
fi
if ! kill -0 "$wm_pid" 2>/dev/null; then
cat /tmp/orca-e2e-window-manager.log
exit 1
fi
sleep 0.1
done
if [ "$ready" != true ]; then
echo 'Window manager did not acquire the Xvfb root window' >&2
exit 1
fi
"$@"
+8 -2
View File
@@ -127,9 +127,12 @@ jobs:
esac
# Bare: a work-tree repo refuses to fetch over its own checked-out
# branch. tree:0 keeps the fetch to the commit graph — no trees, no
# blobs — so this stays cheap next to the build it fronts.
# blobs — so this stays cheap next to the build it fronts. reftable
# because this repo has branches that differ only in casing, and the
# files backend cannot store both on a case-insensitive runner disk —
# it fails the entire fetch, not just the one ref.
scratch="$RUNNER_TEMP/vet-requested-ref"
git init -q --bare "$scratch"
git init -q --bare --ref-format=reftable "$scratch"
git -C "$scratch" fetch -q --filter=tree:0 "$REPO_URL" '+refs/heads/*:refs/heads/*' '+refs/tags/*:refs/tags/*'
# Branch first to keep actions/checkout's old tie-break: bare
# rev-parse would prefer the tag when a branch shares its name.
@@ -157,6 +160,9 @@ jobs:
- name: Checkout the requested ref
uses: actions/checkout@v6
env:
# Full-history checkout must also preserve case-twin branch and tag names.
GIT_DEFAULT_REF_FORMAT: reftable
with:
# Why an input at all rather than just github.ref: the whole point is to
# build code that has not landed, and the workflow definition itself
@@ -91,7 +91,11 @@ jobs:
test -n "${CAPACITY_SERVICE_ACCOUNT}"
test -n "${DIRECTOR_RUNTIME_SERVICE_ACCOUNT}"
# Full history: the monitor evidence this job verifies is sealed at an ancestor commit,
# and the provenance check fails closed on a commit a shallow clone left out.
- uses: actions/checkout@v4
with:
fetch-depth: 0
- uses: pnpm/action-setup@v4
with: { package_json_file: cloud/package.json }
@@ -177,11 +181,12 @@ jobs:
env:
ORCA_RELAY_ADMIN_ID_TOKEN: ${{ steps.deploy-auth.outputs.id_token }}
run: |
RETRY_ARGS=()
if test "${WAVE_INDEX}" != 0; then RETRY_ARGS=(--retry-freshness); fi
# Freshness-only failures are publish lag, not health, on every wave
# including the first; the CLI still caps the retry at the wave's
# evidence-age budget, so this cannot mutate on aged evidence.
pnpm incident:relay-preflight -- \
--state-file "${OUTPUT_DIRECTORY}/relay-${MONITOR_RUN_ID}-dry-run.state.json" \
--wave-index "${WAVE_INDEX}" "${RETRY_ARGS[@]}"
--wave-index "${WAVE_INDEX}" --retry-freshness
- name: Require durable rehome disabled and exact selector
env:
@@ -271,10 +276,22 @@ jobs:
env:
ORCA_RELAY_ADMIN_ID_TOKEN: ${{ steps.deploy-auth.outputs.id_token }}
run: |
CURRENT_RUNTIME="$(curl --fail-with-body --max-time 30 \
--request POST "${CELL_ORIGIN}/v1/admin/runtime-status" \
--header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \
--header 'Content-Type: application/json' --data '{"v":1}')"
# A single transient 5xx (LB warm-up behind a fresh instance) must not
# fail a canary; 4xx (auth, generation mismatch) still fails fast.
admin_post() {
local out="${RUNNER_TEMP}/$1.json"
if ! curl --fail-with-body --max-time 30 \
--retry 3 --retry-delay 2 --retry-connrefused --output "${out}" \
--request POST "$2" \
--header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \
--header 'Content-Type: application/json' --data "$3"; then
cat "${out}" >&2
return 1
fi
cat "${out}"
}
CURRENT_RUNTIME="$(admin_post current-runtime \
"${CELL_ORIGIN}/v1/admin/runtime-status" '{"v":1}')"
# A rollback that failed between template apply and admission restore
# leaves the cell already on the rollback image; resume from that
# state instead of demanding the pre-rollback predecessor.
@@ -370,11 +387,9 @@ jobs:
if .regionalRehomeProtocol == null then "regionalRehomeProtocol" else empty end
] | if length > 0 then "runtime predecessor normalized legacy fields=" + join(",") else empty end' \
<<< "${CURRENT_RUNTIME}"
CURRENT_DIRECTOR_STATUS="$(curl --fail-with-body --max-time 30 \
--request POST "${DIRECTOR_ORIGIN}/v1/admin/cell-status" \
--header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \
--header 'Content-Type: application/json' \
--data "$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")"
CURRENT_DIRECTOR_STATUS="$(admin_post current-cell-status \
"${DIRECTOR_ORIGIN}/v1/admin/cell-status" \
"$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")"
SOURCE_INCARNATION="$(jq -er '.status.runtime.cellIncarnation' \
<<< "${CURRENT_DIRECTOR_STATUS}")"
if test "${ROLLBACK_RESUME}" = true && ! jq -e \
@@ -418,13 +433,13 @@ jobs:
# result's generation is authoritative either way.
ISOLATE_RESULT="$(node dev/scripts/prepare-relay-production-capacity-canary.mjs \
--director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \
--cell-id "${TARGET_CELL_ID}" --mode isolate)"
--cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode isolate)"
echo "${ISOLATE_RESULT}"
ISOLATE_GENERATION="$(jq -er '.generation' <<< "${ISOLATE_RESULT}")"
echo "SELECTOR_GENERATION_AFTER_ISOLATE=${ISOLATE_GENERATION}" >> "${GITHUB_ENV}"
node dev/scripts/prepare-relay-production-capacity-canary.mjs \
--director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \
--cell-id "${TARGET_CELL_ID}" --mode drain
--cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode drain
node dev/scripts/verify-relay-capacity-transition.mjs \
--director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \
--cell-id "${TARGET_CELL_ID}" --hard-cap "${EXPECTED_HARD_CAP}" \
@@ -486,6 +501,7 @@ jobs:
--rollback-image "${DESIRED_IMAGE}" \
--rehome-director-service-account "${DIRECTOR_RUNTIME_SERVICE_ACCOUNT}" \
--rehome-audience https://relay.onorca.dev/v1/admin/host-drain \
--regional-rehome-protocol "${DESIRED_REHOME_PROTOCOL}" \
| jq -e '.changes == 2' >/dev/null
fi
gcloud compute instance-groups managed wait-until "${MIG_NAME}" --stable \
@@ -511,7 +527,8 @@ jobs:
--unobserved-bound "${EXPECTED_UNOBSERVED_BOUND}" --image "${DESIRED_IMAGE}" \
--rollback-image "${IMAGE_REPOSITORY}@${CURRENT_IMAGE_DIGEST}" \
--rehome-director-service-account "${DIRECTOR_RUNTIME_SERVICE_ACCOUNT}" \
--rehome-audience https://relay.onorca.dev/v1/admin/host-drain
--rehome-audience https://relay.onorca.dev/v1/admin/host-drain \
--regional-rehome-protocol "${DESIRED_REHOME_PROTOCOL}"
terraform -chdir=infra/terraform apply -auto-approve \
"${RUNNER_TEMP}/relay-same-cap.tfplan"
gcloud compute instance-groups managed wait-until "${MIG_NAME}" --stable \
@@ -532,6 +549,20 @@ jobs:
env:
ORCA_RELAY_ADMIN_ID_TOKEN: ${{ steps.post-auth.outputs.id_token }}
run: |
# A single transient 5xx (LB warm-up behind a fresh instance) must not
# fail a canary; 4xx (auth, generation mismatch) still fails fast.
admin_post() {
local out="${RUNNER_TEMP}/$1.json"
if ! curl --fail-with-body --max-time 30 \
--retry 3 --retry-delay 2 --retry-connrefused --output "${out}" \
--request POST "$2" \
--header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \
--header 'Content-Type: application/json' --data "$3"; then
cat "${out}" >&2
return 1
fi
cat "${out}"
}
node dev/scripts/verify-relay-capacity-transition.mjs \
--director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \
--cell-id "${TARGET_CELL_ID}" --hard-cap "${EXPECTED_HARD_CAP}" \
@@ -539,19 +570,15 @@ jobs:
--heartbeat fresh --admission migration-only --draining forbidden \
--activity allowed --expected-image-digests "${DESIRED_IMAGE_DIGEST}" \
--regional-rehome-protocol "${DESIRED_REHOME_PROTOCOL}" --timeout-ms 900000
TARGET_RUNTIME="$(curl --fail-with-body --max-time 30 \
--request POST "${CELL_ORIGIN}/v1/admin/runtime-status" \
--header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \
--header 'Content-Type: application/json' --data '{"v":1}')"
TARGET_RUNTIME="$(admin_post target-runtime \
"${CELL_ORIGIN}/v1/admin/runtime-status" '{"v":1}')"
jq -e --arg digest "${DESIRED_IMAGE_DIGEST}" \
--argjson protocol "${DESIRED_REHOME_PROTOCOL}" \
'.imageDigest == $digest and (.regionalRehomeProtocol // 0) == $protocol' \
<<< "${TARGET_RUNTIME}" >/dev/null
TARGET_DIRECTOR_STATUS="$(curl --fail-with-body --max-time 30 \
--request POST "${DIRECTOR_ORIGIN}/v1/admin/cell-status" \
--header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \
--header 'Content-Type: application/json' \
--data "$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")"
TARGET_DIRECTOR_STATUS="$(admin_post target-cell-status \
"${DIRECTOR_ORIGIN}/v1/admin/cell-status" \
"$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")"
TARGET_INCARNATION="$(jq -er '.status.runtime.cellIncarnation' \
<<< "${TARGET_DIRECTOR_STATUS}")"
if test "${ROLLBACK_RESUME}" = true; then
@@ -588,7 +615,7 @@ jobs:
echo "MUTATION_STARTED=true" >> "${GITHUB_ENV}"
ACTIVATE_RESULT="$(node dev/scripts/prepare-relay-production-capacity-canary.mjs \
--director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \
--cell-id "${TARGET_CELL_ID}" --mode activate)"
--cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode activate)"
echo "${ACTIVATE_RESULT}"
SELECTOR_GENERATION_AFTER_ACTIVATE="$(jq -er '.generation' \
<<< "${ACTIVATE_RESULT}")"
@@ -627,7 +654,7 @@ jobs:
test "${MUTATION_STARTED:-false}" = true || exit 0
ISOLATE_RESULT="$(node dev/scripts/prepare-relay-production-capacity-canary.mjs \
--director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \
--cell-id "${TARGET_CELL_ID}" --mode isolate)"
--cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode isolate)"
echo "${ISOLATE_RESULT}"
# The isolate result carries the authoritative post-isolate generation;
# fixed offsets are wrong whenever an earlier isolate was a no-op.
@@ -87,13 +87,18 @@ jobs:
gate:
if: ${{ vars.ORCA_CLOUD_OPERATIONS_ENABLED == 'true' && (github.ref == 'refs/heads/main') }}
runs-on: blacksmith-2vcpu-ubuntu-2204
timeout-minutes: 10
# Headroom for the full-history checkout the canary provenance check needs.
timeout-minutes: 15
environment: production
outputs:
cells: ${{ steps.wave.outputs.cells }}
job-mode: ${{ steps.wave.outputs.job-mode }}
steps:
# Full history: the canary authority a batch verifies is sealed at an ancestor commit, and
# the provenance check fails closed on a commit a shallow clone left out.
- uses: actions/checkout@v4
with:
fetch-depth: 0
- uses: actions/setup-node@v4
with: { node-version: 24 }
@@ -95,7 +95,11 @@ jobs:
;;
esac
# Full history: the monitor evidence this job verifies is sealed at an ancestor commit,
# and the provenance check fails closed on a commit a shallow clone left out.
- uses: actions/checkout@v4
with:
fetch-depth: 0
- uses: actions/setup-node@v4
with:
+5 -4
View File
@@ -25,9 +25,10 @@ defaults:
working-directory: cloud
jobs:
# Public-repository hosted runners preserve Blacksmith allowance for macOS.
security:
name: Secret scan
runs-on: blacksmith-2vcpu-ubuntu-2204
runs-on: ubuntu-22.04
steps:
- uses: actions/checkout@v4
with:
@@ -53,7 +54,7 @@ jobs:
# Compiles the workspace. No Postgres service: nothing here reaches a
# database, and the service container costs ~13s of startup.
build:
runs-on: blacksmith-4vcpu-ubuntu-2204
runs-on: ubuntu-22.04
steps:
- uses: actions/checkout@v4
@@ -73,7 +74,7 @@ jobs:
# package it needs through the relay pretest hook, so it does not depend on
# `pnpm build` having run.
test:
runs-on: blacksmith-4vcpu-ubuntu-2204
runs-on: ubuntu-22.04
services:
postgres:
image: postgres:16-alpine
@@ -107,7 +108,7 @@ jobs:
# Fork pull requests reach this job, so it never configures a backend, never plans, and never
# holds a credential. Only the relay root ships here; foundation and apps stay private.
terraform:
runs-on: blacksmith-2vcpu-ubuntu-2204
runs-on: ubuntu-22.04
steps:
- uses: actions/checkout@v4
+5 -2
View File
@@ -149,9 +149,12 @@ jobs:
fi
# Reachability is the trust test: GitHub serves PR-only commits by SHA,
# so resolving the object is not proof a branch or tag of this repo
# reaches it. Bare + tree:0 keeps this to the commit graph.
# reaches it. Bare + tree:0 keeps this to the commit graph; reftable
# because branches that differ only in casing cannot both be stored by
# the files backend on a case-insensitive runner disk, which fails the
# entire fetch rather than the one ref.
scratch="$RUNNER_TEMP/vet-requested-ref"
git init -q --bare "$scratch"
git init -q --bare --ref-format=reftable "$scratch"
git -C "$scratch" fetch -q --filter=tree:0 "$REPO_URL" '+refs/heads/*:refs/heads/*' '+refs/tags/*:refs/tags/*'
if ! git -C "$scratch" rev-parse --verify --quiet "$REQUESTED_SHA^{commit}" >/dev/null; then
echo "::error::Commit $REQUESTED_SHA is not in stablyai/orca."
+12 -8
View File
@@ -27,6 +27,10 @@ on:
description: Ref to check out (defaults to the workflow ref)
required: false
type: string
test_files:
description: JSON array of specs to run; empty runs the full suite
required: false
type: string
schedule:
# Why: GitHub cron uses UTC; these slots map to 10am and 3pm
# America/Phoenix for the default-branch E2E run.
@@ -146,7 +150,7 @@ jobs:
# Native cache misses need the compiler, Electron needs Xvfb, and paired
# Quick Open needs ripgrep. Install them in one apt transaction per shard.
- name: Install native build and headless UI tools
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk python3 ripgrep xvfb zsh
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk python3 ripgrep xvfb zsh openbox x11-utils
- uses: ./.github/actions/install-node-dependencies
with:
@@ -167,7 +171,7 @@ jobs:
# ORCA_E2E_FORWARD_APP_LOGS keeps startup failures visible when Electron
# launches but never creates a BrowserWindow.
- name: Run E2E tests (${{ matrix.shard_name }})
run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 ORCA_E2E_WEB_CLIENT=1 ORCA_RELAY_PATH="$GITHUB_WORKSPACE/out/relay" pnpm run test:e2e --shard=${{ matrix.shard }}
run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 ORCA_E2E_WEB_CLIENT=1 ORCA_RELAY_PATH="$GITHUB_WORKSPACE/out/relay" pnpm run test:e2e --shard=${{ matrix.shard }}
# Why: Playwright retains traces/screenshots only on failure. Uploading
# them as an artifact makes post-mortem debugging on CI possible without
@@ -201,7 +205,7 @@ jobs:
# unbounded inventory fallback; the paired fixture exercises that real boundary.
# Why openssh-client: the Docker-SSH fixture shells out to ssh/ssh-keygen, and this
# lane now receives those specs from pr.yml's SSH source mapping.
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 ripgrep xvfb zsh
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 ripgrep xvfb zsh openbox x11-utils
- uses: ./.github/actions/install-node-dependencies
with:
@@ -241,7 +245,7 @@ jobs:
if grep -l '@headful' "${TEST_FILES[@]}" >/dev/null; then
E2E_PROJECT_ARGS+=(--project=electron-headful)
fi
xvfb-run --auto-servernum env "${E2E_ENV[@]}" \
xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env "${E2E_ENV[@]}" \
pnpm run test:e2e "${TEST_FILES[@]}" --workers=1 "${E2E_PROJECT_ARGS[@]}"
- name: Upload Playwright traces
@@ -278,7 +282,7 @@ jobs:
ref: ${{ inputs.ref || github.ref }}
- name: Install native build and headless UI tools
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 xvfb zsh
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 ripgrep xvfb zsh openbox x11-utils
- uses: ./.github/actions/install-node-dependencies
with:
@@ -293,7 +297,7 @@ jobs:
# Why: this is the release-path proof that the deployed Linux relay keeps
# its PTY and explorer live across a real watcher SIGSEGV.
- name: Run Docker SSH watcher isolation E2E
run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-watcher-isolation
run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-watcher-isolation
# Why: Playwright empties test-results/ when it starts, so each step here used to
# destroy the previous step's traces. Only the last lane's failure was ever
@@ -310,7 +314,7 @@ jobs:
# readiness across live SSH, headed paired, and headless serve topologies.
- name: Run Docker SSH terminal parking + startup readiness E2E
if: always()
run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-terminal-parking
run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-terminal-parking
- name: Keep terminal-parking traces
if: always()
@@ -326,7 +330,7 @@ jobs:
# legible as an SSH-named failure.
- name: Run remaining Docker SSH E2E
if: always()
run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker
run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker
- name: Keep remaining-ssh-docker traces
if: always()
@@ -98,12 +98,17 @@ jobs:
$env:SKIP_BUILD = '1'
$env:ORCA_E2E_FORWARD_APP_LOGS = '1'
pnpm run --if-present test:e2e:workspace-session-golden
if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }
pnpm run --if-present test:e2e:windows-fresh-startup-golden
if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }
pnpm run --if-present test:e2e:tab-bar-agent-launch-golden
if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }
if (Test-Path tests/e2e/golden-fresh-profile-terminal.spec.ts) {
pnpm run test:e2e -- tests/e2e/golden-fresh-profile-terminal.spec.ts tests/e2e/golden-shell-command.spec.ts
if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }
}
pnpm run --if-present test:e2e:source-control-golden
if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }
- name: Upload Playwright traces
if: failure()
+49 -41
View File
@@ -26,7 +26,7 @@ name: Hourly macOS Dev Build
# HOURLY_RELEASE_APP_ID the App's numeric id
# HOURLY_RELEASE_APP_PRIVATE_KEY the App's .pem private key
#
# Installation tokens live one hour, which is why this mints twice. Install and
# Installation tokens live one hour, so the build job mints twice. Install and
# build need no token at all, and notarization can hold the publish step for tens
# of minutes; minting again once the build is done starts the clock at the first
# call that actually uses it rather than burning a third of it on `pnpm install`.
@@ -60,33 +60,15 @@ env:
HOURLY_RETAIN_COUNT: 72
jobs:
build-hourly-mac:
# Avoid occupying the limited Mac pool when main has not moved.
preflight:
if: github.repository == 'stablyai/orca'
runs-on: ubuntu-latest
timeout-minutes: 5
outputs:
tag: ${{ steps.release.outputs.tag }}
version: ${{ steps.hourly.outputs.version }}
should_build: ${{ steps.freshness.outputs.should_build }}
head_sha: ${{ steps.freshness.outputs.head_sha }}
published: ${{ steps.publish_live.outcome == 'success' && 'true' || 'false' }}
runs-on: blacksmith-6vcpu-macos-15
# Why 150: it must exceed the worst case the retry budgets below can produce
# (install 3x10 + publish 2x45 = 120, plus ~25 for checkout/build/verify), or
# the job is killed mid-retry and no cleanup step runs at all. A typical run
# is far shorter — this is the notary queue's tail, not its median.
timeout-minutes: 150
env:
NODE_OPTIONS: --max-old-space-size=4096
steps:
- name: Checkout
uses: actions/checkout@v6
with:
ref: main
fetch-depth: 0
# Why: this job only reads stablyai/orca and never pushes; every write
# goes to the hourly repo through a minted App token passed by env.
# Not persisting the checkout credential shrinks the blast radius if a
# build step is compromised (zizmor: artipacked).
persist-credentials: false
- name: Mint hourly repo token
id: app_token
uses: actions/create-github-app-token@v2
@@ -95,18 +77,19 @@ jobs:
private-key: ${{ secrets.HOURLY_RELEASE_APP_PRIVATE_KEY }}
owner: stablyai
repositories: orca-hourly
permission-contents: read
# Why: main is often idle overnight. Rebuilding an unchanged commit burns a
# runner hour and adds a redundant tag to the retention window.
- name: Check whether main moved since the last hourly
id: freshness
shell: bash
env:
GH_TOKEN: ${{ steps.app_token.outputs.token }}
MAIN_REPO_TOKEN: ${{ github.token }}
FORCED: ${{ github.event_name == 'workflow_dispatch' && inputs.force }}
run: |
set -euo pipefail
head_sha="$(git rev-parse HEAD)"
head_sha="$(GH_TOKEN="$MAIN_REPO_TOKEN" gh api "repos/$GITHUB_REPOSITORY/commits/main" --jq .sha)"
[[ "$head_sha" =~ ^[0-9a-f]{40}$ ]] || { echo "::error::Could not resolve main"; exit 1; }
echo "head_sha=$head_sha" >>"$GITHUB_OUTPUT"
if [[ "$FORCED" == "true" ]]; then
echo "should_build=true" >>"$GITHUB_OUTPUT"
@@ -133,21 +116,55 @@ jobs:
echo "main moved to $head_sha (last hourly built $last_sha); building."
fi
build-hourly-mac:
needs: preflight
if: needs.preflight.outputs.should_build == 'true'
outputs:
tag: ${{ steps.release.outputs.tag }}
version: ${{ steps.hourly.outputs.version }}
head_sha: ${{ needs.preflight.outputs.head_sha }}
published: ${{ steps.publish_live.outcome == 'success' && 'true' || 'false' }}
runs-on: blacksmith-6vcpu-macos-15
# Why 150: it must exceed the worst case the retry budgets below can produce
# (install 3x10 + publish 2x45 = 120, plus ~25 for checkout/build/verify), or
# the job is killed mid-retry and no cleanup step runs at all. A typical run
# is far shorter — this is the notary queue's tail, not its median.
timeout-minutes: 150
env:
NODE_OPTIONS: --max-old-space-size=4096
steps:
- name: Checkout
uses: actions/checkout@v6
with:
ref: ${{ needs.preflight.outputs.head_sha }}
fetch-depth: 0
# Why: this job only reads stablyai/orca and never pushes; every write
# goes to the hourly repo through a minted App token passed by env.
# Not persisting the checkout credential shrinks the blast radius if a
# build step is compromised (zizmor: artipacked).
persist-credentials: false
- name: Mint hourly repo token
id: app_token
uses: actions/create-github-app-token@v2
with:
app-id: ${{ secrets.HOURLY_RELEASE_APP_ID }}
private-key: ${{ secrets.HOURLY_RELEASE_APP_PRIVATE_KEY }}
owner: stablyai
repositories: orca-hourly
- name: Setup pnpm
if: steps.freshness.outputs.should_build == 'true'
uses: pnpm/setup@v2
with:
install: false
- name: Setup Node.js
if: steps.freshness.outputs.should_build == 'true'
uses: actions/setup-node@v6
with:
node-version-file: package.json
cache: pnpm
- name: Cache electron-builder downloads
if: steps.freshness.outputs.should_build == 'true'
uses: actions/cache@v5
with:
path: |
@@ -158,7 +175,6 @@ jobs:
electron-builder-mac-
- name: Install dependencies
if: steps.freshness.outputs.should_build == 'true'
uses: nick-fields/retry@v4
with:
timeout_minutes: 10
@@ -169,7 +185,6 @@ jobs:
# Why: signing is what makes an hourly installable over an existing Orca, so
# a missing cert must fail here rather than after a 20-minute build.
- name: Verify macOS signing environment
if: steps.freshness.outputs.should_build == 'true'
run: node config/scripts/verify-macos-release-env.mjs
env:
CSC_LINK: ${{ secrets.MAC_CERTS }}
@@ -180,7 +195,6 @@ jobs:
- name: Compute hourly version
id: hourly
if: steps.freshness.outputs.should_build == 'true'
shell: bash
env:
GH_TOKEN: ${{ steps.app_token.outputs.token }}
@@ -211,7 +225,7 @@ jobs:
node config/scripts/hourly-build-version.mjs \
>"$RUNNER_TEMP/hourly-identity.txt"
grep -E '^(version|build_number)=' "$RUNNER_TEMP/hourly-identity.txt"
# Why check rather than trust: the checkout above pins `ref: main`, but a
# Why check rather than trust: the checkout above pins the resolved main commit, but a
# workflow_dispatch runs this file from whatever branch was dispatched. A
# branch that edits this step while main still has the old script yields
# an empty name and an untitled release — silent, and only visible once
@@ -223,7 +237,6 @@ jobs:
cat "$RUNNER_TEMP/hourly-identity.txt" >>"$GITHUB_OUTPUT"
- name: Build app
if: steps.freshness.outputs.should_build == 'true'
run: pnpm build:release
env:
NODE_OPTIONS: --max-old-space-size=4096
@@ -239,7 +252,6 @@ jobs:
# part the full budget.
- name: Re-mint hourly repo token for publish
id: app_token_publish
if: steps.freshness.outputs.should_build == 'true'
uses: actions/create-github-app-token@v2
with:
app-id: ${{ secrets.HOURLY_RELEASE_APP_ID }}
@@ -249,13 +261,12 @@ jobs:
- name: Create hourly release
id: release
if: steps.freshness.outputs.should_build == 'true'
shell: bash
env:
GH_TOKEN: ${{ steps.app_token_publish.outputs.token }}
TAG: v${{ steps.hourly.outputs.version }}
NAME: ${{ steps.hourly.outputs.name }}
SHA: ${{ steps.freshness.outputs.head_sha }}
SHA: ${{ needs.preflight.outputs.head_sha }}
run: |
set -euo pipefail
# Kept at 12 even though the title shows 7: the freshness check above
@@ -291,7 +302,6 @@ jobs:
echo "tag=$TAG" >>"$GITHUB_OUTPUT"
- name: Publish hourly macOS artifacts
if: steps.freshness.outputs.should_build == 'true'
uses: nick-fields/retry@v4
with:
# Why 45 like the release pipeline: an attempt is pack + notarize +
@@ -322,7 +332,6 @@ jobs:
# release missing that manifest is a tag the picker offers and the download
# 404s on, so fail loudly instead of leaving a broken entry.
- name: Verify update manifest published
if: steps.freshness.outputs.should_build == 'true'
shell: bash
env:
GH_TOKEN: ${{ steps.app_token_publish.outputs.token }}
@@ -352,7 +361,6 @@ jobs:
# means the picker can never offer a release whose assets are incomplete.
- name: Publish the verified release
id: publish_live
if: steps.freshness.outputs.should_build == 'true'
shell: bash
env:
GH_TOKEN: ${{ steps.app_token_publish.outputs.token }}
+6 -21
View File
@@ -15,8 +15,13 @@ on:
# Why: this job holds the only checks that load the Fastfile, so edits to
# it or to the release workflow it guards must re-run them.
- '.github/workflows/mobile.yml'
- '.github/actions/install-node-dependencies/**'
- '.github/workflows/mobile-ios-release.yml'
concurrency:
group: mobile-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
verify:
runs-on: ubuntu-latest
@@ -35,10 +40,7 @@ jobs:
- name: Checkout
uses: actions/checkout@v6
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
- uses: ./.github/actions/install-node-dependencies
# bundler-cache installs mobile/Gemfile.lock, so this job is also what
# proves the pinned fastlane the release workflow depends on still
@@ -50,23 +52,6 @@ jobs:
bundler-cache: true
working-directory: mobile
- name: Setup pnpm
uses: pnpm/setup@v2
with:
install: false
# Why: the mobile typecheck imports shared types from ../src/shared, and
# some of those files import runtime deps (tweetnacl, ws) resolved from
# the repo-root node_modules. Without a root install, tsc fails with
# "Cannot find module 'tweetnacl'/'ws'". Mobile is a separate pnpm project
# (not in the root workspace), so this is a distinct install.
# --ignore-scripts skips the root postinstall (Electron native-module
# rebuild) which is irrelevant to a type-only check and would only add
# time and failure surface on this ubuntu mobile runner.
- name: Install root dependencies
working-directory: .
run: pnpm install --frozen-lockfile --ignore-scripts
- name: Install dependencies
run: pnpm install --frozen-lockfile
@@ -0,0 +1,63 @@
name: Performance contracts
on:
schedule:
- cron: '15 9 * * *'
workflow_dispatch:
pull_request:
paths:
- '.github/workflows/performance-contracts.yml'
- 'config/vitest.performance.config.ts'
- 'config/oxlint-performance-audit.json'
- 'config/oxlint-plugins/*performance.mjs'
- 'config/oxlint-plugins/quadratic-buffer-concat.mjs'
- 'config/scripts/*-plugin.test.mjs'
# Keep in sync with the contract list in config/vitest.performance.config.ts;
# without these a rename lands green and only breaks the next nightly.
- 'src/main/sqlite/sync-database.test.ts'
- 'src/main/runtime/orchestration/db/row-column-lists.test.ts'
- 'src/relay/fs-path-metadata-symlink-concurrency.test.ts'
- 'src/renderer/src/components/editor/rich-markdown-list-tokenizers.test.ts'
- 'src/renderer/src/components/editor/rich-markdown-lowlight-cache.test.ts'
- 'src/renderer/src/components/terminal-pane/agent-completion-coordinator-queued-inspection-disposal.test.ts'
- 'src/renderer/src/lib/pane-manager/pane-terminal-output-scheduler-queue-retention.test.ts'
permissions:
contents: read
concurrency:
group: performance-contracts-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
contracts:
strategy:
fail-fast: false
matrix:
os: [ubuntu-latest, macos-latest, windows-latest]
runs-on: ${{ matrix.os }}
timeout-minutes: 20
steps:
- uses: actions/checkout@v6
with:
persist-credentials: false
- uses: ./.github/actions/install-node-dependencies
- name: Run operation-count and retention contracts
run: pnpm test:perf:contracts --reporter=default --reporter=json --outputFile=performance-contracts.json
# Source-only scan: identical on every OS, so run it once.
- name: Audit production performance patterns
if: always() && matrix.os == 'ubuntu-latest'
shell: bash
run: pnpm --silent audit:perf > performance-audit.json
- uses: actions/upload-artifact@v7
if: always()
with:
name: performance-contracts-${{ matrix.os }}
path: performance-contracts.json
if-no-files-found: error
- uses: actions/upload-artifact@v7
if: always() && matrix.os == 'ubuntu-latest'
with:
name: performance-audit
path: performance-audit.json
if-no-files-found: error
+59 -62
View File
@@ -41,6 +41,10 @@ jobs:
managed_hook_node18: ${{ steps.filter.outputs.managed_hook_node18 }}
package: ${{ steps.filter.outputs.package }}
package_windows: ${{ steps.filter.outputs.package_windows }}
e2e_should_run: ${{ steps.e2e_filter.outputs.should_run }}
test_files: ${{ steps.e2e_filter.outputs.test_files }}
ssh_source_changed: ${{ steps.e2e_filter.outputs.ssh_source_changed }}
native_ime_source_changed: ${{ steps.e2e_filter.outputs.native_ime_source_changed }}
steps:
- name: Checkout
uses: actions/checkout@v6
@@ -66,6 +70,38 @@ jobs:
printf '%s\n' "$CHANGED"
printf '%s\n' "$CHANGED" | node config/scripts/pr-code-change-scope.mjs | tee -a "$GITHUB_OUTPUT"
# Reuse the path-detector checkout instead of queuing another runner.
- name: Filter changed E2E specs
id: e2e_filter
if: github.event.pull_request.draft != true && steps.filter.outputs.should_run == 'true'
run: |
set -euo pipefail
BASE="${{ github.event.pull_request.base.sha }}"
HEAD="${{ github.event.pull_request.head.sha }}"
CHANGED="$(git diff --name-only --diff-filter=AMCR --merge-base "$BASE" "$HEAD")"
# Source routes are executable contracts so a test can prove exact
# authorities, exclusions, and sentinels without evaluating workflow shell.
TEST_FILES_JSON="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs)"
echo "test_files=$TEST_FILES_JSON" >> "$GITHUB_OUTPUT"
# Why a separate signal: the Docker-SSH lane must trigger on SSH source, not on a
# spec name surviving in a route's list. Same routes, so the two cannot drift.
SSH_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --ssh-source)"
echo "ssh_source_changed=$SSH_SOURCE_CHANGED" >> "$GITHUB_OUTPUT"
echo "SSH source changed: $SSH_SOURCE_CHANGED"
# Why its own signal: the real-IME lane is a whole ibus session, not a spec, so it must
# trigger on IME source rather than on a spec name in some route's list.
NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)"
echo "native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED" >> "$GITHUB_OUTPUT"
echo "Native IME source changed: $NATIVE_IME_SOURCE_CHANGED"
SHOULD_RUN="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --reusable-workflow)"
if [ "$SHOULD_RUN" = true ]; then
echo "should_run=true" >> "$GITHUB_OUTPUT"
echo "Changed E2E specs: $TEST_FILES_JSON"
else
echo "should_run=false" >> "$GITHUB_OUTPUT"
echo "No specs requiring the reusable E2E workflow"
fi
static_analysis:
name: static analysis
needs: [code_paths]
@@ -712,7 +748,11 @@ jobs:
- name: Package unpacked app
env:
ORCA_REUSE_PREPARED_NATIVE_RUNTIME: '1'
run: pnpm exec electron-builder --config config/electron-builder.config.cjs --linux AppImage deb rpm --x64 --publish never
# PR artifacts are only inspected locally; gzip avoids release-size xz compression.
run: >-
pnpm exec electron-builder --config config/electron-builder.config.cjs
--linux AppImage deb rpm --x64 --publish never
--config.deb.compression=gz --config.rpm.compression=gzip
- name: Verify root-package marker payloads
run: |
@@ -792,10 +832,13 @@ jobs:
node_modules/.pnpm/@vscode+windows-process-tree@*/node_modules/@vscode/windows-process-tree/build
key: native-modules-${{ runner.os }}-${{ steps.deps.outputs.native-cache-scope }}-${{ runner.arch }}-node-node${{ steps.deps.outputs.node-version }}-${{ hashFiles('pnpm-lock.yaml', '.github/actions/install-node-dependencies/action.yml', 'config/scripts/ensure-native-runtime.mjs', 'config/scripts/rebuild-native-deps.mjs', 'config/patches/node-pty@1.1.0.patch', 'config/patches/@vscode__windows-process-tree@0.8.0.patch') }}
# vitest runs here directly rather than through `pnpm test`, so the addon
# assertions only hold once install-node-dependencies has rebuilt natives.
- name: Test Windows-specific boundaries
run: >-
pnpm exec vitest run --config config/vitest.config.ts
config/scripts/rebuild-native-deps.test.mjs
config/scripts/rebuild-native-deps-windows-process-tree.test.mjs
src/main/browser/browser-client-page-renderer-lifecycle.electron.test.ts
src/main/browser/browser-route-tcp-egress.electron.test.ts
src/main/browser/browser-route-webrtc-egress.electron.test.ts
@@ -804,9 +847,14 @@ jobs:
src/main/providers/windows-conpty-wide-char-duplication.node-pty.test.ts
src/main/providers/pty-repaint-wide-char-buffer.node-pty.test.ts
src/shared/child-process/windows-command-line.win32.test.ts
src/shared/child-process/windows-cmd-shim-resolution.test.ts
src/shared/child-process/windows-cmd-shim-resolution.win32.test.ts
src/main/agent-hooks/windows-hook-payload-delivery.test.ts
src/main/agent-hooks/windows-direct-cmd-hook-command.test.ts
src/main/windows/windows-pty-job.win32.test.ts
src/main/windows/windows-host-job.win32.test.ts
src/main/windows/windows-process-tree-command-line-patch.test.ts
src/main/windows-live-tree-kill.win32.test.ts
src/main/wsl/wsl-runner.test.ts
src/main/wsl/wsl-guest-environment.test.ts
src/main/wsl/wsl-invocation-boundary.test.ts
@@ -814,14 +862,18 @@ jobs:
src/main/wsl/wsl-w1-w3-contract.test.ts
src/shared/source-scan/source-tree-scan.test.ts
src/main/cli/wsl-cli-powershell-boundary.test.ts
src/main/computer/desktop-script-runtime-host.win32.test.ts
src/main/cursor/hook-service.test.ts
src/main/orca-profiles/profile-index-store.test.ts
src/main/startup/windows-install-dir-acl-repair.win32.test.ts
src/main/runtime/repo-worktree-admin-fingerprint.test.ts
src/main/runtime/worktree-scan-admin-fingerprint-gate.test.ts
src/shared/secure-file-fsync-flags.test.ts
src/shared/secure-path-windows-acl.win32.test.ts
src/main/runtime/unreadable-secret-store-preservation.win32.test.ts
src/main/ipc/pty-codex-account-attribution.test.ts
src/main/ipc/pty-spawn-env-codex-resume-provenance.test.ts
src/relay/windows-port-scan.win32.test.ts
# Why the :parallel variant: identical to build:release except the three
# electron-vite targets overlap instead of running back to back. The Linux package
@@ -860,65 +912,10 @@ jobs:
- name: Smoke packaged CLI
run: node config/scripts/smoke-packaged-cli.mjs --app-dir=dist/win-unpacked
# Why: PR E2E is advisory and only validates changed specs; scheduled and
# release runs retain full-suite coverage.
e2e-paths:
name: detect changed e2e specs
needs: [code_paths]
runs-on: ubuntu-latest
if: github.event.pull_request.draft != true && needs.code_paths.outputs.should_run == 'true'
# Why: detector only needs to read the checkout; do not inherit repo defaults.
permissions:
contents: read
outputs:
should_run: ${{ steps.filter.outputs.should_run }}
test_files: ${{ steps.filter.outputs.test_files }}
ssh_source_changed: ${{ steps.filter.outputs.ssh_source_changed }}
native_ime_source_changed: ${{ steps.filter.outputs.native_ime_source_changed }}
steps:
- name: Checkout
uses: actions/checkout@v6
with:
# Why blob:none: full history is needed for the merge-base diff, but historical
# file contents are not. Blobs are ~89% of this repo's pack, and Git fetches the
# few this job actually reads on demand.
fetch-depth: 0
filter: blob:none
persist-credentials: false
- name: Filter changed E2E specs
id: filter
run: |
set -euo pipefail
BASE="${{ github.event.pull_request.base.sha }}"
HEAD="${{ github.event.pull_request.head.sha }}"
CHANGED="$(git diff --name-only --diff-filter=AMCR --merge-base "$BASE" "$HEAD")"
# Source routes are executable contracts so a test can prove exact
# authorities, exclusions, and sentinels without evaluating workflow shell.
TEST_FILES_JSON="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs)"
echo "test_files=$TEST_FILES_JSON" >> "$GITHUB_OUTPUT"
# Why a separate signal: the Docker-SSH lane must trigger on SSH source, not on a
# spec name surviving in a route's list. Same routes, so the two cannot drift.
SSH_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --ssh-source)"
echo "ssh_source_changed=$SSH_SOURCE_CHANGED" >> "$GITHUB_OUTPUT"
echo "SSH source changed: $SSH_SOURCE_CHANGED"
# Why its own signal: the real-IME lane is a whole ibus session, not a spec, so it must
# trigger on IME source rather than on a spec name in some route's list.
NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)"
echo "native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED" >> "$GITHUB_OUTPUT"
echo "Native IME source changed: $NATIVE_IME_SOURCE_CHANGED"
if [ "$TEST_FILES_JSON" != '[]' ]; then
echo "should_run=true" >> "$GITHUB_OUTPUT"
echo "Changed E2E specs: $TEST_FILES_JSON"
else
echo "should_run=false" >> "$GITHUB_OUTPUT"
echo "No changed E2E specs"
fi
e2e:
name: e2e
needs: e2e-paths
if: needs.e2e-paths.outputs.should_run == 'true'
needs: code_paths
if: needs.code_paths.outputs.e2e_should_run == 'true'
# Why: reusable e2e.yml only checkouts, builds, and uploads artifacts.
permissions:
contents: read
@@ -927,8 +924,8 @@ jobs:
# The synthetic pull-request merge ref can disappear while this reusable
# workflow is queued. The head SHA is immutable and works for every PR.
ref: ${{ github.event.pull_request.head.sha }}
test_files: ${{ needs.e2e-paths.outputs.test_files }}
ssh_source_changed: ${{ needs.e2e-paths.outputs.ssh_source_changed }}
test_files: ${{ needs.code_paths.outputs.test_files }}
ssh_source_changed: ${{ needs.code_paths.outputs.ssh_source_changed }}
# Why this is not in verify's needs: it is the first PR-gate run of a harness whose reliability
# is only known from nightly main runs (20/20 green, 2026-08-09..2026-08-29, p50 3m25s). It
@@ -938,8 +935,8 @@ jobs:
# require `success || skipped` outside the strict loop — see the note on `e2e`.
terminal_ime_native:
name: real IME
needs: e2e-paths
if: needs.e2e-paths.outputs.native_ime_source_changed == 'true'
needs: code_paths
if: needs.code_paths.outputs.native_ime_source_changed == 'true'
# Why: the reusable workflow only checks out, builds, and uploads artifacts.
permissions:
contents: read
+146 -30
View File
@@ -809,13 +809,7 @@ jobs:
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
TAG: ${{ needs.cut.outputs.tag }}
run: |
if gh release view "$TAG" --repo "$GITHUB_REPOSITORY" >/dev/null 2>&1; then
echo "Release $TAG already exists."
exit 0
fi
node config/scripts/create-draft-release.mjs "$TAG"
run: node config/scripts/create-draft-release.mjs "$TAG"
terminal-rendering-golden:
needs: cut
@@ -858,16 +852,17 @@ jobs:
if: runner.os == 'Linux'
run: sudo apt-get update && sudo apt-get install -y build-essential python3 xvfb
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
- name: Setup pnpm
uses: pnpm/setup@v2
with:
install: false
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
cache: pnpm
# Why: Linux terminal golden E2E uses the same native install path as
# release CI, which needs pnpm to bypass its non-executable gyp_main.py.
- name: Use external node-gyp to avoid pnpm's bundled copy (Linux only)
@@ -1074,16 +1069,17 @@ jobs:
if: runner.os == 'Linux'
run: sudo apt-get update && sudo apt-get install -y build-essential python3 xvfb
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
- name: Setup pnpm
uses: pnpm/setup@v2
with:
install: false
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
cache: pnpm
# Why: keep the non-blocking evidence lane on the same Linux native
# install path as the blocking golden and release build jobs.
- name: Use external node-gyp to avoid pnpm's bundled copy (Linux only)
@@ -1425,6 +1421,17 @@ jobs:
command: ${{ matrix.release_command }}
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
# Why: the NSIS uninstaller only exists inside electron-builder's
# uninstaller pass, which deletes it right after embedding it. The sign
# hook in config/scripts/windows-uninstaller-signing.cjs copies it out
# here so it can ride the inner-binaries SignPath request below.
# Why runner.temp and never the workspace: `files` in
# config/electron-builder.config.cjs is all-negation, so app-builder
# prepends `**/*` and packs whatever is left in the checkout root. This
# step retries up to 3 times; attempt 1 writes the file after packing,
# but attempts 2 and 3 would then pack the unsigned uninstaller into
# app.asar - the exact defect this chain exists to remove.
ORCA_WIN_UNINSTALLER_EXPORT_PATH: ${{ runner.temp }}\uninstaller-signing\unsigned\orca-uninstaller.exe
- name: Verify Windows node-pty ConPTY runtime
if: matrix.platform == 'win' && github.run_attempt == 1
@@ -1451,7 +1458,10 @@ jobs:
# Why: SignPath cannot deep-sign inside NSIS installers, so inner PE
# files (Orca.exe, node-pty *.node, DLLs) are signed via a separate zip
# request, then the installer is rebuilt from the signed tree before the
# existing installer signing request below. Every step in this chain is
# existing installer signing request below. The NSIS uninstaller rides
# this same request (it is the MDE update cluster: old-uninstaller.exe /
# Uninstall Orca.exe), captured through electron-builder's sign hook and
# swapped back in during the rebuild — no third approval wait. Every step is
# fail-open (continue-on-error + outcome gating): any failure ships the
# original installer with unsigned inner binaries, exactly like releases
# did before this chain existed. Rehearsed end to end in run 28988432001
@@ -1498,6 +1508,36 @@ jobs:
Write-Host "Skipped $($skipped.Count) already-signed files:"
$skipped | ForEach-Object { Write-Host " $_" }
# Why the uninstaller rides this request: it is the file MDE flagged in
# the whole update cluster (old-uninstaller.exe / Uninstall Orca.exe),
# and folding it in here costs no extra approval wait. Why it is kept
# out of inner-signing-list.txt: that list drives the copy-back into
# dist/win-unpacked, and the uninstaller does not live there — it is
# re-injected through the sign hook during the rebuild instead.
# Why this name and not "Uninstall Orca.exe": the restore loop below
# matches staged files by suffix (`-like "*$relative"`) and takes the
# first hit, so any staged path ending in "Orca.exe" is separated from
# the real Orca.exe only by Get-ChildItem's enumeration order. That
# order happens to favour the root file today, but it is not a
# documented guarantee; a name that cannot suffix-match is.
# Why the whole block is caught rather than just Test-Path'd: this
# step's outcome gates the upload of every inner binary, so a locked
# file or a full disk here would cost all of them their signatures -
# worse than shipping no uninstaller signature at all.
try {
$exportedUninstaller = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\unsigned\orca-uninstaller.exe'
if (Test-Path -LiteralPath $exportedUninstaller) {
$uninstallerStagePath = Join-Path $stage.FullName 'uninstaller\orca-uninstaller.exe'
New-Item -ItemType Directory -Force -Path (Split-Path $uninstallerStagePath) -ErrorAction Stop | Out-Null
Copy-Item -LiteralPath $exportedUninstaller -Destination $uninstallerStagePath -Force -ErrorAction Stop
Write-Host 'Staged the NSIS uninstaller for signing: uninstaller\orca-uninstaller.exe'
} else {
Write-Host "::warning::No exported NSIS uninstaller at $exportedUninstaller; this release ships an unsigned uninstaller (fail-open)."
}
} catch {
Write-Host "::warning::Could not stage the NSIS uninstaller ($_); this release ships an unsigned uninstaller (fail-open)."
}
- name: Upload unsigned inner binaries for SignPath
id: upload-unsigned-inner
if: matrix.platform == 'win' && github.run_attempt == 1 && steps.stage-inner.outcome == 'success'
@@ -1642,6 +1682,31 @@ jobs:
throw "Signed inner artifact did not round-trip cleanly ($($failures.Count) failures)."
}
# Why gated separately from the inner restore above: if SignPath's
# windows-inner-binaries-zip artifact configuration does not (yet) cover the
# uninstaller/ directory, the uninstaller comes back missing. That must cost
# only the uninstaller signature — the rebuild below still runs and still
# ships the signed inner binaries, exactly as it does today.
- name: Restore signed uninstaller for the installer rebuild
id: restore-signed-uninstaller
if: matrix.platform == 'win' && github.run_attempt == 1 && steps.restore-signed-inner.outcome == 'success'
continue-on-error: true
shell: pwsh
run: |
$signed = Get-ChildItem -Path signed-inner -Recurse -File -Filter 'orca-uninstaller.exe' |
Select-Object -First 1
if ($null -eq $signed) {
throw 'SignPath did not return uninstaller/orca-uninstaller.exe; check the windows-inner-binaries-zip artifact configuration covers it.'
}
$signature = Get-AuthenticodeSignature -FilePath $signed.FullName
if ($null -eq $signature.SignerCertificate) {
throw 'The returned NSIS uninstaller carries no signature.'
}
$signedDir = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\signed'
New-Item -ItemType Directory -Force -Path $signedDir | Out-Null
Copy-Item -LiteralPath $signed.FullName -Destination (Join-Path $signedDir 'orca-uninstaller.exe') -Force
Write-Host ("{0,-14} uninstaller <{1}>" -f $signature.Status, $signature.SignerCertificate.Subject)
# Why this step exists: electron-builder's CopyElevateHelper re-copies a
# pristine elevate.exe from its download cache over resources\elevate.exe
# on EVERY nsis pack — including the --prepackaged rebuild below — which
@@ -1651,9 +1716,12 @@ jobs:
# no-op. Known quirk: the cache persists across releases via actions/cache,
# so later runs may see elevate.exe as already signed and skip staging it —
# that is fine (the signature is timestamped) and the evidence gate checks
# elevate.exe in the shipped installer unconditionally. If this ever causes
# trouble, delete this step; the only effect is elevate.exe shipping
# unsigned again, which the evidence gate will flag.
# elevate.exe in the shipped installer unconditionally.
#
# The cache lookup lives in a script because the inline path this step used
# (`<cache>\nsis`) matches no app-builder-lib layout, and `SilentlyContinue`
# plus `exit 0` turned that miss into a green step — v1.4.193 and v1.4.194
# shipped an unsigned elevate.exe that way. A miss now fails the step.
- name: Replace cached elevate.exe with the signed copy
id: sign-elevate-cache
if: matrix.platform == 'win' && github.run_attempt == 1 && steps.restore-signed-inner.outcome == 'success'
@@ -1665,20 +1733,26 @@ jobs:
Write-Host '::warning::No elevate.exe in win-unpacked resources; nothing to protect from the rebuild clobber.'
exit 0
}
# Why this guard stays: windows-signing-rehearsal.yml shares the
# electron-builder-win-<lockfile hash> cache key with this workflow, so a
# test-certificate elevate.exe must never be staged into a release cache.
$signature = Get-AuthenticodeSignature -FilePath $signed
$subject = if ($null -eq $signature.SignerCertificate) { '<none>' } else { $signature.SignerCertificate.Subject }
if ($signature.Status -ne 'Valid' -or $subject -notlike '*CN=SignPath Foundation*') {
Write-Host "::warning::win-unpacked elevate.exe is not SignPath-signed ($($signature.Status), $subject); skipping cache swap."
exit 0
}
$cached = @(Get-ChildItem "$env:LOCALAPPDATA\electron-builder\Cache\nsis" -Recurse -Filter elevate.exe -ErrorAction SilentlyContinue)
if ($cached.Count -eq 0) {
Write-Host '::warning::No cached elevate.exe found (electron-builder cache layout changed?); the rebuild will pack the unsigned copy and the evidence gate will flag it.'
exit 0
}
foreach ($file in $cached) {
Copy-Item -Path $signed -Destination $file.FullName -Force
Write-Host "Replaced $($file.FullName) with the SignPath-signed copy."
node config/scripts/replace-cached-nsis-elevate.mjs $signed
if ($LASTEXITCODE -ne 0) {
$message = 'Cached elevate.exe swap found nothing to replace; the rebuilt installer ships an unsigned UAC elevation helper (issue #7785).'
if ($env:GITHUB_STEP_SUMMARY) {
try {
Add-Content -Path $env:GITHUB_STEP_SUMMARY -Value "**Windows elevate.exe cache swap:** FAILED — $message" -ErrorAction Stop
} catch {
Write-Host "::warning::Could not write the elevate.exe swap verdict to the job summary: $_"
}
}
throw $message
}
- name: Rebuild NSIS installer from signed unpacked app
@@ -1686,6 +1760,11 @@ jobs:
if: matrix.platform == 'win' && github.run_attempt == 1 && steps.restore-signed-inner.outcome == 'success'
continue-on-error: true
shell: pwsh
env:
# Why unconditional: the sign hook keys off the file existing, which it
# only does when the restore step above succeeded. A missing file logs a
# warning and embeds the freshly built unsigned uninstaller instead.
ORCA_WIN_UNINSTALLER_SIGNED_PATH: ${{ runner.temp }}\uninstaller-signing\signed\orca-uninstaller.exe
run: |
# Why: keep the pre-rebuild artifacts so a failed rebuild can fall
# back to shipping them unchanged (fail-open).
@@ -1716,6 +1795,7 @@ jobs:
with:
name: orca-windows-unsigned-${{ needs.cut.outputs.tag }}
path: dist/orca-windows-setup.exe
compression-level: 0
if-no-files-found: error
# Why: SignPath Foundation production certificates require manual review,
@@ -1876,6 +1956,7 @@ jobs:
env:
ORCA_WINDOWS_INNER_SIGNATURE_REQUIRED: 'false'
INNER_SIGNING_COMPLETED: ${{ steps.rebuild-nsis-signed.outcome == 'success' }}
UNINSTALLER_SIGNING_COMPLETED: ${{ steps.restore-signed-uninstaller.outcome == 'success' }}
run: |
$required = $env:ORCA_WINDOWS_INNER_SIGNATURE_REQUIRED -eq 'true'
@@ -1956,6 +2037,39 @@ jobs:
if ($targets -notcontains 'resources\elevate.exe') {
$targets += 'resources\elevate.exe'
}
# Why the uninstaller is not in $targets: NSIS embeds it in its own
# compressed data section (`File /oname=${UNINSTALL_FILENAME}` in
# app-builder-lib templates/nsis/include/installer.nsh), not in the
# app 7z payload extracted above - the bundled 7za cannot see it.
# What the receipt proves and does not: the digest comparison is
# equal by construction (the hook digests the bytes it copied from
# this same file), so the real signal is that the receipt exists at
# all - the import leg ran, and these are the bytes it embedded. The
# signature check below is the part with teeth. The shipped-artifact
# check lives in windows-signing-rehearsal.yml, which installs the
# installer and inspects the uninstaller it drops on disk.
if ($env:UNINSTALLER_SIGNING_COMPLETED -eq 'true') {
$signedUninstaller = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\signed\orca-uninstaller.exe'
$receipt = "$signedUninstaller.embedded-sha256"
if (-not (Test-Path -LiteralPath $receipt)) {
$failures.Add('the sign hook did not embed the signed uninstaller into the rebuilt installer')
} else {
$embedded = (Get-Content -LiteralPath $receipt -Raw).Trim()
$actual = (Get-FileHash -LiteralPath $signedUninstaller -Algorithm SHA256).Hash.ToLowerInvariant()
$signature = Get-AuthenticodeSignature -FilePath $signedUninstaller
$subject = if ($null -eq $signature.SignerCertificate) { '<none>' } else { $signature.SignerCertificate.Subject }
$line = "{0,-14} {1} <{2}>" -f $signature.Status, 'Uninstall Orca.exe (embedded)', $subject
$report.Add($line)
Write-Host $line
if ($embedded -ne $actual) {
$failures.Add("the rebuilt installer embedded different uninstaller bytes than the signed one ($embedded vs $actual)")
} elseif ($signature.Status -ne 'Valid' -or $subject -notlike '*CN=SignPath Foundation*') {
$failures.Add("not signed by SignPath Foundation: Uninstall Orca.exe ($($signature.Status), $subject)")
}
}
} else {
Write-Host '::warning::The NSIS uninstaller was not signed on this run; it is excluded from the evidence gate (fail-open).'
}
foreach ($relative in $targets) {
$path = Join-Path $root $relative
if (-not (Test-Path $path)) {
@@ -1988,7 +2102,9 @@ jobs:
Add-GateEvidence "VERDICT: FAILED — $message"
Add-GateSummary "FAILED — $message"
} else {
$ok = "All $($targets.Count) inner binaries in the shipped installer are signed by SignPath Foundation."
# $report, not $targets: the embedded uninstaller is reported but
# is not one of the extracted payload targets.
$ok = "All $($report.Count) checked binaries are signed by SignPath Foundation."
Add-GateEvidence "VERDICT: PASSED — $ok"
Add-GateSummary "PASSED — $ok"
Write-Host $ok
@@ -0,0 +1,38 @@
name: Release ref validation
on:
pull_request:
paths:
- '.github/workflows/adhoc-mac-build.yml'
- '.github/workflows/dev-channel-win-build.yml'
- '.github/workflows/release-ref-validation.yml'
- 'config/scripts/workflow-ref-reachability.test.mjs'
- 'config/scripts/workflow-ref-mirror-case-safety.test.mjs'
workflow_dispatch:
permissions:
contents: read
concurrency:
group: release-ref-validation-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
validate:
strategy:
fail-fast: false
matrix:
os: [macos-15, windows-2022]
runs-on: ${{ matrix.os }}
timeout-minutes: 10
steps:
- uses: actions/checkout@v6
with:
persist-credentials: false
- uses: ./.github/actions/install-node-dependencies
- name: Verify case-twin refs and release trust boundary
run: >-
pnpm exec vitest run --config config/vitest.config.ts
config/scripts/workflow-ref-reachability.test.mjs
config/scripts/workflow-ref-mirror-case-safety.test.mjs
config/scripts/dev-channel-windows-workflow-contract.test.mjs
@@ -22,6 +22,10 @@ on:
- main
paths: *skill-roundtrip-paths
concurrency:
group: skill-roundtrip-${{ github.event_name }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
jobs:
roundtrip:
strategy:
@@ -41,7 +45,9 @@ jobs:
steps:
- uses: actions/checkout@v6
with:
# Historical skill snapshots need tags, but only their blobs are read.
fetch-depth: 0
filter: blob:none
persist-credentials: false
- uses: actions/setup-node@v6
with:
+2 -16
View File
@@ -38,23 +38,9 @@ jobs:
xfwm4
xvfb
- name: Setup Node.js
uses: actions/setup-node@v6
- uses: ./.github/actions/install-node-dependencies
with:
node-version-file: package.json
- name: Setup pnpm
uses: pnpm/setup@v2
with:
install: false
- name: Use external node-gyp to avoid pnpm bundled copy
run: |
npm install -g node-gyp@11.5.0
echo "npm_config_node_gyp=$(npm root -g)/node-gyp/bin/node-gyp.js" >> "$GITHUB_ENV"
- name: Install dependencies
run: pnpm install --frozen-lockfile
native-runtime: electron
- name: Build Electron app for E2E
run: pnpm exec electron-vite build --mode e2e
+6 -5
View File
@@ -67,16 +67,17 @@ jobs:
- name: Install native build tools and xvfb
run: sudo apt-get update && sudo apt-get install -y build-essential python3 xvfb zsh
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
- name: Setup pnpm
uses: pnpm/setup@v2
with:
install: false
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
cache: pnpm
# Why: this scheduled/manual workflow uses the same native install path as
# PR and E2E CI, which needs pnpm to bypass its bundled gyp_main.py.
- name: Use external node-gyp to avoid pnpm's bundled copy
+204 -12
View File
@@ -3,9 +3,11 @@
# Why: SignPath cannot deep-sign inside NSIS installers, so shipping signed
# inner binaries (Orca.exe, node-pty *.node, DLLs — see issue #7785) requires
# a two-request flow: sign the unpacked PE files first, then build the NSIS
# installer from the signed tree, then sign the installer. This workflow
# rehearses that entire flow from a branch, end to end, without publishing
# anything — so the release pipeline on main is never at risk while we verify.
# installer from the signed tree, then sign the installer. The NSIS uninstaller
# rides that same first request — it is captured through electron-builder's sign
# hook and swapped back in during the rebuild — so it adds no third approval.
# This workflow rehearses that entire flow from a branch, end to end, without
# publishing anything — so the release pipeline on main is never at risk.
#
# Runs only via manual dispatch. Use the test-signing policy for iteration
# (auto-approved test certificate) and release-signing to rehearse the
@@ -81,15 +83,27 @@ jobs:
env:
NODE_OPTIONS: --max-old-space-size=4096
- name: Package unpacked Windows app
# Why a full --win build and not --dir: the NSIS uninstaller only exists
# inside the installer build, and it is the file the MDE update cluster
# flags. --dir would never produce it, so the rehearsal would not rehearse
# the uninstaller leg at all. This mirrors release-cut's first Windows pass.
- name: Package Windows app and export the NSIS uninstaller
shell: pwsh
env:
# runner.temp, never the workspace: the all-negation `files` list in
# config/electron-builder.config.cjs packs whatever is left in the
# checkout root into app.asar.
ORCA_WIN_UNINSTALLER_EXPORT_PATH: ${{ runner.temp }}\uninstaller-signing\unsigned\orca-uninstaller.exe
run: |
node config/scripts/ensure-native-runtime.mjs --runtime=electron
if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }
pnpm exec electron-builder --config config/electron-builder.config.cjs --win --dir --publish never
pnpm exec electron-builder --config config/electron-builder.config.cjs --win --publish never
if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }
if (-not (Test-Path 'dist/win-unpacked/Orca.exe')) {
throw 'electron-builder --dir did not produce dist/win-unpacked/Orca.exe'
throw 'electron-builder --win did not produce dist/win-unpacked/Orca.exe'
}
if (-not (Test-Path -LiteralPath $env:ORCA_WIN_UNINSTALLER_EXPORT_PATH)) {
throw "The sign hook did not export the NSIS uninstaller to $env:ORCA_WIN_UNINSTALLER_EXPORT_PATH"
}
# Why: only unsigned PE files go to SignPath. Files that already carry a
@@ -132,6 +146,17 @@ jobs:
Write-Host "Skipped $($skipped.Count) already-signed files:"
$skipped | ForEach-Object { Write-Host " $_" }
# Why kept out of inner-signing-list.txt: that list drives the copy-back
# into dist/win-unpacked, and the uninstaller does not live there — it is
# re-injected through the electron-builder sign hook during the rebuild.
# No catch here, unlike the release job: the rehearsal exists to prove
# the flow, so a staging failure must fail it loudly.
$exportedUninstaller = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\unsigned\orca-uninstaller.exe'
$uninstallerStagePath = Join-Path $stage.FullName 'uninstaller\orca-uninstaller.exe'
New-Item -ItemType Directory -Force -Path (Split-Path $uninstallerStagePath) | Out-Null
Copy-Item -LiteralPath $exportedUninstaller -Destination $uninstallerStagePath -Force
Write-Host 'Staged the NSIS uninstaller for signing: uninstaller\orca-uninstaller.exe'
- name: Upload unsigned inner binaries for SignPath
id: upload-unsigned-inner
uses: actions/upload-artifact@v7
@@ -200,8 +225,27 @@ jobs:
throw "Signed inner artifact did not round-trip cleanly ($($failures.Count) failures)."
}
- name: Restore signed uninstaller for the installer rebuild
shell: pwsh
run: |
$signed = Get-ChildItem -Path signed-inner -Recurse -File -Filter 'orca-uninstaller.exe' |
Select-Object -First 1
if ($null -eq $signed) {
throw 'SignPath did not return uninstaller/orca-uninstaller.exe; check the inner-binaries artifact configuration covers it.'
}
$signature = Get-AuthenticodeSignature -FilePath $signed.FullName
if ($null -eq $signature.SignerCertificate) {
throw 'The returned NSIS uninstaller carries no signature.'
}
$signedDir = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\signed'
New-Item -ItemType Directory -Force -Path $signedDir | Out-Null
Copy-Item -LiteralPath $signed.FullName -Destination (Join-Path $signedDir 'orca-uninstaller.exe') -Force
Write-Host ("{0,-14} uninstaller <{1}>" -f $signature.Status, $signature.SignerCertificate.Subject)
- name: Build NSIS installer from signed unpacked app
shell: pwsh
env:
ORCA_WIN_UNINSTALLER_SIGNED_PATH: ${{ runner.temp }}\uninstaller-signing\signed\orca-uninstaller.exe
run: |
pnpm exec electron-builder --config config/electron-builder.config.cjs --win --publish never --prepackaged "$env:GITHUB_WORKSPACE\dist\win-unpacked"
if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }
@@ -215,6 +259,7 @@ jobs:
with:
name: orca-windows-installer-unsigned-${{ github.run_id }}
path: dist/orca-windows-setup.exe
compression-level: 0
if-no-files-found: error
- name: Submit Windows installer signing request
@@ -288,20 +333,33 @@ jobs:
run: |
$report = New-Object System.Collections.Generic.List[string]
$failures = New-Object System.Collections.Generic.List[string]
$advisories = New-Object System.Collections.Generic.List[string]
$requireValid = $env:SIGNING_POLICY -eq 'release-signing'
function Test-Signature([string]$label, [string]$path) {
# -Advisory records a problem without failing the run. It exists for
# exactly one file (resources\elevate.exe, below) and must not be
# widened casually: the point of this workflow is to fail when signing
# is broken.
function Test-Signature([string]$label, [string]$path, [switch]$Advisory) {
$signature = Get-AuthenticodeSignature -FilePath $path
$subject = if ($null -eq $signature.SignerCertificate) { '<none>' } else { $signature.SignerCertificate.Subject }
$line = "{0,-14} {1} <{2}>" -f $signature.Status, $label, $subject
$script:report.Add($line)
Write-Host $line
$problem = $null
if ($null -eq $signature.SignerCertificate -or $signature.Status -eq 'NotSigned') {
$script:failures.Add("unsigned: $label")
$problem = "unsigned: $label"
} elseif ($script:requireValid -and $signature.Status -ne 'Valid') {
$script:failures.Add("not Valid under release-signing: $label ($($signature.Status))")
$problem = "not Valid under release-signing: $label ($($signature.Status))"
} elseif ($script:requireValid -and $subject -notlike '*CN=SignPath Foundation*') {
$script:failures.Add("unexpected signer: $label ($subject)")
$problem = "unexpected signer: $label ($subject)"
}
if ($null -eq $problem) { return }
if ($Advisory) {
$script:advisories.Add($problem)
Write-Host "::warning::$problem - known pre-existing issue, not failing the rehearsal"
} else {
$script:failures.Add($problem)
}
}
@@ -323,21 +381,155 @@ jobs:
& $7za x 'dist/orca-windows-setup.exe' '-oextracted-app' -y | Out-Null
$root = Resolve-Path 'extracted-app'
# The receipt only proves the import leg ran; it cannot prove what NSIS
# embedded, because the uninstaller lives in a compressed NSIS data
# section rather than the app 7z payload above and the bundled 7za has
# no NSIS handler. So the rehearsal - unlike the release job, which
# must not mutate the runner it publishes from - goes all the way: it
# installs the installer silently and inspects the uninstaller the
# installer actually wrote to disk. That is the file MDE flags.
$signedUninstaller = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\signed\orca-uninstaller.exe'
$receipt = "$signedUninstaller.embedded-sha256"
if (-not (Test-Path -LiteralPath $receipt)) {
$failures.Add('the sign hook did not embed the signed uninstaller into the rebuilt installer')
} else {
Test-Signature 'relayed: orca-uninstaller.exe' $signedUninstaller
}
# Why a full 7-Zip attempt first: it is non-invasive. The runner image
# ships the complete 7z.exe, which - unlike the reduced 7za - has an
# NSIS handler. If it cannot read the section either, fall back to a
# real silent install.
$installedUninstaller = $null
$installedVia = $null
$expectedDigest = if (Test-Path -LiteralPath $receipt) { (Get-Content -LiteralPath $receipt -Raw).Trim() } else { $null }
$full7z = 'C:\Program Files\7-Zip\7z.exe'
if (Test-Path -LiteralPath $full7z) {
New-Item -ItemType Directory -Path nsis-extract -Force | Out-Null
& $full7z x -tnsis 'dist/orca-windows-setup.exe' '-onsis-extract' -y 2>&1 | Out-Null
$installedUninstaller = Get-ChildItem -Path nsis-extract -Recurse -File -Filter 'Uninstall*.exe' -ErrorAction SilentlyContinue |
Select-Object -First 1
# Why the digest guard before trusting this route: 7-Zip's NSIS
# handler emits partial or garbled output on some NSIS builds, and a
# truncated extract would score NotSigned and fail the rehearsal as
# "the shipped uninstaller is unsigned" when nothing is wrong. Only
# trust it when it reproduces the bytes the relay embedded; otherwise
# fall through to the install route, which is ground truth. A name
# miss (the handler labelling the entry by its source name) falls
# through the same way.
if ($null -ne $installedUninstaller -and $null -ne $expectedDigest -and
(Get-FileHash -LiteralPath $installedUninstaller.FullName -Algorithm SHA256).Hash.ToLowerInvariant() -ne $expectedDigest) {
Write-Host "7-Zip's NSIS output did not match the relayed digest; falling back to a silent install."
$installedUninstaller = $null
}
if ($null -ne $installedUninstaller) {
$installedVia = "7-Zip's NSIS handler"
Write-Host "Read the embedded uninstaller with 7-Zip's NSIS handler: $($installedUninstaller.FullName)"
} else {
Write-Host "7-Zip's NSIS handler did not yield a usable uninstaller; falling back to a silent install."
}
}
if ($null -eq $installedUninstaller) {
# Nothing here is published, so mutating this runner is free.
# Why -PassThru and a bounded wait rather than -Wait: a bare -Wait on
# an installer that ever prompts hangs to the job's 360-minute cap.
$installerProcess = Start-Process -FilePath (Resolve-Path 'dist/orca-windows-setup.exe') -ArgumentList '/S' -PassThru
if (-not $installerProcess.WaitForExit(300000)) {
$installerProcess | Stop-Process -Force -ErrorAction SilentlyContinue
$failures.Add('the silent install did not exit within 5 minutes; it is likely prompting')
}
# Why a poll rather than one Stop-Process: the oneClick installer
# launches the app as it finishes, so Orca.exe can appear *after* the
# installer process exits. A single silenced Stop-Process would miss
# it and leave Orca plus orca-terminal-daemon.exe holding handles
# under %LOCALAPPDATA%\Programs for the rest of the job.
for ($attempt = 0; $attempt -lt 20; $attempt++) {
$running = @(Get-Process -Name 'Orca' -ErrorAction SilentlyContinue)
if ($running.Count -gt 0) {
$running | Stop-Process -Force -ErrorAction SilentlyContinue
break
}
Start-Sleep -Milliseconds 500
}
Get-Process -Name 'orca-terminal-daemon' -ErrorAction SilentlyContinue |
Stop-Process -Force -ErrorAction SilentlyContinue
$installedUninstaller = Get-ChildItem -Path "$env:LOCALAPPDATA\Programs" -Recurse -File -Filter 'Uninstall*.exe' -ErrorAction SilentlyContinue |
Where-Object { $_.FullName -like '*Orca*' } |
Select-Object -First 1
if ($null -ne $installedUninstaller) { $installedVia = 'a silent install' }
}
if ($null -eq $installedUninstaller) {
$failures.Add('could not obtain the uninstaller the installer ships; neither 7-Zip nor a silent install produced it')
} else {
# Why this digest comparison is the point of the whole rehearsal:
# unlike the release job's, it hashes a file NSIS itself wrote out
# rather than the file the hook copied, so it is the only check that
# proves the shipped installer embedded the SignPath-signed bytes. On
# the 7-Zip route the guard above already forced equality; on the
# install route this is the first time it is tested.
if ($null -ne $expectedDigest) {
$shippedDigest = (Get-FileHash -LiteralPath $installedUninstaller.FullName -Algorithm SHA256).Hash.ToLowerInvariant()
if ($shippedDigest -ne $expectedDigest) {
$failures.Add("the uninstaller the installer ships is not the relayed one (via $installedVia): $shippedDigest vs $expectedDigest")
}
}
Test-Signature "shipped: Uninstall Orca.exe (via $installedVia)" $installedUninstaller.FullName
}
foreach ($relative in Get-Content 'inner-signing-list.txt') {
$path = Join-Path $root $relative
if (-not (Test-Path $path)) {
$failures.Add("missing from installer payload: $relative")
continue
}
Test-Signature "installed: $relative" $path
# Why elevate.exe alone is advisory: app-builder-lib re-copies the
# pristine cached elevate.exe over resources\elevate.exe on EVERY nsis
# pack - AppPackageHelper.packArch calls elevateHelper.copy() before
# buildAppPackage (nsisUtil.js), and CopyElevateHelper.copy does
# `copyFile(elevatePath, outFile, false)` then `signIf(outFile)`, which
# signs nothing because this build configures no certificate. So the
# signed copy restored into win-unpacked is clobbered by the rebuild.
# This predates the uninstaller relay and is not caused by it: with no
# `sign` hook, signIf already returned false at "no signing info
# identified" (windowsSignToolManager.js), so no signtool call was
# displaced. release-cut.yml mitigates it separately by pre-seeding the
# electron-builder cache ("Replace cached elevate.exe with the signed
# copy"); this workflow has no such step, which is why the clobber is
# visible here and not there. Mirroring that step here would not help:
# it only swaps when the copy is already Valid and SignPath-signed, so
# it no-ops under the test certificate.
#
# DO NOT relax that Valid + SignPath-signed guard to make this
# rehearsal go green. This workflow and release-cut.yml share the
# cache key `electron-builder-win-<lockfile hash>`, and that guard is
# the only thing stopping a test certificate from being seeded into
# the cache a real release restores from. Shipping users a binary
# signed by "Test certificate for 'Orca agent ide [OSS]'" is worse
# than shipping it unsigned.
#
# Fixing elevate.exe belongs in its own PR - it is a UAC elevation
# helper, and it deserves more scrutiny than a footnote in an
# uninstaller change.
if ($relative -eq 'resources\elevate.exe') {
Test-Signature "installed: $relative" $path -Advisory
} else {
Test-Signature "installed: $relative" $path
}
}
if ($advisories.Count -gt 0) {
$report.Add('')
$report.Add('ADVISORY (known pre-existing, did not fail this run):')
$advisories | ForEach-Object { $report.Add(" $_") }
}
Set-Content -Path 'signing-evidence.txt' -Value ($report -join "`n")
if ($failures.Count -gt 0) {
$failures | ForEach-Object { Write-Host "::error::$_" }
throw "Signing rehearsal failed with $($failures.Count) problems."
}
Write-Host "All $((Get-Content 'inner-signing-list.txt').Count) inner binaries plus the installer are signed."
Write-Host "All checked binaries are signed, including the uninstaller the installer writes to disk ($($advisories.Count) advisory)."
- name: Upload rehearsal evidence and installer
if: always()
+2
View File
@@ -110,6 +110,8 @@ docs/**
!docs/reference/macos-press-and-hold.md
!docs/reference/orcad-operations.md
!docs/reference/relay-grace-time-reconfiguration.md
!docs/reference/windows-cmd-shim-resolution.md
!docs/reference/windows-daemon-host-relocation.md
!docs/reference/windows-edr-posture.md
!docs/reference/windows-process-enumeration.md
!docs/reference/wsl-runner-verification.md
+5
View File
@@ -2,6 +2,10 @@
"$schema": "./node_modules/oxlint/configuration_schema.json",
"plugins": ["typescript", "react", "react-hooks", "react-perf", "unicorn"],
"jsPlugins": [
{
"name": "sort-comparator-performance",
"specifier": "./config/oxlint-plugins/sort-comparator-performance.mjs"
},
{
"name": "mobile-pairing",
"specifier": "./config/oxlint-plugins/mobile-pairing-qrcode-import.mjs"
@@ -23,6 +27,7 @@
"correctness": "error"
},
"rules": {
"sort-comparator-performance/no-repeated-collator": "warn",
"app-store-performance/require-selector": "error",
"app-store-performance/no-identity-selector": "error",
"app-store-performance/no-fresh-selector-result": "error",
+8 -1
View File
@@ -4,6 +4,12 @@ All UI work — layout, color, typography, spacing, component selection, UX beha
## Electron UI Validation
Always run tests and agent-launched apps in the background with `ORCA_BACKGROUND_LAUNCH=1`.
Never steal monitor focus or reveal test windows: no `show()`, `showInactive()`, `bringToFront()`,
`app.focus()`, or OS activation. Use CDP screenshots of hidden renderers. Keep native-focus and
visible-window tests paused on the user's desktop; run them on an isolated display or CI.
Rebuild modified launch-policy code before running an app; stale build wrappers are not safe.
Use the `$electron` skill and Playwright CDP for rendered Orca UI checks. Do not use computer-use for Orca UI validation.
# Style
@@ -47,8 +53,9 @@ Orca targets macOS, Linux, and Windows. Keep all platform-dependent behavior beh
- **Shortcut labels in UI**: Display `⌘` / `⇧` on Mac and `Ctrl+` / `Shift+` on other platforms.
- **File paths**: Use `path.join` or Electron/Node path utilities — never assume `/` or `\`.
- **Windows setup scripts**: the setup/issue-command runner is a `.cmd` batch file unless the script starts with a `#!` line — never derive that from the user's terminal-shell preference, and never launch a `.cmd` runner with a bare `cmd.exe /c` from a Git Bash pane (MSYS rewrites the `/c`). See [`docs/reference/windows-setup-shell.md`](./docs/reference/windows-setup-shell.md).
- **Windows child processes**: start them through `runProcess`/`spawnProcess` in `src/shared/child-process/` — never `child_process` directly. It pins `windowsHide`, refuses `shell: true`, and encodes `.cmd`/`.bat` arguments so neither `CommandLineToArgvW` nor `cmd.exe` mangles them. A ratchet test fails on any new direct import.
- **Windows child processes**: start them through `runProcess`/`spawnProcess` in `src/shared/child-process/` — never `child_process` directly. It pins `windowsHide`, refuses `shell: true`, and encodes `.cmd`/`.bat` arguments so neither `CommandLineToArgvW` nor `cmd.exe` mangles them. A ratchet test fails on any new direct import. Recognised npm/pnpm `.cmd` shims are resolved to their real target so the spawn skips `cmd.exe` entirely; see [`docs/reference/windows-cmd-shim-resolution.md`](./docs/reference/windows-cmd-shim-resolution.md) before adding a shim shape or debugging one.
- **Windows process enumeration**: read the table through `src/main/windows/windows-process-table.ts`, never by forking `powershell.exe`. See [`docs/reference/windows-process-enumeration.md`](./docs/reference/windows-process-enumeration.md).
- **Windows daemon-host relocation**: the terminal daemon runs from a copy of the app runtime under `%LOCALAPPDATA%`, which is what survives an auto-update. Before touching that copy, its exe name, or the NSIS uninstall macro, read [`docs/reference/windows-daemon-host-relocation.md`](./docs/reference/windows-daemon-host-relocation.md).
- **Windows EDR signal**: don't add `-ExecutionPolicy Bypass`, `-EncodedCommand`, `cmd.exe /c` with escaped free text, per-operation interpreter spawning, or runtime `Add-Type` compilation without reading [`docs/reference/windows-edr-posture.md`](./docs/reference/windows-edr-posture.md) first — behavioural EDR scores each of those, and being signed does not clear them.
- **WSL commands**: build argv with `buildWslExecArgs` (always `--exec` — under `--`, `wsl.exe` expands `$name` in every argument and silently rewrites the script), and fence anything whose stdout you parse with `buildWslCapturedLoginShellCommand`, because the interactive login shell prints the distro banner to stdout. See [`docs/reference/wsl-command-execution.md`](./docs/reference/wsl-command-execution.md).
- **Linux native modules**: keep the glibc floor at Ubuntu 20.04 / glibc 2.31. A module compiled from source on a newer runner can reference symbol versions absent on the floor and crash the app on startup. See [`docs/reference/linux-glibc-compatibility.md`](./docs/reference/linux-glibc-compatibility.md); packaging fails if a bundled native binary needs newer glibc.
+2 -2
View File
@@ -238,9 +238,9 @@ Pair with your desktop app to monitor and steer your agents from your phone.
- **Discord:** Join the community on **[Discord](https://discord.gg/fzjDKHxv8Q)**.
- **Twitter / X:** Follow **[@orca_build](https://x.com/orca_build)** for updates and announcements.
- **WeChat:** Scan to join the Orca community WeChat group 8.
- **WeChat:** Scan to join the Orca community WeChat group 8. Group 8 may be full; if so, scan the Group 9 QR code instead.
<img src="docs/assets/wechat-qr-group8.jpg" alt="WeChat group 8 QR code for the Orca community" width="160" />
<img src="docs/assets/wechat-qr-group8.jpg" alt="WeChat group 8 QR code for the Orca community" width="160" />&nbsp;&nbsp;<img src="docs/assets/wechat-qr-group9.jpg" alt="WeChat group 9 QR code for the Orca community" width="160" />
- **Feedback &amp; Ideas:** We ship fast. Missing something? [Request a new feature](https://github.com/stablyai/orca/issues).
- **Privacy:** See the [privacy &amp; telemetry docs](https://www.onorca.dev/docs/telemetry) for what anonymous usage data Orca collects and how to opt out.
@@ -6,7 +6,10 @@ import {
livePreflightGcloud,
runIncidentLivePreflight
} from './incident-live-preflight-cli.js'
import type { IncidentSample } from './incident-monitor.js'
import {
INCIDENT_MONITOR_THRESHOLDS,
type IncidentSample
} from './incident-monitor.js'
import type { AdmissionSelector } from './incident-selector.js'
const directories: string[] = []
@@ -69,6 +72,7 @@ function sample(): IncidentSample {
expectedSelector: selector,
cells: [{
cellId: 'production-gce-c1',
region: 'us-central1',
runtimeKnown: true,
powered: true,
expectedAdmissionState: 'existing-only'
@@ -250,6 +254,30 @@ describe('relay incident live preflight', () => {
)).rejects.toThrow('cloud-monitoring/threshold_max')
})
// Why: a frozen wave has to name what froze it without re-reading the sample.
it('names the signal and its numbers in the failure message', async () => {
const slowCell = sample()
slowCell.sources['active-probe']!.signals[
'cell.production-gce-c1.latency_ms'
]!.value = 2_568
await expect(runIncidentLivePreflight(
['--state-file', stateFile()],
{ now: () => now, collect: async () => slowCell }
)).rejects.toThrow(
'relay live preflight failed: active-probe/threshold_max cell.production-gce-c1.latency_ms observed=2568 threshold=2000'
)
// A failure with no signal keeps the source/code token and drops the rest.
const stale = sample()
stale.sources['active-probe']!.observedAt = new Date(now - 60_001).toISOString()
await expect(runIncidentLivePreflight(
['--state-file', stateFile()],
{ now: () => now, collect: async () => stale }
)).rejects.toThrow(
'relay live preflight failed: active-probe/source_stale observed=60001 threshold=60000'
)
})
it('enforces the signed migration policy', async () => {
const inactiveTarget = sample()
inactiveTarget.sources['director-admin']!.signals[
@@ -313,7 +341,7 @@ describe('relay incident live preflight', () => {
it('retries freshness-only failures when explicitly requested', async () => {
const stale = sample()
stale.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt =
new Date(now - 180_001).toISOString()
new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString()
const missing = sample()
delete missing.sources['relay-logs']
const collect = vi.fn()
@@ -331,11 +359,44 @@ describe('relay incident live preflight', () => {
expect(wait).toHaveBeenNthCalledWith(2, 15_000)
})
it('retries a first-wave stale sample and passes on the fresh one', async () => {
const stale = sample()
stale.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt =
new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString()
const collect = vi.fn().mockResolvedValueOnce(stale).mockResolvedValueOnce(sample())
const wait = vi.fn(async () => undefined)
await expect(runIncidentLivePreflight(
['--state-file', stateFile(), '--wave-index', '0', '--retry-freshness'],
{ now: () => now, collect, wait }
)).resolves.toBeUndefined()
expect(collect).toHaveBeenCalledTimes(2)
expect(wait).toHaveBeenCalledOnce()
})
it('stops retrying when the next wait would exceed the evidence-age bound', async () => {
const completedAt = now - 290_000
const stale = sample()
stale.sources['cloud-monitoring']!.observedAt = new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString()
const collect = vi.fn(async () => stale)
const wait = vi.fn(async () => undefined)
await expect(runIncidentLivePreflight(
['--state-file', stateFile('strict', {
startedAt: new Date(completedAt - 17 * 60_000).toISOString(),
windowStartedAt: new Date(completedAt - 16 * 60_000).toISOString(),
lastSampleAt: new Date(completedAt - 30_000).toISOString(),
completedAt: new Date(completedAt).toISOString()
}), '--retry-freshness'],
{ now: () => now, collect, wait }
)).rejects.toThrow('cloud-monitoring/source_stale')
expect(collect).toHaveBeenCalledOnce()
expect(wait).not.toHaveBeenCalled()
})
it('does not retry a threshold failure', async () => {
const unhealthy = sample()
unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.value = 0.9
unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt =
new Date(now - 180_001).toISOString()
new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString()
const collect = vi.fn(async () => unhealthy)
const wait = vi.fn(async () => undefined)
await expect(runIncidentLivePreflight(
@@ -348,7 +409,7 @@ describe('relay incident live preflight', () => {
it('fails closed after the bounded freshness retry window', async () => {
const stale = sample()
stale.sources['cloud-monitoring']!.observedAt = new Date(now - 180_001).toISOString()
stale.sources['cloud-monitoring']!.observedAt = new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString()
const collect = vi.fn(async () => stale)
const wait = vi.fn(async () => undefined)
await expect(runIncidentLivePreflight(
@@ -7,7 +7,9 @@ import { suppliedIdentityToken } from './incident-monitor-cli.js'
import { AdmissionSelectorSchema, type AdmissionSelector } from './incident-selector.js'
import {
evaluateIncidentSample,
FRESHNESS_FAILURE_CODES,
preDrainDryRunPassed,
type IncidentFailure,
type IncidentSample
} from './incident-monitor.js'
import { createIncidentSampleCollector } from './incident-monitor-sources.js'
@@ -18,12 +20,6 @@ const MONITOR_EVIDENCE_MAX_AGE_MS = 5 * 60_000
// Matches the same-cap cell job timeout-minutes; bounds each predecessor wave.
const WAVE_PREDECESSOR_TIMEOUT_MS = 75 * 60_000
const WAVE_INDEX_PATTERN = /^[0-3]$/
const FRESHNESS_FAILURE_CODES = new Set([
'signal_missing',
'signal_stale',
'source_missing',
'source_stale'
])
export function livePreflightGcloud(
gcloud: ReturnType<typeof createGcloudClient>,
@@ -74,6 +70,17 @@ const PreflightStateSchema = z.object({
}
})
// Keep the source/code prefix other tooling matches on, then name the signal and
// its numbers so a frozen wave is attributable without re-reading the sample.
function describeFailure(failure: IncidentFailure): string {
const detail = [
failure.signal,
failure.observed === undefined ? null : `observed=${failure.observed}`,
failure.threshold === undefined ? null : `threshold=${failure.threshold}`
].filter((part): part is string => part !== null && part !== undefined)
return [`${failure.source}/${failure.code}`, ...detail].join(' ')
}
export async function runIncidentLivePreflight(
argv: string[],
dependencies: {
@@ -173,10 +180,14 @@ export async function runIncidentLivePreflight(
const freshnessOnly = evaluation.failures.every((failure) =>
FRESHNESS_FAILURE_CODES.has(failure.code)
)
if (!freshnessOnly || attempt === attempts) {
// Waiting must never carry the mutation past the same evidence-age bound
// the entry check enforces, so the wave budget also caps the retry window.
const budgetExhausted =
now() + FRESHNESS_RETRY_INTERVAL_MS - completedAt > maxEvidenceAgeMs
if (!freshnessOnly || attempt === attempts || budgetExhausted) {
throw new Error(
`relay live preflight failed: ${evaluation.failures
.map((failure) => `${failure.source}/${failure.code}`)
.map(describeFailure)
.join(',')}`
)
}
@@ -44,6 +44,7 @@ function sample(at: number): IncidentSample {
expectedSelector: selector,
cells: [{
cellId,
region: 'us-central1',
runtimeKnown: true,
powered: true,
expectedAdmissionState: 'general'
@@ -50,6 +50,8 @@ const StateSchema = z.object({
continuityEvents: z.array(z.object({
recordedAt: z.string(),
windowSequence: z.number().int().nonnegative(),
// Pre-2026-09-05 state files predate tolerated freshness gaps.
tolerated: z.boolean().default(false),
failures: z.array(z.object({
code: z.string(),
source: z.enum(['active-probe', 'cloud-monitoring', 'relay-logs', 'director-admin']),
@@ -93,7 +93,7 @@ describe('incident monitor sources', () => {
})
it('zero-fills an expired sparse lock-wait point', async () => {
let pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs
let pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs
const fetchImpl: typeof fetch = async () => Response.json({
timeSeries: [{
points: [{
@@ -141,7 +141,7 @@ describe('incident monitor sources', () => {
it('freshens a sparse zero without masking a recent nonzero lock wait', async () => {
let value = 0
const pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs
const pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs
const readAt = now + 11_879
const fetchImpl: typeof fetch = async () => Response.json({
timeSeries: [{
@@ -317,6 +317,49 @@ describe('incident monitor sources', () => {
}, endAt)).toBeNull()
})
// Why: the per-region cell latency bar is only correct if the tfvars region
// reaches the evaluator on every cell expectation.
it('carries the configured region onto every cell expectation', async () => {
const gcloud: GcloudClient = {
accessToken: async () => 'unused',
identityToken: async () => 'unused'
}
const selector = {
generation: 1,
membership: {
existingOnly: [],
migrationOnly: [],
general: productionCells
}
}
const fetchImpl: typeof fetch = async (_input, init) => {
const body = JSON.parse(String(init?.body)) as { cellId?: string; sourceCellId?: string }
if (!body.cellId && !body.sourceCellId) return Response.json({ selector })
if (body.cellId) {
return Response.json({
status: {
enabled: true,
connectionCapacity: { hardCap: 600 },
runtime: { lastHeartbeatAt: now - 1_000, heartbeatFresh: true }
}
})
}
return Response.json({
blocked: 0,
blockedExpiredUnregistered: 0,
registeredTargetInactive: 0
})
}
const result = await directorSignals('production', selector, gcloud, now, fetchImpl)
const regionById = new Map(result.cells.map((cell) => [cell.cellId, cell.region]))
expect(regionById.get('production-gce-c1')).toBe('us-central1')
expect(regionById.get('production-gce-c27')).toBe('asia-east2')
expect(result.cells).toHaveLength(productionCells.length)
for (const cell of RELAY_OPS_ENVIRONMENTS.production.cells) {
expect(regionById.get(cell.cellId)).toBe(cell.region)
}
})
it('aggregates admin state without returning tokens or response identities', async () => {
const identityToken = 'secret.header.signature'
const sensitiveIdentity = 'user@example.test'
@@ -95,7 +95,7 @@ export const GOOGLE_METRICS: GoogleMetricDefinition[] = [
'resource.type="cloudsql_database" AND metric.label."wait_event_type"="Lock"',
aggregation: 'latest-max',
emptyIsZero: true,
zeroAfterMs: INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs
zeroAfterMs: INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs
},
{
signal: 'cloud_sql.deadlocks',
@@ -558,6 +558,7 @@ export async function directorSignals(
selector,
cells: statuses.map(({ cell }) => ({
cellId: cell.cellId,
region: cell.region,
runtimeKnown: true,
powered: true,
expectedAdmissionState: selectorCellState(expectedSelector, cell.cellId)
+295 -12
View File
@@ -2,6 +2,7 @@ import { describe, expect, it } from 'vitest'
import {
evaluateIncidentSample,
INCIDENT_CHECKPOINT_MINUTES,
INCIDENT_FRESHNESS_TOLERANCE_SAMPLES,
INCIDENT_MONITOR_THRESHOLDS,
INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS,
initialIncidentMonitorState,
@@ -32,6 +33,7 @@ function healthySample(at = startedAt): IncidentSample {
expectedSelector: selector,
cells: [{
cellId: 'production-gce-c1',
region: 'us-central1',
runtimeKnown: true,
powered: true,
expectedAdmissionState: 'general'
@@ -150,6 +152,52 @@ describe('incident monitor evaluator', () => {
})
})
// Why: an asia-east2 cell's /ready reaches auth and Cloud SQL in us-central1, so
// from the US runner it measures p50 0.88 s / max 2.7 s and the flat 2 000 bar
// froze three healthy gates on 2026-09-05 (c27 at 2568/2668/2685 ms).
it('holds cell endpoint latency to a per-region bar', () => {
const asiaTail = healthySample()
asiaTail.cells[0]!.region = 'asia-east2'
asiaTail.sources['active-probe']!.signals['cell.production-gce-c1.latency_ms'] =
signal(2_685)
expect(evaluateIncidentSample(asiaTail, startedAt)).toMatchObject({
status: 'green',
failures: []
})
const asiaIncident = healthySample()
asiaIncident.cells[0]!.region = 'asia-east2'
asiaIncident.sources['active-probe']!.signals['cell.production-gce-c1.latency_ms'] =
signal(4_001)
expect(evaluateIncidentSample(asiaIncident, startedAt)).toMatchObject({
status: 'freeze',
failures: [
expect.objectContaining({
code: 'threshold_max',
source: 'active-probe',
signal: 'cell.production-gce-c1.latency_ms',
observed: 4_001,
threshold: 4_000
})
]
})
const usIncident = healthySample()
usIncident.sources['active-probe']!.signals['cell.production-gce-c1.latency_ms'] =
signal(2_001)
expect(evaluateIncidentSample(usIncident, startedAt)).toMatchObject({
status: 'freeze',
failures: [
expect.objectContaining({
code: 'threshold_max',
signal: 'cell.production-gce-c1.latency_ms',
observed: 2_001,
threshold: 2_000
})
]
})
})
it('allows missing auth readiness and legacy existing-only connections', () => {
const sample = healthySample()
const legacySelector = {
@@ -182,12 +230,49 @@ describe('incident monitor evaluator', () => {
code: 'source_missing',
source: 'relay-logs'
})
const stale = healthySample(startedAt - 180_001)
const stale = healthySample(
startedAt - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1
)
const failures = evaluateIncidentSample(stale, startedAt).failures
expect(failures.some((failure) => failure.source === 'cloud-monitoring')).toBe(true)
expect(failures.some((failure) => failure.source === 'active-probe')).toBe(true)
})
// Why: production run 33944873727 at 2026-09-05T04:46:09Z read
// cloud_sql.lock_waits 189 286 ms old and restarted a 15-minute window on
// Google's publish lag. Cloud SQL documents 60 s sampling plus up to 165 s of
// invisibility, so that age is Google's clock, not our fleet.
it('reads a 189-second cloud signal as fresh and holds the other sources at 180 s', () => {
const lagged = healthySample()
lagged.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] =
signal(0, startedAt - 189_286)
expect(evaluateIncidentSample(lagged, startedAt)).toMatchObject({
status: 'green',
failures: []
})
const laggedDirector = healthySample()
laggedDirector.sources['director-admin']!.observedAt =
new Date(startedAt - 189_286).toISOString()
expect(evaluateIncidentSample(laggedDirector, startedAt).failures).toContainEqual(
expect.objectContaining({ code: 'source_stale', source: 'director-admin' })
)
})
it('still fails a cloud signal past the documented publish lag', () => {
const dark = healthySample()
dark.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = signal(
0,
startedAt - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1
)
expect(evaluateIncidentSample(dark, startedAt).failures).toContainEqual(
expect.objectContaining({
code: 'signal_stale',
source: 'cloud-monitoring',
signal: 'cloud_sql.lock_waits'
})
)
})
it('freezes on SQL, director, relay pool, heartbeat, and migration breaches', () => {
const sample = healthySample()
sample.sources['cloud-monitoring']!.signals['cloud_sql.cpu'] = signal(0.81)
@@ -363,12 +448,14 @@ describe('incident monitor evaluator', () => {
] = signal(0)
sample.cells.push({
cellId: 'production-gce-c2',
region: 'us-central1',
runtimeKnown: true,
powered: true,
expectedAdmissionState: 'general'
})
sample.cells.push({
cellId: 'production-gce-c3',
region: 'us-central1',
runtimeKnown: true,
powered: true,
expectedAdmissionState: 'general'
@@ -592,7 +679,7 @@ describe('incident monitor lifecycle', () => {
'restarts a %i-minute continuous window after stale telemetry',
async (durationMinutes) => {
let now = startedAt
let staleInjected = false
let staleSamples = INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1
const checkpoints: Array<[number, number]> = []
const state = initialIncidentMonitorState({
incidentId: 'incident-1',
@@ -612,9 +699,11 @@ describe('incident monitor lifecycle', () => {
now += ms
},
collect: async () => {
if (!staleInjected && now === startedAt + 5 * 60_000) {
staleInjected = true
return healthySample(now - 180_001)
if (staleSamples > 0 && now >= startedAt + 5 * 60_000) {
staleSamples--
return healthySample(
now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1
)
}
return healthySample(now)
},
@@ -623,16 +712,20 @@ describe('incident monitor lifecycle', () => {
checkpoints.push([summary.windowSequence, summary.checkpointMinute])
}
})
const restartMinute = 5 + INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1
expect(result.windowSequence).toBe(1)
expect(result.windowStartedAt).toBe(
new Date(startedAt + 6 * 60_000).toISOString()
new Date(startedAt + restartMinute * 60_000).toISOString()
)
expect(result.completedAt).toBe(
new Date(startedAt + (durationMinutes + 6) * 60_000).toISOString()
new Date(startedAt + (durationMinutes + restartMinute) * 60_000).toISOString()
)
expect(result.sampleCount).toBe(durationMinutes + 1)
expect(result.continuityEvents).toHaveLength(1)
expect(result.continuityEvents[0]!.failures).toEqual(
expect(result.continuityEvents.map((event) => event.tolerated)).toEqual([
...Array<boolean>(INCIDENT_FRESHNESS_TOLERANCE_SAMPLES).fill(true),
false
])
expect(result.continuityEvents.at(-1)!.failures).toEqual(
expect.arrayContaining([
expect.objectContaining({ code: 'source_stale' })
])
@@ -642,6 +735,188 @@ describe('incident monitor lifecycle', () => {
}
)
// Why: run 33944873727 on 2026-09-05 restarted at 04:46:09Z on a single
// 189-second cloud reading and then blew the 25-minute lineage cap, so a
// green fleet produced no verdict at all. One unread sample now continues the
// window; the sample is still checked against every threshold it can read.
it('carries a 15-minute window through a single stale cloud sample', async () => {
let now = startedAt
const state = initialIncidentMonitorState({
incidentId: 'incident-1',
environment: 'production',
expectedSelector: selector,
preDrainDryRun: true,
migrationPolicy: 'strict',
recoverySourceCellId: null,
capacityCellId: null,
startedAt: new Date(startedAt).toISOString(),
durationMinutes: 15,
intervalMs: 60_000
})
const result = await runIncidentMonitor(state, {
now: () => now,
wait: async (ms) => {
now += ms
},
collect: async () => {
const sample = healthySample(now)
if (now === startedAt + 10 * 60_000) {
sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] =
signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1)
}
return sample
},
persist: async () => {},
checkpoint: async () => {}
})
expect(result.windowSequence).toBe(0)
expect(result.windowStartedAt).toBe(new Date(startedAt).toISOString())
expect(result.completedAt).toBe(new Date(startedAt + 15 * 60_000).toISOString())
expect(result.sampleCount).toBe(16)
expect(result.frozenAt).toBeNull()
expect(result.continuityEvents).toEqual([{
recordedAt: new Date(startedAt + 10 * 60_000).toISOString(),
windowSequence: 0,
tolerated: true,
failures: [expect.objectContaining({
code: 'signal_stale',
source: 'cloud-monitoring',
signal: 'cloud_sql.lock_waits'
})]
}])
expect(preDrainDryRunPassed(result)).toBe(true)
})
it('gives a signal a fresh budget only after it reads fresh again', async () => {
let now = startedAt
const staleMinutes = new Set([3, 5, 6, 9, 10])
const state = initialIncidentMonitorState({
incidentId: 'incident-1',
environment: 'production',
expectedSelector: selector,
preDrainDryRun: true,
migrationPolicy: 'strict',
recoverySourceCellId: null,
capacityCellId: null,
startedAt: new Date(startedAt).toISOString(),
durationMinutes: 15,
intervalMs: 60_000
})
const result = await runIncidentMonitor(state, {
now: () => now,
wait: async (ms) => {
now += ms
},
collect: async () => {
const sample = healthySample(now)
if (staleMinutes.has((now - startedAt) / 60_000)) {
sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] =
signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1)
}
return sample
},
persist: async () => {},
checkpoint: async () => {}
})
expect(result.windowSequence).toBe(0)
expect(result.continuityEvents).toHaveLength(staleMinutes.size)
expect(result.continuityEvents.every((event) => event.tolerated)).toBe(true)
expect(preDrainDryRunPassed(result)).toBe(true)
})
it('does not hand a resumed monitor a fresh tolerance budget', async () => {
let now = startedAt + 3 * 60_000
const resumed = {
...initialIncidentMonitorState({
incidentId: 'incident-1',
environment: 'production',
expectedSelector: selector,
preDrainDryRun: true,
migrationPolicy: 'strict',
recoverySourceCellId: null,
capacityCellId: null,
startedAt: new Date(startedAt).toISOString(),
durationMinutes: 15,
intervalMs: 60_000
}),
windowStartedAt: new Date(startedAt).toISOString(),
lastSampleAt: new Date(startedAt + 2 * 60_000).toISOString(),
sampleCount: 3,
totalSampleCount: 3,
continuityEvents: Array.from(
{ length: INCIDENT_FRESHNESS_TOLERANCE_SAMPLES },
(_, index) => ({
recordedAt: new Date(startedAt + (index + 1) * 60_000).toISOString(),
windowSequence: 0,
tolerated: true,
failures: [{
code: 'signal_stale',
source: 'cloud-monitoring' as const,
signal: 'cloud_sql.lock_waits'
}]
})
)
}
const stop = new Error('stop after the resumed sample')
await expect(runIncidentMonitor(resumed, {
now: () => now,
wait: async () => {
throw stop
},
collect: async () => {
const sample = healthySample(now)
sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] =
signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1)
return sample
},
persist: async (state) => {
expect(state.windowSequence).toBe(1)
expect(state.windowStartedAt).toBeNull()
expect(state.continuityEvents.at(-1)!.tolerated).toBe(false)
},
checkpoint: async () => {}
})).rejects.toThrow(stop)
})
it('freezes on a threshold breach that arrives with a tolerated stale signal', async () => {
let now = startedAt
const state = initialIncidentMonitorState({
incidentId: 'incident-1',
environment: 'production',
expectedSelector: selector,
preDrainDryRun: true,
migrationPolicy: 'strict',
recoverySourceCellId: null,
capacityCellId: null,
startedAt: new Date(startedAt).toISOString(),
durationMinutes: 15,
intervalMs: 60_000
})
const result = await runIncidentMonitor(state, {
now: () => now,
wait: async (ms) => {
now += ms
},
collect: async () => {
const sample = healthySample(now)
if (now === startedAt + 2 * 60_000) {
sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] =
signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1)
sample.sources['cloud-monitoring']!.signals['cloud_sql.cpu'] = signal(0.81, now)
}
return sample
},
persist: async () => {},
checkpoint: async () => {}
})
expect(result.frozenAt).toBe(new Date(startedAt + 2 * 60_000).toISOString())
expect(result.failures).toContainEqual(expect.objectContaining({
code: 'threshold_max',
signal: 'cloud_sql.cpu'
}))
expect(preDrainDryRunPassed(result)).toBe(false)
})
it('resets at the next fresh sample after a runner gap', async () => {
let now = startedAt + 10 * 60_000
const state = {
@@ -690,13 +965,21 @@ describe('incident monitor lifecycle', () => {
durationMinutes: 15,
intervalMs: 60_000
})
let staleSamples = INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1
const result = await runIncidentMonitor(state, {
now: () => now,
wait: async (ms) => {
now += ms
},
collect: async () =>
healthySample(now === startedAt + 10 * 60_000 ? now - 180_001 : now),
collect: async () => {
if (staleSamples > 0 && now >= startedAt + 10 * 60_000) {
staleSamples--
return healthySample(
now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1
)
}
return healthySample(now)
},
persist: async () => {},
checkpoint: async () => {}
})
@@ -706,7 +989,7 @@ describe('incident monitor lifecycle', () => {
)
expect(result.frozenAt).not.toBeNull()
expect(result.windowSequence).toBe(1)
expect(result.sampleCount).toBe(15)
expect(result.sampleCount).toBe(13)
expect(result.failures).toContainEqual({
code: 'continuity_deadline_exceeded',
source: 'active-probe',
+111 -7
View File
@@ -1,3 +1,4 @@
import type { RelayOpsRegion } from './environment-config.js'
import {
exactAdmissionSelector,
type AdmissionSelector,
@@ -6,10 +7,39 @@ import {
export const INCIDENT_MONITOR_THRESHOLDS = {
activeProbeMaxAgeMs: 60_000,
cloudDataMaxAgeMs: 180_000,
// Why: Cloud Monitoring publishes on Google's clock, not ours. Per the metric
// list read 2026-09-05, Cloud Run instance_count / cpu / memory /
// max_request_concurrencies / request_count are "Sampled every 60 seconds.
// After sampling, data is not visible for up to 120 seconds" (60+120=180 s),
// and Cloud SQL cpu / memory / num_backends / backends_in_wait /
// deadlock_count say "up to 165 seconds" (60+165=225 s). Window-sum signals
// age differently: observedAt is the newest point in the 5-minute query
// window, so a label series that stops emitting reads as 300 s old while its
// summed value is still complete. 330 s clears the worst of the three (the
// 300 s query window) plus ~30 s of collect-to-evaluate latency. The old
// 180 s bar restarted healthy 15-minute windows at 181 s, 189 s and 255 s on
// 2026-09-04/05, once burning the whole 25-minute lineage with no verdict.
cloudDataMaxAgeMs: 330_000,
// Why: the director admin API answers live on our own request, so hold its
// freshness bar where it sat while it shared cloudDataMaxAgeMs.
directorAdminMaxAgeMs: 180_000,
// Why: how long a nonzero backends-in-wait point is carried before it reads as
// zero. Held at the pre-2026-09-05 cloud bar: carrying it for the full
// cloudDataMaxAgeMs would hand the evaluator a point older than its own
// freshness bar as soon as collection latency is added.
cloudLockWaitCarryMs: 180_000,
relayLogMaxAgeMs: 180_000,
heartbeatMaxAgeMs: 45_000,
endpointLatencyMs: 2_000,
// Why: a cell's /ready fetches the auth JWKS and runs SELECT 1 against Cloud SQL,
// both in us-central1, so from the US runner asia-east2 cells measure p50 0.88 s /
// max 2.7 s against 0.08-0.5 s for us-central1. The flat 2 000 bar froze three
// healthy 15-minute gates on 2026-09-05 (c27 at 2568/2668/2685 ms); hard faults
// are still caught by the .health/.ready equal-1 checks and the 8 s fetch timeout.
cellEndpointLatencyMs: {
'us-central1': 2_000,
'asia-east2': 4_000
} as const satisfies Record<RelayOpsRegion, number>,
cloudSqlCpuUtilization: 0.8,
cloudSqlMemoryUtilization: 0.9,
// Why: healthy latest-sum backends idle near 100 but spike to 216 in 1-minute
@@ -106,6 +136,7 @@ export type IncidentSource = {
export type IncidentCellExpectation = {
cellId: string
region: RelayOpsRegion
runtimeKnown: boolean
powered: boolean
expectedAdmissionState: AdmissionState
@@ -175,6 +206,7 @@ export type IncidentMonitorState = {
continuityEvents: {
recordedAt: string
windowSequence: number
tolerated: boolean
failures: IncidentFailure[]
}[]
frozenAt: string | null
@@ -307,7 +339,7 @@ const SOURCE_MAX_AGE: Record<IncidentSourceName, number> = {
'active-probe': INCIDENT_MONITOR_THRESHOLDS.activeProbeMaxAgeMs,
'cloud-monitoring': INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs,
'relay-logs': INCIDENT_MONITOR_THRESHOLDS.relayLogMaxAgeMs,
'director-admin': INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs
'director-admin': INCIDENT_MONITOR_THRESHOLDS.directorAdminMaxAgeMs
}
function ageMs(timestamp: string, nowMs: number): number {
@@ -384,7 +416,7 @@ function checkCell(
'active-probe',
probe,
`cell.${cell.cellId}.latency_ms`,
INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs,
INCIDENT_MONITOR_THRESHOLDS.cellEndpointLatencyMs[cell.region],
'max'
],
[
@@ -608,14 +640,59 @@ function checkpointMinutes(durationMinutes: number): number[] {
return INCIDENT_CHECKPOINT_MINUTES.filter((minute) => minute <= durationMinutes)
}
const CONTINUITY_FAILURE_CODES = new Set([
'collector_failed',
'monitor_gap',
// Freshness-only failures: we could not read a signal this sample. Distinct from
// collector_failed / monitor_gap, where the whole sample is absent.
export const FRESHNESS_FAILURE_CODES = new Set([
'signal_missing',
'signal_stale',
'source_missing',
'source_stale'
])
const CONTINUITY_FAILURE_CODES = new Set([
'collector_failed',
'monitor_gap',
...FRESHNESS_FAILURE_CODES
])
// Why: Cloud Monitoring overshoots its own publish bar, and one unread sample is
// not evidence of an unhealthy fleet. Under the 25-minute lineage cap a restart
// past minute 10 costs the entire verdict, so a healthy fleet produced none on
// 2026-09-05. A signal may miss this many consecutive samples before the window
// restarts; the sample is still evaluated against every threshold it can read,
// and a threshold breach still freezes the run outright.
export const INCIDENT_FRESHNESS_TOLERANCE_SAMPLES = 2
function freshnessKey(failure: IncidentFailure): string {
return `${failure.source}/${failure.signal ?? '*'}`
}
// Rebuild the per-signal tolerated streak from the trailing continuity events so a
// resumed monitor cannot hand a signal a fresh budget.
function resumeFreshnessStreaks(
state: IncidentMonitorState
): Map<string, number> {
const events = state.continuityEvents
const streaks = new Map<string, number>()
const last = events[events.length - 1]
if (!last?.tolerated) return streaks
for (const key of new Set(last.failures.map(freshnessKey))) {
let streak = 0
let laterAt: number | null = null
for (let index = events.length - 1; index >= 0; index--) {
const event = events[index]!
const recordedAt = Date.parse(event.recordedAt)
if (!event.tolerated) break
if (laterAt !== null && laterAt - recordedAt > state.intervalMs * 1.5) break
if (!event.failures.some((failure) => freshnessKey(failure) === key)) break
streak++
laterAt = recordedAt
}
streaks.set(key, streak)
}
return streaks
}
function resetContinuousWindow(
state: IncidentMonitorState,
recordedAt: string,
@@ -631,6 +708,7 @@ function resetContinuousWindow(
state.continuityEvents.push({
recordedAt,
windowSequence: state.windowSequence,
tolerated: false,
failures
})
}
@@ -681,6 +759,7 @@ export async function runIncidentMonitor(
await dependencies.persist(state)
return state
}
const freshnessStreaks = resumeFreshnessStreaks(state)
while (state.completedAt === null) {
if (dependencies.now() > lineageDeadlineMs) {
completeContinuityDeadline(state, dependencies.now(), lineageStartMs)
@@ -715,9 +794,34 @@ export async function runIncidentMonitor(
const thresholdFailures = evaluation.failures.filter((failure) =>
!CONTINUITY_FAILURE_CODES.has(failure.code)
)
if (continuityFailures.length > 0) {
const toleratedKeys = new Set(
state.windowStartedAt !== null &&
continuityFailures.length > 0 &&
continuityFailures.every((failure) => FRESHNESS_FAILURE_CODES.has(failure.code))
? continuityFailures.map(freshnessKey)
: []
)
for (const key of [...freshnessStreaks.keys()]) {
if (!toleratedKeys.has(key)) freshnessStreaks.delete(key)
}
let tolerated = toleratedKeys.size > 0
for (const key of toleratedKeys) {
const streak = (freshnessStreaks.get(key) ?? 0) + 1
freshnessStreaks.set(key, streak)
if (streak > INCIDENT_FRESHNESS_TOLERANCE_SAMPLES) tolerated = false
}
if (continuityFailures.length > 0 && !tolerated) {
freshnessStreaks.clear()
resetContinuousWindow(state, evaluation.evaluatedAt, continuityFailures)
} else {
if (tolerated) {
state.continuityEvents.push({
recordedAt: evaluation.evaluatedAt,
windowSequence: state.windowSequence,
tolerated: true,
failures: continuityFailures
})
}
if (state.windowStartedAt === null) {
state.windowStartedAt = evaluation.evaluatedAt
}
@@ -13,6 +13,41 @@ const runService = {
latestReadyRevision: 'projects/project/revisions/revision-one'
}
const sleepingStagingGcloud: GcloudClient = { accessToken: async () => 'a'.repeat(40) }
// Staging's Cloud SQL is stopped, so this inventory reads REST only and probes no endpoint.
type MigOutcome = 'ok' | 'throw' | 'missing'
const sleepingStagingFetch = (migOutcome: (migName: string) => MigOutcome): typeof fetch =>
async (input) => {
const url = new URL(String(input))
if (url.hostname === 'run.googleapis.com') return Response.json(runService)
if (url.hostname === 'sqladmin.googleapis.com') return Response.json({
state: 'STOPPED',
databaseVersion: 'POSTGRES_17',
settings: { activationPolicy: 'NEVER', availabilityType: 'ZONAL', tier: 'db-custom-1-3840' }
})
if (url.hostname === 'certificatemanager.googleapis.com') return Response.json({
managed: { domains: ['*.relay-staging.onorca.dev'], state: 'ACTIVE' }
})
if (url.pathname.includes('/instanceGroupManagers/')) {
const name = url.pathname.split('/').at(-1)!
const outcome = migOutcome(name)
if (outcome === 'throw') throw new TypeError('fetch failed')
if (outcome === 'missing') return new Response(null, { status: 404 })
return Response.json({
name,
targetSize: 0,
size: '0',
instanceGroup: `projects/project/zones/zone/instanceGroups/${name}`,
instanceTemplate: `projects/project/global/instanceTemplates/template-${name}`,
status: { isStable: true }
})
}
if (url.pathname.includes('/instanceTemplates/')) return Response.json({ properties: {} })
if (url.pathname.endsWith('/getHealth')) return Response.json([])
throw new Error(`Unexpected request to ${url.hostname}${url.pathname}`)
}
describe('readResourceInventory', () => {
it('does not delay a healthy endpoint sample', async () => {
let calls = 0
@@ -23,8 +58,10 @@ describe('readResourceInventory', () => {
calls += 1
return new Response(null, { status: 200 })
},
async () => {
waits += 1
{
wait: async () => {
waits += 1
}
}
)
@@ -45,8 +82,10 @@ describe('readResourceInventory', () => {
calls.set(path, call)
return new Response(null, { status: path === '/ready' && call === 1 ? 503 : 200 })
},
async (ms) => {
waits.push(ms)
{
wait: async (ms) => {
waits.push(ms)
}
}
)
@@ -65,17 +104,130 @@ describe('readResourceInventory', () => {
calls += 1
return new Response(null, { status: 503 })
},
async (ms) => {
waits.push(ms)
{
wait: async (ms) => {
waits.push(ms)
}
}
)
expect(result.health).toBe(false)
expect(result.ready).toBe(false)
expect(calls).toBe(4)
// A refusing endpoint is a reading, so only the independent retry runs.
expect(waits).toEqual([11_000])
})
it('treats a thrown fetch as no reading and re-asks that path once', async () => {
const calls: string[] = []
const waits: number[] = []
const result = await probeEndpointHealth(
'https://c9.relay.onorca.dev',
async (input) => {
const path = new URL(String(input)).pathname
calls.push(path)
if (path === '/health' && calls.filter((call) => call === '/health').length === 1) {
throw new TypeError('fetch failed')
}
return new Response(null, { status: 200 })
},
{
wait: async (ms) => {
waits.push(ms)
}
}
)
expect(result.health).toBe(true)
expect(result.ready).toBe(true)
expect(calls.filter((call) => call === '/health')).toEqual(['/health', '/health'])
expect(waits).toEqual([1_000])
})
it('fails closed when both attempts of a path throw', async () => {
const calls: string[] = []
const waits: number[] = []
const result = await probeEndpointHealth(
'https://c9.relay.onorca.dev',
async (input) => {
const path = new URL(String(input)).pathname
calls.push(path)
if (path === '/health') throw new TypeError('fetch failed')
return new Response(null, { status: 200 })
},
{
wait: async (ms) => {
waits.push(ms)
}
}
)
expect(result.health).toBe(false)
expect(calls.filter((call) => call === '/health')).toHaveLength(4)
expect(waits).toEqual([1_000, 11_000, 1_000])
})
it('accepts an auth-shaped endpoint that serves no readiness path', async () => {
const calls: string[] = []
const waits: number[] = []
const result = await probeEndpointHealth(
'https://login.onorca.dev',
async (input) => {
const path = new URL(String(input)).pathname
calls.push(path)
return new Response(null, { status: path === '/ready' ? 404 : 200 })
},
{
requiresReady: false,
wait: async (ms) => {
waits.push(ms)
}
}
)
expect(result.health).toBe(true)
expect(result.ready).toBeNull()
expect(calls).toEqual(['/health'])
expect(waits).toEqual([])
})
it('still requires readiness for the director and cells', async () => {
const waits: number[] = []
const result = await probeEndpointHealth(
'https://relay.onorca.dev',
async (input) => new Response(null, {
status: new URL(String(input)).pathname === '/ready' ? 503 : 200
}),
{
wait: async (ms) => {
waits.push(ms)
}
}
)
expect(result.health).toBe(true)
expect(result.ready).toBe(false)
expect(waits).toEqual([11_000])
})
it('measures latency as the answering round trip, not the retry delay', async () => {
let healthCalls = 0
const result = await probeEndpointHealth(
'https://c9.relay.onorca.dev',
async (input) => {
if (new URL(String(input)).pathname !== '/health') return new Response(null, { status: 200 })
healthCalls += 1
if (healthCalls === 1) throw new TypeError('fetch failed')
return new Response(null, { status: 200 })
},
{ wait: async (ms) => await new Promise((resolve) => setTimeout(resolve, Math.min(ms, 60))) }
)
expect(result.health).toBe(true)
expect(result.latencyMs).not.toBeNull()
expect(result.latencyMs!).toBeLessThan(60)
})
it('uses aggregate REST inventory without probing sleeping staging endpoints', async () => {
const gcloud: GcloudClient = { accessToken: async () => 'a'.repeat(40) }
let publicProbeCalls = 0
@@ -132,6 +284,74 @@ describe('readResourceInventory', () => {
expect(JSON.stringify(result)).not.toContain('SECRET_TEXT')
})
it('re-asks a MIG read that failed once before calling a cell powered-unknown', async () => {
const parkedCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]!
const waits: number[] = []
let parkedMigCalls = 0
const result = await readResourceInventory(
RELAY_OPS_ENVIRONMENTS.staging,
sleepingStagingGcloud,
sleepingStagingFetch((migName) => {
if (!migName.endsWith(parkedCell.hostname)) return 'ok'
parkedMigCalls += 1
return parkedMigCalls === 1 ? 'throw' : 'ok'
}),
{ wait: async (ms) => { waits.push(ms) } }
)
const parked = result.cells.find((cell) => cell.cellId === parkedCell.cellId)!
// The MIG was fine and parked at zero; one transient read must not erase that reading.
expect(parked.targetSize).toBe(0)
expect(parkedMigCalls).toBe(2)
expect(waits).toEqual([1_000])
expect(result.warnings).toEqual([])
})
it('reports a MIG unavailable only when the retry fails too', async () => {
const parkedCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]!
const waits: number[] = []
let parkedMigCalls = 0
const result = await readResourceInventory(
RELAY_OPS_ENVIRONMENTS.staging,
sleepingStagingGcloud,
sleepingStagingFetch((migName) => {
if (!migName.endsWith(parkedCell.hostname)) return 'ok'
parkedMigCalls += 1
return 'throw'
}),
{ wait: async (ms) => { waits.push(ms) } }
)
const parked = result.cells.find((cell) => cell.cellId === parkedCell.cellId)!
expect(parked.targetSize).toBeNull()
expect(parked.backendHealth).toBe('unknown')
expect(parkedMigCalls).toBe(2)
expect(waits).toEqual([1_000])
expect(result.warnings).toEqual([
`${parkedCell.hostname.toUpperCase()} MIG inventory is unavailable.`
])
})
it('does not re-ask a MIG read the API answered with 404', async () => {
const missingCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]!
const waits: number[] = []
let missingMigCalls = 0
const result = await readResourceInventory(
RELAY_OPS_ENVIRONMENTS.staging,
sleepingStagingGcloud,
sleepingStagingFetch((migName) => {
if (!migName.endsWith(missingCell.hostname)) return 'ok'
missingMigCalls += 1
return 'missing'
}),
{ wait: async (ms) => { waits.push(ms) } }
)
expect(result.cells.find((cell) => cell.cellId === missingCell.cellId)!.targetSize).toBeNull()
expect(missingMigCalls).toBe(1)
expect(waits).toEqual([])
})
it('represents missing credentials as unknown inventory, never sleeping', async () => {
const gcloud: GcloudClient = {
accessToken: async () => { throw new Error('sensitive context') }
+94 -23
View File
@@ -102,6 +102,9 @@ export type ResourceInventory = {
const unavailableEndpoint = (): EndpointHealth => ({ health: null, ready: null, latencyMs: null })
const independentEndpointRetryDelayMs = 11_000
const transientProbeRetryDelayMs = 1_000
const sleep = async (ms: number): Promise<void> =>
await new Promise((resolvePromise) => setTimeout(resolvePromise, ms))
function finalSegment(value: string): string {
return value.split('/').at(-1) ?? value
@@ -120,6 +123,12 @@ function parseService(value: unknown): ServiceInventory {
}
}
class GoogleApiError extends Error {
constructor(readonly status: number) {
super(`Google API returned ${status}`)
}
}
async function googleRequest(
fetchImpl: typeof fetch,
token: string,
@@ -134,45 +143,97 @@ async function googleRequest(
},
signal: AbortSignal.timeout(30_000)
})
if (!response.ok) throw new Error(`Google API returned ${response.status}`)
if (!response.ok) throw new GoogleApiError(response.status)
return await response.json()
}
async function endpointProbe(origin: string, fetchImpl: typeof fetch): Promise<EndpointHealth> {
const startedAt = performance.now()
const check = async (path: '/health' | '/ready'): Promise<boolean> => {
// A 404 is the API's answer about the resource; anything else is the absence of a reading, so re-ask.
async function readOnceMore(
read: () => Promise<unknown>,
wait: (ms: number) => Promise<void>
): Promise<unknown> {
try {
return await read()
} catch (error) {
if (error instanceof GoogleApiError && error.status === 404) throw error
await wait(transientProbeRetryDelayMs)
return await read()
}
}
// A reading the endpoint actually produced: ok is its answer, latencyMs is that answer's round trip.
type PathReading = { ok: boolean; latencyMs: number | null }
async function probePath(
origin: string,
path: '/health' | '/ready',
fetchImpl: typeof fetch,
wait: (ms: number) => Promise<void>
): Promise<PathReading> {
// null means the request never produced an answer (DNS/TCP/TLS failure or the 8s abort).
const attempt = async (): Promise<PathReading | null> => {
const startedAt = performance.now()
try {
const response = await fetchImpl(`${origin}${path}`, {
redirect: 'error',
signal: AbortSignal.timeout(8_000)
})
return response.ok
return { ok: response.ok, latencyMs: Math.round(performance.now() - startedAt) }
} catch {
return false
return null
}
}
const [health, ready] = await Promise.all([check('/health'), check('/ready')])
return { health, ready, latencyMs: Math.round(performance.now() - startedAt) }
const first = await attempt()
if (first) return first
// A thrown fetch is the absence of a reading, not an unhealthy answer, so re-ask before concluding.
await wait(transientProbeRetryDelayMs)
return (await attempt()) ?? { ok: false, latencyMs: null }
}
async function endpointProbe(
origin: string,
fetchImpl: typeof fetch,
requiresReady: boolean,
wait: (ms: number) => Promise<void>
): Promise<EndpointHealth> {
const [health, ready] = await Promise.all([
probePath(origin, '/health', fetchImpl, wait),
requiresReady ? probePath(origin, '/ready', fetchImpl, wait) : null
])
// Latency is the slowest answering round trip in this probe; retry delays are not serving latency.
const latencies = [health.latencyMs, ready?.latencyMs ?? null].filter(
(value): value is number => value !== null
)
return {
health: health.ok,
ready: ready ? ready.ok : null,
latencyMs: latencies.length > 0 ? Math.max(...latencies) : null
}
}
export type EndpointProbeOptions = {
// Auth serves no /ready by design, so it is judged on /health and latency alone.
requiresReady?: boolean
wait?: (ms: number) => Promise<void>
}
export async function probeEndpointHealth(
origin: string,
fetchImpl: typeof fetch,
wait: (ms: number) => Promise<void> = async (ms) =>
await new Promise((resolvePromise) => setTimeout(resolvePromise, ms))
options: EndpointProbeOptions = {}
): Promise<EndpointHealth> {
const first = await endpointProbe(origin, fetchImpl)
if (
first.health &&
first.ready &&
first.latencyMs !== null &&
first.latencyMs <= INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs
) {
return first
}
const requiresReady = options.requiresReady ?? true
const wait = options.wait ?? sleep
const accepted = (probe: EndpointHealth): boolean =>
probe.health === true &&
(!requiresReady || probe.ready === true) &&
probe.latencyMs !== null &&
probe.latencyMs <= INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs
const first = await endpointProbe(origin, fetchImpl, requiresReady, wait)
if (accepted(first)) return first
// Outwait Relay's ten-second readiness cache before treating the retry as independent.
await wait(independentEndpointRetryDelayMs)
return await endpointProbe(origin, fetchImpl)
return await endpointProbe(origin, fetchImpl, requiresReady, wait)
}
function imageDigest(template: z.infer<typeof TemplateSchema>): string | null {
@@ -285,11 +346,17 @@ function unavailableInventory(environment: RelayOpsEnvironment, warning: string)
}
}
export type ResourceInventoryOptions = {
wait?: (ms: number) => Promise<void>
}
export async function readResourceInventory(
environment: RelayOpsEnvironment,
gcloud: GcloudClient,
fetchImpl: typeof fetch = fetch
fetchImpl: typeof fetch = fetch,
options: ResourceInventoryOptions = {}
): Promise<ResourceInventory> {
const wait = options.wait ?? sleep
let token: string
try {
token = await gcloud.accessToken()
@@ -316,7 +383,10 @@ export async function readResourceInventory(
token,
`https://certificatemanager.googleapis.com/v1/projects/${environment.project}/locations/global/certificates/${environment.certificateName}`
),
...environment.cells.map((cell) => googleRequest(fetchImpl, token, migUrl(cell)))
// One transient Compute read must never become a verdict on a cell's power state.
...environment.cells.map((cell) =>
readOnceMore(async () => await googleRequest(fetchImpl, token, migUrl(cell)), wait)
)
])
const warnings: string[] = []
const directorValue = parsed(settled[0]!, RunServiceSchema, 'Director service inventory is unavailable.', warnings)
@@ -338,7 +408,8 @@ export async function readResourceInventory(
? [unavailableEndpoint(), unavailableEndpoint()]
: await Promise.all([
probeEndpointHealth(environment.directorOrigin, fetchImpl),
probeEndpointHealth(environment.authOrigin, fetchImpl)
// The auth service exposes no /ready, so requiring it would fail every first probe.
probeEndpointHealth(environment.authOrigin, fetchImpl, { requiresReady: false })
])
const cells = await Promise.all(environment.cells.map((cell, index) =>
readCell(environment, cell, migValues[index] ?? null, token, fetchImpl)
@@ -44,6 +44,12 @@ describePostgres('PostgreSQL assignment connection headroom', () => {
`DELETE FROM relay_assignments
WHERE user_id LIKE 'connection-headroom-postgres-%'`
)
// A snapshot left by an aborted run rejects the replayed watermark
// with stale_connection_snapshot.
await database.query(
`DELETE FROM relay_cell_connection_snapshots WHERE cell_id = ?`,
[cell.id]
)
await database.query(
`DELETE FROM relay_cell_connection_runtime WHERE cell_id = ?`,
[cell.id]
@@ -38,6 +38,10 @@ describePostgres('PostgreSQL control supersession', () => {
[identity.userId]
)
await database.query(`DELETE FROM relay_assignments WHERE user_id = ?`, [identity.userId])
// A snapshot left by an aborted run rejects the replayed watermark with stale_connection_snapshot.
await database.query(`DELETE FROM relay_cell_connection_snapshots WHERE cell_id = ?`, [
cell.id
])
await database.query(`DELETE FROM relay_cell_connection_runtime WHERE cell_id = ?`, [cell.id])
await database.query(`DELETE FROM relay_cell_connection_limits WHERE cell_id = ?`, [cell.id])
await database.query(`DELETE FROM relay_cell_runtime WHERE cell_id = ?`, [cell.id])
+83 -25
View File
@@ -645,9 +645,17 @@ export class RelayAssignmentStore {
): Promise<RelayAssignment | null> {
const now = this.now()
return await this.database.transaction(async (transaction) => {
const lockedCells = inventoryFirst
? await this.lockCellInventory(transaction, lockMode)
// Why: the retry exists to take a cell row before the assignment row, the
// order placement uses. It only ever needs the one cell this host is
// pinned to, so read the pin unlocked and lock that row alone; taking all
// 23 queued every sticky refresh in the fleet behind every other one.
const pinnedCellId = inventoryFirst
? await this.pinnedCellId(transaction, identity)
: undefined
const lockedCells =
pinnedCellId === undefined
? undefined
: await this.lockCellRows(transaction, [pinnedCellId], lockMode)
const existing = await this.assignmentRow(transaction, identity, inventoryFirst)
if (!existing) return null
const activityLeases = await this.lockAssignmentActivities(transaction, identity, true)
@@ -661,6 +669,11 @@ export class RelayAssignmentStore {
}
const currentCellId = text(existing, 'cell_id')
// The pin moved between the unlocked read and the assignment lock, so the
// row held is the wrong one. Same recovery as losing the lock: retry.
if (pinnedCellId !== undefined && pinnedCellId !== currentCellId) {
throw new Error('database_lock_unavailable')
}
const hadControl = holdsControlLease(
activityLeases,
currentCellId,
@@ -701,14 +714,9 @@ export class RelayAssignmentStore {
if (hadControl) {
await this.touchAssignment(transaction, identity, leaseExpiresAt, now)
} else {
const nextReservation = integer(currentRow, 'reserved_requests') + 1
if (nextReservation > integer(currentRow, 'capacity_requests')) {
throw new Error('relay_capacity_exhausted')
}
await transaction.query(
`UPDATE relay_cells SET reserved_requests = ?, updated_at = ? WHERE cell_id = ?`,
[nextReservation, now, currentCellId]
)
// Delta, not the value read from the snapshot: an absolute write here
// would clobber any concurrent movement of the same counter.
await this.adjustCellReservationAtomically(transaction, currentCellId, 1)
await this.adjustActivityCount(transaction, identity, 'control', 1, leaseExpiresAt, now)
await this.insertPendingControlLease(
transaction,
@@ -3202,8 +3210,7 @@ export class RelayAssignmentStore {
)
const requestDelta = ACTIVITY_REQUEST_UNITS[kind] * (after - before)
if (requestDelta !== 0) {
await this.lockCellInventory(transaction, 'request')
await this.adjustCellReservation(transaction, text(row, 'cell_id'), requestDelta)
await this.adjustCellReservationAtomically(transaction, text(row, 'cell_id'), requestDelta)
}
})
})
@@ -3263,9 +3270,12 @@ export class RelayAssignmentStore {
}
const units = ACTIVITY_REQUEST_UNITS[input.kind]
if (existing) {
await this.lockCellInventory(transaction, 'request')
// Why: a client-chosen activity id can move between cells, so lock the
// one or two rows this path touches in cell_id order, the same order
// placement takes the inventory in, and no cycle can form.
await this.lockCellRows(transaction, [text(existing, 'cell_id'), input.cellId])
await this.removeActivityLease(transaction, identity, existing, now)
await this.adjustCellReservation(transaction, input.cellId, units)
await this.adjustCellReservationAtomically(transaction, input.cellId, units)
}
await this.adjustActivityCount(transaction, identity, input.kind, 1, expiresAt, now)
await transaction.query(
@@ -3580,8 +3590,7 @@ export class RelayAssignmentStore {
)
await this.touchAssignment(transaction, identity, expiresAt, now)
} else {
await this.lockCellInventory(transaction, 'request')
await this.adjustCellReservation(transaction, input.cellId, 1)
await this.adjustCellReservationAtomically(transaction, input.cellId, 1)
await this.adjustActivityCount(transaction, identity, 'control', 1, expiresAt, now)
await transaction.query(
`INSERT INTO relay_assignment_activity_leases
@@ -6863,13 +6872,24 @@ export class RelayAssignmentStore {
targetCellId
]
)
const cells = await this.lockCellInventory(transaction, 'pool-default')
// Only the two cells this repairs need holding. The id set below is an
// existence check against a table that only reconcileCells writes, so it
// reads unlocked instead of dragging the other 21 rows into the section.
const cellIds = new Set(
(await transaction.query(`SELECT cell_id FROM relay_cells`)).map((row) =>
text(row, 'cell_id')
)
)
const cells = await this.lockCellRows(
transaction,
[sourceCellId, targetCellId],
'pool-default'
)
const assignmentKeys = new Set(
assignments.map((row) =>
assignmentKey(text(row, 'user_id'), text(row, 'relay_host_id'))
)
)
const cellIds = new Set(cells.map((row) => text(row, 'cell_id')))
const assignmentCounts = new Map<
string,
{ counts: Record<AssignmentActivityKind, number>; leaseExpiresAt: number }
@@ -6923,9 +6943,7 @@ export class RelayAssignmentStore {
)
}
for (const row of cells.filter((cell) =>
[sourceCellId, targetCellId].includes(text(cell, 'cell_id'))
)) {
for (const row of cells) {
const cellId = text(row, 'cell_id')
const expected = cellUnits.get(cellId) ?? 0
if (expected > integer(row, 'capacity_requests')) {
@@ -6954,6 +6972,43 @@ export class RelayAssignmentStore {
return rows
}
// Per-connection paths touch one or two cells. Locking exactly those rows,
// in the same ascending order the inventory lock uses (ORDER BY fixes the
// row-lock order), keeps them off the fleet-wide lock without a cycle.
// The wait policy follows the caller for the same reason the inventory lock's
// does: a sweep must not fail terminally on ordinary contention. Hold time is
// deliberately not sampled here — the metric tracks the fleet-wide lock these
// rows replace, and mixing in short single-row holds would flatter it.
private async lockCellRows(
database: RelayDatabase,
cellIds: string[],
mode: CellInventoryLockMode = 'request'
): Promise<SqlRow[]> {
const distinct = [...new Set(cellIds)]
const { measureHoldMs: _sampled, ...wait } = cellInventoryLockOptions(mode)
return await database.queryLocked(
`SELECT * FROM relay_cells WHERE cell_id IN (${distinct.map(() => '?').join(', ')})
ORDER BY cell_id ASC`,
distinct,
wait
)
}
// Unlocked on purpose: this only names the row to lock next, and the caller
// re-checks the pin once the assignment row is held.
private async pinnedCellId(
database: RelayDatabase,
identity: AssignmentIdentity
): Promise<string | undefined> {
const row = (
await database.query(
`SELECT cell_id FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`,
[identity.userId, identity.relayHostId]
)
)[0]
return row ? text(row, 'cell_id') : undefined
}
private async lockGeneralCellInventory(
database: RelayDatabase,
mode: CellInventoryLockMode
@@ -6972,10 +7027,11 @@ export class RelayAssignmentStore {
private async leastLoadedCell(
database: RelayDatabase,
lockedCells: SqlRow[] | undefined,
// Required: the one caller has already locked the inventory it selects from,
// and an optional parameter left a second fleet-wide lock reachable here.
rows: SqlRow[],
preferredRegion: RelayRegion
): Promise<CellRow | null> {
const rows = lockedCells ?? (await this.lockCellInventory(database, 'pool-default'))
const regions = new Map(
(await database.query(`SELECT cell_id, region FROM relay_cell_regions`)).map((row) => [
text(row, 'cell_id'),
@@ -7590,7 +7646,10 @@ export class RelayAssignmentStore {
) {
throw new Error('activity_lease_shape_mismatch')
}
const cells = await this.lockCellInventory(database, 'request')
// Why: this recomputes one cell's reservation from its leases, so only that
// row needs to be held; the 23-row inventory lock here serialised every
// desktop control rebind in the fleet behind every other one.
const cellRow = (await this.lockCellRows(database, [cellId]))[0]
await database.query(
`DELETE FROM relay_assignment_activity_leases
WHERE user_id = ? AND relay_host_id = ? AND activity_kind = 'control'
@@ -7611,7 +7670,6 @@ export class RelayAssignmentStore {
[cellId]
)
)[0]!
const cellRow = cells.find((cell) => text(cell, 'cell_id') === cellId)
const cellUnits = integer(cellUnitsRow, 'request_units')
if (!cellRow) throw new Error('assigned_cell_missing')
if (cellUnits > integer(cellRow, 'capacity_requests')) {
@@ -18,17 +18,21 @@ type CensusEntry = { method: string; mode: CensusMode; reach: Reachability }
// assignment-store.ts, in source order. A new site fails this test until it is
// classified here, which is the point.
const CENSUS: CensusEntry[] = [
{ method: 'assignStickyOnce', mode: 'caller', reach: 'both' },
// assignStickyOnce is gone from this list: its retry now locks only the row
// the host is pinned to (lockCellRows), which is what a sticky refresh
// touches. Placement below is the one genuinely fleet-wide decision left.
{ method: 'assignOnce', mode: 'caller', reach: 'both' },
{ method: 'assignOnce', mode: 'caller', reach: 'both' },
{ method: 'assignOnce', mode: 'nowait', reach: 'both' },
{ method: 'assignOnce', mode: 'nowait', reach: 'both' },
{ method: 'assignOnce', mode: 'nowait', reach: 'both' },
{ method: 'refreshDrainMigrationLeasesOnce', mode: 'request', reach: 'request' },
// Reachable from neither: changeActivity has no production callers, only tests.
{ method: 'changeActivity', mode: 'request', reach: 'orphan' },
{ method: 'acquireActivity', mode: 'request', reach: 'request' },
{ method: 'activateControl', mode: 'request', reach: 'request' },
// changeActivity, acquireActivity, activateControl and
// removeSupersededSameCellControls no longer take the inventory: they lock
// only the one or two cell rows they touch, in cell_id order (lockCellRows),
// so they cannot cycle with placement's ordered inventory lock, and the
// 23-row lock there had serialised every reconnect in the fleet behind every
// other one.
{ method: 'startEvacuation', mode: 'request', reach: 'request' },
{ method: 'completeEvacuationFromDeadSourceOnce', mode: 'request', reach: 'request' },
{ method: 'completeEvacuationFromDeadSourceOnce', mode: 'nowait', reach: 'request' },
@@ -47,9 +51,33 @@ const CENSUS: CensusEntry[] = [
{ method: 'abortExpiredEvacuations', mode: 'nowait', reach: 'sweep' },
{ method: 'releaseExpiredActivityLeases', mode: 'nowait', reach: 'sweep' },
{ method: 'releaseExpiredActivity', mode: 'nowait', reach: 'sweep' },
{ method: 'reconcileReservationAccounting', mode: 'pool-default', reach: 'both' },
{ method: 'leastLoadedCell', mode: 'pool-default', reach: 'both' },
{ method: 'removeSupersededSameCellControls', mode: 'request', reach: 'request' }
// reconcileReservationAccounting and leastLoadedCell are gone too: the first
// repairs exactly two cells' counters and now holds only those rows, and the
// second selects from the inventory its single caller has already locked.
]
// Every inline `FROM relay_cells ... FOR UPDATE` outside the named lock helpers,
// in source order: whole-table locks in reconciliation and sticky placement,
// and single-row locks for a cell the method is already scoped to (heartbeat,
// fence, drain generation, configuration, or a reservation adjust that runs
// under a lock its caller already holds). A new inline lock fails the census
// below until it is listed here; per-connection paths that touch more than one
// cell go through lockCellRows so the order is fixed.
const NAMED_LOCK_HELPERS = ['lockCellInventory', 'lockGeneralCellInventory', 'lockCellRows']
const INLINE_CELL_LOCK_SITES = [
'reconcileCellsWithOptions',
'assignStickyOnce',
'recordCellHeartbeat',
'attestCellFence',
'adoptLegacyCellFence',
'commitLegacyCellFenceAdoption',
'prepareCellFenceAttempt',
'attestCellFenceAttempt',
'attestCellFenceAttempt',
'configureCell',
'assertDrainCellGeneration',
'adjustCellReservation'
]
// The background sweeps, and nothing else. A method reachable from one of these
@@ -151,6 +179,42 @@ describe('cell inventory lock call-site census', () => {
)
})
// Why: the census only sees lockCellInventory calls, so a hand-written
// `relay_cells ... FOR UPDATE` would escape classification entirely.
it('routes every relay_cells row lock through a named lock helper', () => {
const lines = storeSource()
const rawSites: string[] = []
// Whole statements, not a fixed window: a wide column list or a raw
// FOR UPDATE inside query() must not slip past.
const source = lines.join('\n')
const bounds: { name: string; start: number }[] = []
lines.forEach((line, index) => {
const declaration = DECLARATION.exec(line)
if (declaration) bounds.push({ name: declaration[1]!, start: index })
})
const methodAt = (offset: number): string => {
const lineIndex = source.slice(0, offset).split('\n').length - 1
let name = '<module>'
for (const bound of bounds) if (bound.start <= lineIndex) name = bound.name
return name
}
const tick = String.fromCharCode(96)
const statementCall = new RegExp(
'\\.(queryLocked|query)\\(\\s*' + tick + '([^' + tick + ']*)' + tick,
'g'
)
for (const call of source.matchAll(statementCall)) {
const statement = call[2]!
if (!/\bFROM\s+relay_cells\b/.test(statement)) continue
const locks = call[1] === 'queryLocked' || /\bFOR\s+UPDATE\b/.test(statement)
if (!locks) continue
const method = methodAt(call.index)
if (NAMED_LOCK_HELPERS.includes(method)) continue
rawSites.push(method)
}
expect(rawSites).toEqual(INLINE_CELL_LOCK_SITES)
})
it('leaves no call site taking the inventory without naming a mode', () => {
const source = readFileSync(new URL('./assignment-store.ts', import.meta.url), 'utf8')
const unclassified = source
@@ -0,0 +1,206 @@
import { afterAll, beforeAll, describe, expect, it } from 'vitest'
import { RelayAssignmentStore } from './assignment-store.js'
import { openRelayDatabase, type RelayDatabase } from './database.js'
const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL
const describePostgres = databaseUrl ? describe : describe.skip
// Sorted ascending, and the host is pinned to the LAST id on purpose: the
// fleet-wide lock is one ordered scan, so it holds every earlier row while it
// waits on the pinned one. Pinning to the first id would make the two locking
// models indistinguishable.
const cells = ['a', 'b', 'c'].map((suffix) => ({
id: `percell-postgres-${suffix}`,
url: `https://percell-postgres-${suffix}.example.com`,
capacityRequests: 1_000,
connectionHardCap: 600 as const,
connectionUnobservedBound: 50
}))
const [cellA, cellB, cellC] = cells as [(typeof cells)[0], (typeof cells)[0], (typeof cells)[0]]
const identity = { userId: 'percell-postgres-user', relayHostId: 'percellhost00001' }
function heartbeat(cell: (typeof cells)[number]) {
return {
cellId: cell.id,
cellUrl: cell.url,
cellIncarnation: '11111111-1111-4111-8111-111111111111',
startedAt: 50,
ready: true,
observedRequests: 0,
totalConnections: 0,
inFlightConnections: 0,
reservedConnectionUnits: 0,
enforcedConnectionUnits: 0,
connectionInclusionWatermark: 1,
connectionHardCap: 600 as const,
connectionUnobservedBound: 50
}
}
describePostgres('PostgreSQL per-cell inventory locking', () => {
const databases: RelayDatabase[] = []
beforeAll(async () => {
for (let index = 0; index < 3; index++) {
databases.push(await openRelayDatabase({ databaseUrl, dataDir: '' }))
}
})
async function removeTestRows(database: RelayDatabase): Promise<void> {
await database.query(
`DELETE FROM relay_control_connection_reservations WHERE user_id LIKE 'percell-postgres-%'`
)
for (const table of [
'relay_assignment_activity_leases',
'relay_post_drain_migration_pins',
'relay_assignment_migration_incarnations',
'relay_assignment_migrations',
'relay_assignment_region_preferences',
'relay_assignments'
]) {
await database.query(`DELETE FROM ${table} WHERE user_id LIKE 'percell-postgres-%'`)
}
for (const cell of cells) {
for (const table of [
'relay_cell_connection_snapshots',
'relay_cell_connection_runtime',
'relay_cell_connection_limits',
'relay_cell_runtime',
'relay_cells'
]) {
await database.query(`DELETE FROM ${table} WHERE cell_id = ?`, [cell.id])
}
}
}
afterAll(async () => {
if (databases[0]) await removeTestRows(databases[0])
for (const connection of databases) await connection.close()
})
async function pinHostToLastCell(store: RelayAssignmentStore): Promise<void> {
await store.reconcileCells(cells)
for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell))
await store.setCellEnabled(cellA.id, false)
await store.setCellEnabled(cellB.id, false)
const assignment = await store.assign(identity)
expect(assignment.cellId).toBe(cellC.id)
await store.setCellEnabled(cellA.id, true)
await store.setCellEnabled(cellB.id, true)
}
async function lockWaiterAppeared(database: RelayDatabase): Promise<boolean> {
const deadline = Date.now() + 4_000
while (Date.now() < deadline) {
const rows = await database.query(
`SELECT count(*) AS waiting FROM pg_stat_activity
WHERE datname = current_database() AND wait_event_type = 'Lock'`
)
if (Number(rows[0]!.waiting) > 0) return true
await new Promise((resolve) => setTimeout(resolve, 10))
}
return false
}
// Why: a sticky refresh whose first NOWAIT probe loses retries by taking a
// cell row before the assignment row. That retry used to take the whole
// inventory, so one busy cell stalled every other cell's reconnects.
it('waits only on the pinned cell row while refreshing a sticky assignment', async () => {
await removeTestRows(databases[0]!)
const store = new RelayAssignmentStore(databases[0]!, () => 100)
await pinHostToLastCell(store)
// A host whose control lease was already reaped still holds its pin; that
// is the shape that reaches the cell-row probe instead of touchAssignment.
await databases[0]!.query(
`DELETE FROM relay_assignment_activity_leases WHERE user_id = ?`,
[identity.userId]
)
let releaseRow!: () => void
const rowReleased = new Promise<void>((resolve) => {
releaseRow = resolve
})
let rowHeld!: () => void
const rowHeldPromise = new Promise<void>((resolve) => {
rowHeld = resolve
})
const holder = databases[1]!.transaction(async (transaction) => {
await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cellC.id])
rowHeld()
await rowReleased
})
await rowHeldPromise
const refresh = store.assign(identity)
expect(await lockWaiterAppeared(databases[2]!)).toBe(true)
// The refresh is blocked on cell C. Every earlier row must still be free:
// the ordered fleet-wide scan would be holding both of them by now.
const heldWhileRefreshWaits: string[] = []
await databases[2]!.transaction(async (transaction) => {
for (const cell of [cellA, cellB]) {
try {
await transaction.queryLocked(
`SELECT * FROM relay_cells WHERE cell_id = ?`,
[cell.id],
{ failIfUnavailable: true }
)
} catch {
heldWhileRefreshWaits.push(cell.id)
}
}
})
releaseRow()
await holder
expect(heldWhileRefreshWaits).toEqual([])
expect((await refresh).cellId).toBe(cellC.id)
}, 15_000)
// Why: the counter moves by a delta now instead of an absolute value read
// from a snapshot, so concurrent movement on the same cell must still sum.
it('keeps a cell reservation exact under concurrent same-cell activity', async () => {
await removeTestRows(databases[0]!)
const store = new RelayAssignmentStore(databases[0]!, () => 100)
await store.reconcileCells(cells)
for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell))
await store.setCellEnabled(cellA.id, false)
await store.setCellEnabled(cellB.id, false)
const hosts = Array.from({ length: 6 }, (_, index) => ({
userId: `percell-postgres-user-${index}`,
relayHostId: `percellhost0000${index}`
}))
const stores = databases.map((database) => new RelayAssignmentStore(database, () => 100))
await Promise.all(hosts.map((host, index) => stores[index % stores.length]!.assign(host)))
// One splice each (2 units) on the same cell, from three connections at once.
await Promise.all(
hosts.map((host, index) =>
stores[index % stores.length]!.acquireActivity(host, {
activityId: `splice:percell-${index}`,
kind: 'splice',
cellId: cellC.id
})
)
)
const afterAcquire = await databases[0]!.query(
`SELECT reserved_requests FROM relay_cells WHERE cell_id = ?`,
[cellC.id]
)
// 6 pending control grants + 6 splices at 2 units each.
expect(Number(afterAcquire[0]!.reserved_requests)).toBe(6 + 12)
await Promise.all(
hosts.map((host, index) =>
stores[index % stores.length]!.releaseActivity(host, `splice:percell-${index}`)
)
)
const afterRelease = await databases[0]!.query(
`SELECT reserved_requests FROM relay_cells WHERE cell_id = ?`,
[cellC.id]
)
expect(Number(afterRelease[0]!.reserved_requests)).toBe(6)
await store.setCellEnabled(cellA.id, true)
await store.setCellEnabled(cellB.id, true)
}, 15_000)
})
@@ -0,0 +1,260 @@
import { afterAll, beforeAll, describe, expect, it } from 'vitest'
import { RelayAssignmentStore } from './assignment-store.js'
import { openRelayDatabase, type RelayDatabase } from './database.js'
const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL
const describePostgres = databaseUrl ? describe : describe.skip
// Three cells: the inventory lock covers more than the rows a move touches, and
// a high-to-low move exposes any lock taken out of cell_id order.
const cells = [
{
id: 'rebind-inventory-postgres-a',
url: 'https://rebind-inventory-postgres-a.example.com',
capacityRequests: 1_000,
connectionHardCap: 600 as const,
connectionUnobservedBound: 50
},
{
id: 'rebind-inventory-postgres-b',
url: 'https://rebind-inventory-postgres-b.example.com',
capacityRequests: 1_000,
connectionHardCap: 600 as const,
connectionUnobservedBound: 50
},
{
id: 'rebind-inventory-postgres-c',
url: 'https://rebind-inventory-postgres-c.example.com',
capacityRequests: 1_000,
connectionHardCap: 600 as const,
connectionUnobservedBound: 50
}
]
const identity = { userId: 'rebind-inventory-postgres-user', relayHostId: 'rebindinvhost001' }
function heartbeat(cell: (typeof cells)[number]) {
return {
cellId: cell.id,
cellUrl: cell.url,
cellIncarnation: '11111111-1111-4111-8111-111111111111',
startedAt: 50,
ready: true,
observedRequests: 0,
totalConnections: 0,
inFlightConnections: 0,
reservedConnectionUnits: 0,
enforcedConnectionUnits: 0,
connectionInclusionWatermark: 1,
connectionHardCap: 600 as const,
connectionUnobservedBound: 50
}
}
// Why: every desktop control rebind used to take the fleet-wide relay_cells
// FOR UPDATE lock, so a rebind on one cell queued behind whatever held any
// other cell's row, until COMMIT (55P03 at the request bound). A rebind only
// touches its own cell row, so it must proceed while another cell's row is
// held elsewhere.
describePostgres('PostgreSQL control rebind under a held cell row', () => {
const databases: RelayDatabase[] = []
beforeAll(async () => {
databases.push(
await openRelayDatabase({ databaseUrl, dataDir: '' }),
await openRelayDatabase({ databaseUrl, dataDir: '' })
)
})
async function removeTestRows(database: RelayDatabase): Promise<void> {
await database.query(
`DELETE FROM relay_control_connection_reservations WHERE user_id = ?`,
[identity.userId]
)
for (const table of [
'relay_assignment_activity_leases',
'relay_post_drain_migration_pins',
'relay_assignment_migration_incarnations',
'relay_assignment_migrations',
'relay_assignments'
]) {
await database.query(`DELETE FROM ${table} WHERE user_id = ?`, [identity.userId])
}
for (const cell of cells) {
for (const table of [
'relay_cell_connection_snapshots',
'relay_cell_connection_runtime',
'relay_cell_connection_limits',
'relay_cell_runtime',
'relay_cells'
]) {
await database.query(`DELETE FROM ${table} WHERE cell_id = ?`, [cell.id])
}
}
}
afterAll(async () => {
if (databases[0]) await removeTestRows(databases[0])
for (const connection of databases) await connection.close()
})
it("rebinds and supersedes a control while another cell's row is held", async () => {
// A prior aborted run leaves connection snapshots that reject a replayed watermark.
await removeTestRows(databases[0]!)
const store = new RelayAssignmentStore(databases[0]!, () => 100)
await store.reconcileCells(cells)
for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell))
// Pin the host to cell A so placement is deterministic.
await store.setCellEnabled(cells[1]!.id, false)
await store.setCellEnabled(cells[2]!.id, false)
const assignment = await store.assign(identity)
expect(assignment.cellId).toBe(cells[0]!.id)
await store.setCellEnabled(cells[1]!.id, true)
await store.setCellEnabled(cells[2]!.id, true)
await store.activateControl(identity, {
cellId: cells[0]!.id,
assignmentEpoch: assignment.assignmentEpoch,
generation: 1,
connectionInclusionWatermark: 10
})
// Hold only cell B's row on a second connection, the way a rebind on B
// does, for longer than the request-path lock bound.
let releaseInventory!: () => void
const inventoryReleased = new Promise<void>((resolve) => {
releaseInventory = resolve
})
let inventoryHeld!: () => void
const inventoryHeldPromise = new Promise<void>((resolve) => {
inventoryHeld = resolve
})
const holder = databases[1]!.transaction(async (transaction) => {
await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cells[1]!.id])
inventoryHeld()
await inventoryReleased
})
await inventoryHeldPromise
// A generation-2 rebind on cell A supersedes generation 1. It must not
// wait on cell B's row.
const startedAt = Date.now()
const blockedStatement = async (): Promise<string> => {
const rows = await databases[1]!.query(
`SELECT left(query, 160) AS q FROM pg_stat_activity
WHERE datname = current_database() AND wait_event_type = 'Lock'`
)
return rows.map((row) => String(row.q)).join(' | ')
}
const timeout = new Promise<never>((_, reject) =>
setTimeout(
() =>
void blockedStatement().then((statement) =>
reject(new Error(`rebind on cell A blocked behind cell B's row: ${statement}`))
),
2_000
)
)
const rebound = await Promise.race([
store.activateControl(identity, {
cellId: cells[0]!.id,
assignmentEpoch: assignment.assignmentEpoch,
generation: 2,
connectionInclusionWatermark: 11
}),
timeout
])
const elapsedMs = Date.now() - startedAt
releaseInventory()
await holder
expect(rebound).toBe(`control:${cells[0]!.id}:2`)
expect(elapsedMs).toBeLessThan(2_000)
const controls = await databases[0]!.query(
`SELECT activity_id FROM relay_assignment_activity_leases
WHERE user_id = ? AND activity_kind = 'control' ORDER BY activity_id`,
[identity.userId]
)
expect(controls).toEqual([{ activity_id: `control:${cells[0]!.id}:2` }])
const reserved = await databases[0]!.query(
`SELECT reserved_requests FROM relay_cells WHERE cell_id = ?`,
[cells[0]!.id]
)
expect(Number(reserved[0]!.reserved_requests)).toBe(1)
}, 15_000)
// Why: a phone's activity id is client-chosen and can follow the host across
// a migration, so acquireActivity may touch two cell rows. Moving from the
// higher cell to the lower one is where an unordered lock cycles with
// placement's ascending inventory lock (reproduced live before this fix).
it('moves an activity from a higher cell to a lower one in cell_id order', async () => {
await removeTestRows(databases[0]!)
const [cellA, cellB, cellC] = cells as [typeof cells[0], typeof cells[0], typeof cells[0]]
const store = new RelayAssignmentStore(databases[0]!, () => 100)
await store.reconcileCells(cells)
for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell))
await store.setCellEnabled(cellA.id, false)
await store.setCellEnabled(cellB.id, false)
const assignment = await store.assign(identity)
expect(assignment.cellId).toBe(cellC.id)
await store.setCellEnabled(cellA.id, true)
await store.setCellEnabled(cellB.id, true)
const activityId = 'splice:rebind-inventory-postgres'
await store.acquireActivity(identity, { activityId, kind: 'splice', cellId: cellC.id })
// The migration makes B authoritative; the lease still sits on C.
const migration = await store.startEvacuation(identity, cellB.id)
expect(migration.targetCellId).toBe(cellB.id)
// Hold B elsewhere. An ordered move locks B first and queues here holding
// nothing else. Locking C first (the old lease's row, as an unordered move
// does) or the whole inventory (which takes A) shows up as a held row.
let releaseRow!: () => void
const rowReleased = new Promise<void>((resolve) => {
releaseRow = resolve
})
let rowHeld!: () => void
const rowHeldPromise = new Promise<void>((resolve) => {
rowHeld = resolve
})
const heldWhileMoverWaits: string[] = []
const holder = databases[1]!.transaction(async (transaction) => {
await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cellB.id])
rowHeld()
await rowReleased
for (const cell of [cellA, cellC]) {
try {
await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cell.id], {
failIfUnavailable: true
})
} catch {
heldWhileMoverWaits.push(cell.id)
}
}
})
await rowHeldPromise
const move = store.acquireActivity(identity, { activityId, kind: 'splice', cellId: cellB.id })
let moved = false
void move.then(() => {
moved = true
})
await new Promise((resolve) => setTimeout(resolve, 250))
expect(moved).toBe(false)
releaseRow()
await holder
await move
expect(heldWhileMoverWaits).toEqual([])
const reservations = await databases[0]!.query(
`SELECT cell_id, reserved_requests FROM relay_cells
WHERE cell_id IN (?, ?, ?) ORDER BY cell_id ASC`,
[cellA.id, cellB.id, cellC.id]
)
const reserved = reservations.map((row) => [String(row.cell_id), Number(row.reserved_requests)])
expect(reserved).toEqual([
[cellA.id, 0],
// Migration grant plus the moved splice, as in the SQLite origin-scoped
// reservation case: the lock change did not alter accounting.
[cellB.id, 6],
// The sticky grant stays on the source until the migration completes.
[cellC.id, 1]
])
}, 15_000)
})
@@ -2,7 +2,10 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
const fakes = vi.hoisted(() => ({
configs: [] as Array<Record<string, unknown>>,
query: vi.fn(async () => ({ rows: [], rowCount: 0 })),
// Pool construction and pool shutdown interleaved, so "the schema pool is
// gone before the serving pool opens" is checkable rather than assumed.
lifecycle: [] as string[],
query: vi.fn(async (_sql: string) => ({ rows: [], rowCount: 0 })),
release: vi.fn(),
end: vi.fn(async () => undefined)
}))
@@ -13,20 +16,37 @@ vi.mock('pg', () => ({
totalCount = 1
idleCount = 1
waitingCount = 0
end = fakes.end
on = vi.fn()
connect = vi.fn(async () => ({ query: fakes.query, release: fakes.release }))
private readonly label: string
constructor(config: Record<string, unknown>) {
fakes.configs.push(config)
this.label = `max=${String(config.max)} statement_timeout=${String(config.statement_timeout)}`
fakes.lifecycle.push(`open ${this.label}`)
}
async end(): Promise<void> {
fakes.lifecycle.push(`end ${this.label}`)
await fakes.end()
}
}
}
}))
import { openRelayDatabase } from './database.js'
import { openRelayDatabase, relayPostgresStatementTimeoutMs } from './database.js'
import { applyPostgresSchema } from './postgres-schema-startup.js'
const SCHEMA_POOL = {
max: 1,
application_name: 'orca-relay/director/director/schema',
connectionTimeoutMillis: 2_000,
// Why: DDL must not inherit the request deadline.
statement_timeout: 0,
lock_timeout: 1_000,
idle_in_transaction_session_timeout: 5_000
}
afterEach(() => {
vi.restoreAllMocks()
})
@@ -34,9 +54,11 @@ afterEach(() => {
describe('PostgreSQL relay deadlines', () => {
beforeEach(() => {
fakes.configs.length = 0
fakes.lifecycle.length = 0
fakes.query.mockClear()
fakes.release.mockClear()
fakes.end.mockClear()
delete process.env.ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS
})
it('bounds pool acquisition, statements, locks, and abandoned transactions', async () => {
@@ -48,6 +70,7 @@ describe('PostgreSQL relay deadlines', () => {
})
expect(fakes.configs).toEqual([
expect.objectContaining(SCHEMA_POOL),
expect.objectContaining({
max: 3,
application_name: 'orca-relay/director/director',
@@ -59,6 +82,110 @@ describe('PostgreSQL relay deadlines', () => {
])
await database.close()
})
// Why: an untimed session left open would be a standing way for request work
// to escape the deadline this whole pool config exists to enforce.
it('closes the untimed schema pool before the serving pool opens', async () => {
const database = await openRelayDatabase({
databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay',
dataDir: './unused',
poolMax: 3,
applicationName: 'orca-relay/director/director'
})
expect(fakes.lifecycle).toEqual([
'open max=1 statement_timeout=0',
'end max=1 statement_timeout=0',
'open max=3 statement_timeout=5000'
])
await database.close()
})
it('applies the schema on the untimed pool, never on the serving one', async () => {
fakes.query.mockClear()
const ddl: string[] = []
fakes.query.mockImplementation(async (sql: string) => {
// Every statement issued before the serving pool exists is schema work.
if (fakes.lifecycle.length === 1) ddl.push(sql)
return { rows: [], rowCount: 0 }
})
const database = await openRelayDatabase({
databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay',
dataDir: './unused'
})
expect(ddl.length).toBeGreaterThan(0)
// Statements can open with a leading `--` rationale comment.
const body = (statement: string): string =>
statement.replace(/^(?:\s*--[^\n]*\n)*\s*/, '')
expect(ddl.every((statement) => /^CREATE\b/i.test(body(statement)))).toBe(true)
// The backfill is DML, so it stays on the deadline-bearing serving pool.
expect(ddl.some((statement) => statement.includes('INSERT INTO'))).toBe(false)
await database.close()
})
it('takes the serving statement deadline from the environment', async () => {
process.env.ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS = '2500'
const database = await openRelayDatabase({
databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay',
dataDir: './unused'
})
expect(fakes.configs).toEqual([
expect.objectContaining({ statement_timeout: 0 }),
expect.objectContaining({ statement_timeout: 2_500 })
])
await database.close()
})
it.each(['0', '-1', '2.5', 'soon', ' '])(
'refuses %s as a statement deadline instead of running unbounded',
(value) => {
expect(() =>
relayPostgresStatementTimeoutMs({ ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS: value })
).toThrow('invalid_statement_timeout')
}
)
it.each([undefined, ''])('defaults to 5s when the environment says %s', (value) => {
expect(
relayPostgresStatementTimeoutMs(
value === undefined ? {} : { ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS: value }
)
).toBe(5_000)
})
// Why: a statement deadline that reaches the caller as a crash converts a
// transient stall into a failed assignment. It aborts the transaction exactly
// as a lock timeout does, so it belongs on the same bounded retry.
it('retries a statement timeout on a fresh client', async () => {
vi.spyOn(console, 'warn').mockImplementation(() => undefined)
const database = await openRelayDatabase({
databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay',
dataDir: './unused'
})
let attempts = 0
const result = await database.transaction(async (transaction) => {
attempts += 1
if (attempts === 1) {
await transaction.query('SELECT 1')
throw Object.assign(new Error('canceling statement due to statement timeout'), {
code: '57014'
})
}
return 'committed'
})
expect(result).toBe('committed')
expect(attempts).toBe(2)
expect(console.warn).toHaveBeenCalledWith(
expect.stringContaining('"event":"orca_relay_postgres_transaction_retry"')
)
expect(console.warn).toHaveBeenCalledWith(expect.stringContaining('"code":"57014"'))
await database.close()
})
})
describe('PostgreSQL schema startup', () => {
@@ -0,0 +1,98 @@
import { afterAll, beforeAll, describe, expect, it } from 'vitest'
import { openRelayDatabase, type RelayDatabase } from './database.js'
const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL
const describePostgres = databaseUrl ? describe : describe.skip
const applicationName = 'orca-relay/statement-timeout-postgres'
describePostgres('PostgreSQL statement deadline', () => {
const databases: RelayDatabase[] = []
beforeAll(async () => {
databases.push(await openRelayDatabase({ databaseUrl, dataDir: '' }))
})
afterAll(async () => {
for (const database of databases) await database.close()
})
it('serves requests under the configured deadline', async () => {
const database = await openRelayDatabase({ databaseUrl, dataDir: '', statementTimeoutMs: 300 })
databases.push(database)
expect(await database.query(`SELECT current_setting('statement_timeout') AS statement_timeout`)).toEqual([
{ statement_timeout: '300ms' }
])
})
// Why: a real 57014 aborts the transaction exactly as a lock timeout does. If
// it escapes the bounded retry it becomes a failed assignment instead of a
// slow one.
it('retries a real statement timeout on a fresh client', async () => {
const database = await openRelayDatabase({ databaseUrl, dataDir: '', statementTimeoutMs: 300 })
databases.push(database)
let attempts = 0
const result = await database.transaction(async (transaction) => {
attempts += 1
if (attempts === 1) await transaction.query(`SELECT pg_sleep(2)`)
return attempts
})
expect(result).toBe(2)
}, 15_000)
// Why: DDL runs on its own untimed connection. relay_invites carries a
// CREATE INDEX IF NOT EXISTS, which (unlike CREATE TABLE IF NOT EXISTS)
// really does queue behind an ACCESS EXCLUSIVE lock on the table.
it('applies the schema behind a held ACCESS EXCLUSIVE lock', async () => {
let releaseTable!: () => void
const tableReleased = new Promise<void>((resolve) => {
releaseTable = resolve
})
let tableHeld!: () => void
const tableHeldPromise = new Promise<void>((resolve) => {
tableHeld = resolve
})
const holder = databases[0]!.transaction(async (transaction) => {
await transaction.query(`LOCK TABLE relay_invites IN ACCESS EXCLUSIVE MODE`)
tableHeld()
await tableReleased
})
await tableHeldPromise
const opening = openRelayDatabase({
databaseUrl,
dataDir: '',
applicationName,
// Far too short for a blocked DDL; the serving pool wears it, the schema
// connection must not.
statementTimeoutMs: 200
})
const blockedOnSchemaConnection = async (): Promise<boolean> => {
const deadline = Date.now() + 4_000
while (Date.now() < deadline) {
const rows = await databases[0]!.query(
`SELECT count(*) AS waiting FROM pg_stat_activity
WHERE datname = current_database() AND wait_event_type = 'Lock'
AND application_name = ?`,
[`${applicationName}/schema`]
)
if (Number(rows[0]!.waiting) > 0) return true
await new Promise((resolve) => setTimeout(resolve, 10))
}
return false
}
const blocked = await blockedOnSchemaConnection()
releaseTable()
await holder
const database = await opening
databases.push(database)
expect(blocked).toBe(true)
// The serving pool still carries the short deadline it was opened with.
expect(await database.query(`SELECT current_setting('statement_timeout') AS statement_timeout`)).toEqual([
{ statement_timeout: '200ms' }
])
}, 15_000)
})
+58 -10
View File
@@ -796,12 +796,34 @@ class PostgresTransaction implements RelayDatabase {
const POSTGRES_TRANSACTION_ATTEMPTS = 3
const POSTGRES_RETRY_MAX_DELAY_MS = 25
const POSTGRES_CONNECTION_TIMEOUT_MS = 2_000
const POSTGRES_STATEMENT_TIMEOUT_MS = 5_000
// Derivation: a control renewal must land inside its own 30s tick
// (RELAY_PROTOCOL_LIMITS.controlPingIntervalMs * 2), and a transaction gets
// POSTGRES_TRANSACTION_ATTEMPTS tries, so the worst case a renewal can spend in
// Postgres is attempts * timeout. 5s keeps that at 15s, half the tick, and still
// leaves room for the connect timeout above.
export const POSTGRES_STATEMENT_TIMEOUT_MS = 5_000
const POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS = 5_000
export function relayPostgresStatementTimeoutMs(
env: NodeJS.ProcessEnv = process.env
): number {
const configured = env.ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS
if (configured === undefined || configured === '') return POSTGRES_STATEMENT_TIMEOUT_MS
const milliseconds = Number(configured)
// 0 is PostgreSQL's "no timeout"; refusing it keeps the deadline this exists
// to enforce from being disabled by a typo in an environment variable.
if (!Number.isInteger(milliseconds) || milliseconds < 1) {
throw new Error('invalid_statement_timeout')
}
return milliseconds
}
function retryablePostgresTransactionError(error: unknown): boolean {
const code = String((error as { code?: unknown }).code)
return code === '40P01' || code === '40001' || code === '55P03'
// 57014 is the pool statement_timeout firing. It aborts the transaction the
// same way a lock timeout does, so it belongs on the bounded retry path
// rather than surfacing as a terminal failure to the caller.
return code === '40P01' || code === '40001' || code === '55P03' || code === '57014'
}
export function isRelayDatabaseTransientError(error: unknown): boolean {
@@ -963,11 +985,36 @@ async function applySchema(database: RelayDatabase): Promise<void> {
}
}
async function applySchemaWithPostgresRetries(database: RelayDatabase): Promise<void> {
await applyPostgresSchema(
SCHEMA.split(';').filter((statement) => statement.trim()),
async (statement) => await database.query(statement)
)
// Why: DDL is not a request. A CREATE INDEX on a grown table legitimately runs
// longer than the request statement_timeout, and inheriting that timeout would
// make every startup fail at the same statement instead of finishing once. One
// short-lived connection of its own, ended before the serving pool opens, keeps
// the untimed session off the request path entirely.
async function applySchemaOnUntimedPool(
databaseUrl: string,
applicationName: string | undefined
): Promise<void> {
const pool = new pg.Pool({
connectionString: databaseUrl,
max: 1,
application_name: applicationName ? `${applicationName}/schema` : undefined,
connectionTimeoutMillis: POSTGRES_CONNECTION_TIMEOUT_MS,
statement_timeout: 0,
// Kept: a DDL blocked behind another director's ACCESS EXCLUSIVE lock must
// yield to the bounded schema retry instead of holding the connection.
lock_timeout: POSTGRES_LOCK_TIMEOUT_MS,
idle_in_transaction_session_timeout: POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS
})
absorbPostgresIdleClientErrors(pool)
const database = new PostgresDatabase(pool)
try {
await applyPostgresSchema(
SCHEMA.split(';').filter((statement) => statement.trim()),
async (statement) => await database.query(statement)
)
} finally {
await database.close().catch(() => undefined)
}
}
async function backfillRelayCellRegions(database: RelayDatabase): Promise<void> {
@@ -983,15 +1030,17 @@ export async function openRelayDatabase(input: {
dataDir: string
poolMax?: number
applicationName?: string
statementTimeoutMs?: number
}): Promise<RelayDatabase> {
let database: RelayDatabase
if (input.databaseUrl) {
await applySchemaOnUntimedPool(input.databaseUrl, input.applicationName)
const pool = new pg.Pool({
connectionString: input.databaseUrl,
max: input.poolMax ?? 10,
application_name: input.applicationName,
connectionTimeoutMillis: POSTGRES_CONNECTION_TIMEOUT_MS,
statement_timeout: POSTGRES_STATEMENT_TIMEOUT_MS,
statement_timeout: input.statementTimeoutMs ?? relayPostgresStatementTimeoutMs(),
lock_timeout: POSTGRES_LOCK_TIMEOUT_MS,
idle_in_transaction_session_timeout: POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS
})
@@ -1004,8 +1053,7 @@ export async function openRelayDatabase(input: {
database = new SqliteDatabase(sqlite)
}
try {
if (input.databaseUrl) await applySchemaWithPostgresRetries(database)
else await applySchema(database)
if (!input.databaseUrl) await applySchema(database)
await backfillRelayCellRegions(database)
return database
} catch (error) {
@@ -0,0 +1,82 @@
import { ASSIGNMENT_LIMITS, RELAY_HOST_CLOSE_REASON } from '@orca-cloud/relay-contract'
import { describe, expect, it } from 'vitest'
import { HostCloseReasonMemory } from './host-close-reason-memory.js'
function memoryAt(clock: { now: number }): HostCloseReasonMemory {
return new HostCloseReasonMemory(() => clock.now)
}
describe('HostCloseReasonMemory', () => {
it('remembers only reasons it knows', () => {
const clock = { now: 1_000 }
const memory = memoryAt(clock)
memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT)
memory.record('b', 'quitting')
memory.record('c', Buffer.alloc(0))
memory.record('d', undefined)
expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT)
expect(memory.read('b')).toBeNull()
expect(memory.read('c')).toBeNull()
expect(memory.read('d')).toBeNull()
})
it('accepts the reason as the Buffer a ws close delivers', () => {
const clock = { now: 1_000 }
const memory = memoryAt(clock)
memory.record('a', Buffer.from(RELAY_HOST_CLOSE_REASON.SIGNED_OUT))
expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT)
})
it('expires an entry once its host may have been rebalanced away', () => {
const clock = { now: 1_000 }
const memory = memoryAt(clock)
memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT)
clock.now += ASSIGNMENT_LIMITS.dormantTtlMs - 1
expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT)
clock.now += 1
expect(memory.read('a')).toBeNull()
expect(memory.size()).toBe(0)
})
it('forgets on demand', () => {
const clock = { now: 1_000 }
const memory = memoryAt(clock)
memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT)
memory.forget('a')
expect(memory.read('a')).toBeNull()
})
it('drops the oldest survivors rather than growing without bound', () => {
const clock = { now: 1_000 }
const memory = memoryAt(clock)
for (let index = 0; index < 50_050; index++) {
memory.record(`host-${index}`, RELAY_HOST_CLOSE_REASON.SIGNED_OUT)
}
expect(memory.size()).toBe(50_000)
expect(memory.read('host-0')).toBeNull()
expect(memory.read('host-50049')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT)
})
it('re-recording refreshes recency so a live host is not evicted first', () => {
const clock = { now: 1_000 }
const memory = memoryAt(clock)
memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT)
memory.record('b', RELAY_HOST_CLOSE_REASON.SIGNED_OUT)
memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT)
expect([...['a', 'b'].map((key) => memory.read(key))]).toEqual([
RELAY_HOST_CLOSE_REASON.SIGNED_OUT,
RELAY_HOST_CLOSE_REASON.SIGNED_OUT
])
expect(memory.size()).toBe(2)
})
})
@@ -0,0 +1,72 @@
import {
ASSIGNMENT_LIMITS,
relayHostCloseReasonFrom,
type RelayHostCloseReason
} from '@orca-cloud/relay-contract'
// Retention matches the dormant assignment TTL: past it the host may have been
// rebalanced onto another cell, so this cell is no longer the one a phone asks.
const RETENTION_MS = ASSIGNMENT_LIMITS.dormantTtlMs
// A fleet-wide auth outage signs out every host at once; the cap bounds that
// burst well above any single cell's host count without becoming a leak.
const MAX_ENTRIES = 50_000
// Why in-memory and not Postgres: a phone reaches the cell its host's assignment
// row already names, which is the same cell that watched the control socket
// close. Losing this on a cell restart degrades to the pre-existing generic
// verdict, so the failure mode is the old behaviour rather than a wrong one.
export class HostCloseReasonMemory {
private readonly entries = new Map<string, { reason: RelayHostCloseReason; expiresAt: number }>()
constructor(private readonly now: () => number = Date.now) {}
// Silently ignores anything that is not a known reason, which is every close
// from a host that predates the field and every abrupt 1006.
record(key: string, reason: unknown): void {
const parsed = relayHostCloseReasonFrom(reason)
if (!parsed) {
return
}
this.entries.delete(key)
this.entries.set(key, { reason: parsed, expiresAt: this.now() + RETENTION_MS })
this.evict()
}
forget(key: string): void {
this.entries.delete(key)
}
read(key: string): RelayHostCloseReason | null {
const entry = this.entries.get(key)
if (!entry) {
return null
}
if (entry.expiresAt <= this.now()) {
this.entries.delete(key)
return null
}
return entry.reason
}
size(): number {
return this.entries.size
}
private evict(): void {
const now = this.now()
for (const [key, entry] of this.entries) {
if (entry.expiresAt > now) {
break
}
this.entries.delete(key)
}
// Insertion order is recency order (record deletes before setting), so the
// head is always the oldest survivor.
for (const key of this.entries.keys()) {
if (this.entries.size <= MAX_ENTRIES) {
break
}
this.entries.delete(key)
}
}
}
@@ -0,0 +1,382 @@
import { EventEmitter } from 'node:events'
import { RELAY_CLOSE_CODE } from '@orca-cloud/relay-contract'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import type WebSocket from 'ws'
import type { RelayAssignmentStore } from './assignment-store.js'
import type { RelayConfig } from './config.js'
import type { CredentialReservation, RelayCredentialStore } from './credential-store.js'
import {
CONTROL_LEASE_JITTER_MS,
CONTROL_LEASE_MS,
HostSessionRegistry
} from './host-session-registry.js'
import type { RelayRuntimeObserver } from './relay-observability.js'
import type { RelayTokenClaims } from './relay-token-verifier.js'
import { ProcessQueuedByteBudget } from './splice-forwarder.js'
// Incident 2026-09-04 ~01:05Z: the phone's dial bound ran out while the cell was
// still inside acceptClient's serialized Postgres phase (cell-inventory lock
// contention). The cell then finished the work for a socket nobody held, holding
// an activity lease for the 10s attach deadline before its timer unwound it, and
// logged `host_data_reservation_already_bound`.
class FakeSocket extends EventEmitter {
readonly OPEN = 1
readonly CLOSING = 2
readonly CLOSED = 3
readyState = this.OPEN
readonly send = vi.fn()
readonly close = vi.fn((code?: number, reason?: string) => {
this.readyState = this.CLOSED
this.emit('close', code, Buffer.from(reason ?? ''))
})
readonly terminate = vi.fn(() => {
this.readyState = this.CLOSED
this.emit('close')
})
}
const config = {
port: 8080,
publicUrl: 'https://relay-c3.example.com',
cellUrl: 'https://relay-c3.example.com',
authIssuer: 'https://auth.example.com',
authAudience: 'orca-relay',
jwksUrl: 'https://auth.example.com/jwks',
assignmentSigningKey: new Uint8Array(32),
role: 'cell',
cellId: 'production-gce-c3',
cells: [{ id: 'production-gce-c3', url: 'https://relay-c3.example.com', capacityRequests: 4_000 }],
adminAudience: 'https://relay-c3.example.com/v1/admin/drain',
deployServiceAccount: 'deploy@example.com',
runtimeServiceAccount: 'runtime@example.com',
adminJwksUrl: 'https://auth.example.com/admin-jwks',
databasePoolMax: 10,
publicAssignmentsEnabled: true,
publicAssignmentConcurrency: 2,
publicAssignmentQueueMax: 128,
publicAssignmentWaitMs: 4_000,
publicResolveConcurrency: 1,
publicResolveWaitMs: 5_000,
publicAssignmentRetryAfterSeconds: 5,
dataDir: './test-data'
} satisfies RelayConfig
const identity = {
sub: 'user-1',
prof: 'profile-1',
relayHostId: 'abcdefghijklmnop',
purpose: 'host-control',
exp: 4_102_444_800
} satisfies RelayTokenClaims
function deferred<T>(): { promise: Promise<T>; resolve: (value: T) => void } {
let resolve!: (value: T) => void
const promise = new Promise<T>((next) => (resolve = next))
return { promise, resolve }
}
const reservation: CredentialReservation = {
userId: identity.sub,
relayHostId: identity.relayHostId,
credentialKind: 'resume',
relayDeviceId: 'device-1',
tokenHash: 'hash',
reservationId: 'reservation-1',
leaseExpiresAt: Date.now() + 60_000,
acceptedCredentialVersion: 2,
acceptedAs: 'current'
}
function harness(options: { random?: () => number; now?: () => number } = {}) {
const acquireActivity = vi.fn().mockResolvedValue(undefined)
const releaseActivity = vi.fn().mockResolvedValue(true)
const assignments = {
activateControl: vi.fn().mockResolvedValue('control:production-gce-c3:1'),
markMigrationTargetRegistered: vi.fn().mockResolvedValue(undefined),
resolve: vi.fn().mockResolvedValue({ cellId: config.cellId }),
acquireActivity,
renewControlActivity: vi.fn().mockResolvedValue(undefined),
releaseActivity
} as unknown as RelayAssignmentStore
const store = {
resolveResume: vi.fn().mockResolvedValue({ userId: identity.sub }),
reserveCredential: vi.fn().mockResolvedValue(reservation),
failReservation: vi.fn().mockResolvedValue(undefined)
}
const observer = {
recordAuth: vi.fn(),
recordForwardedBytes: vi.fn(),
recordHttp: vi.fn(),
recordReconnect: vi.fn(),
recordSql: vi.fn(),
recordClientAcceptAbandoned: vi.fn()
} satisfies RelayRuntimeObserver
const registry = new HostSessionRegistry(
config,
vi.fn(),
store as unknown as RelayCredentialStore,
assignments,
new ProcessQueuedByteBudget(),
observer,
options.now,
options.random
)
const activate = (
registry as unknown as {
activate: (
socket: WebSocket,
identity: RelayTokenClaims,
existing: null,
generation: number,
rebind: boolean,
assignmentEpoch: number,
appVersion: string
) => Promise<void>
}
).activate.bind(registry)
return { registry, store, assignments, acquireActivity, releaseActivity, observer, activate }
}
async function activeHost(h: ReturnType<typeof harness>): Promise<FakeSocket> {
const control = new FakeSocket()
await h.activate(control as unknown as WebSocket, identity, null, 1, false, 1, '1.4.197')
return control
}
describe('client accept abandoned mid-DB-phase', () => {
beforeEach(() => vi.useFakeTimers())
afterEach(() => {
vi.clearAllTimers()
vi.useRealTimers()
})
it('stops after a slow activity acquire when the phone already hung up', async () => {
const h = harness()
const control = await activeHost(h)
const slowAcquire = deferred<void>()
h.acquireActivity.mockReturnValueOnce(slowAcquire.promise)
const capacity = { bind: vi.fn(), release: vi.fn() }
const client = new FakeSocket()
const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined)
try {
const accepting = h.registry.acceptClient(
client as unknown as WebSocket,
identity.relayHostId,
'credential',
capacity
)
await vi.advanceTimersByTimeAsync(0)
expect(h.acquireActivity).toHaveBeenCalledOnce()
// The phone's 12s bound fires while the cell still waits on Postgres.
client.close(1000, 'client bound')
capacity.release()
slowAcquire.resolve()
await accepting
// No conn-open reached the desktop; nothing pending; the lease it just took is
// released instead of leaking to expiry cleanup; bind never throws.
expect(control.send).not.toHaveBeenCalledWith(expect.stringContaining('conn-open'))
expect(capacity.bind).not.toHaveBeenCalled()
const session = h.registry.get({ userId: identity.sub, relayHostId: identity.relayHostId })
expect(session?.pendingConns.size).toBe(0)
expect(h.store.failReservation).toHaveBeenCalledWith(reservation)
expect(h.releaseActivity).toHaveBeenCalledWith(
{ userId: identity.sub, relayHostId: identity.relayHostId },
expect.stringMatching(/^confirmation:/)
)
expect(h.observer.recordClientAcceptAbandoned).toHaveBeenCalledWith(
'activity',
expect.any(Number)
)
const line = warn.mock.calls.map((call) => String(call[0])).find((entry) =>
entry.includes('orca_relay_client_accept_abandoned')
)
expect(line).toBeDefined()
expect(JSON.parse(line!)).toMatchObject({ stage: 'activity' })
expect(line).not.toContain(identity.relayHostId)
} finally {
warn.mockRestore()
h.registry.drain(0)
vi.advanceTimersByTime(0)
}
})
it('stops after a slow credential reservation without acquiring an activity lease', async () => {
const h = harness()
await activeHost(h)
const slowReserve = deferred<CredentialReservation>()
h.store.reserveCredential.mockReturnValueOnce(slowReserve.promise)
const client = new FakeSocket()
const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined)
try {
const accepting = h.registry.acceptClient(
client as unknown as WebSocket,
identity.relayHostId,
'credential'
)
await vi.advanceTimersByTimeAsync(0)
client.close(1000, 'client bound')
slowReserve.resolve(reservation)
await accepting
expect(h.acquireActivity).not.toHaveBeenCalled()
expect(h.store.failReservation).toHaveBeenCalledWith(reservation)
expect(h.observer.recordClientAcceptAbandoned).toHaveBeenCalledWith(
'credential',
expect.any(Number)
)
} finally {
warn.mockRestore()
h.registry.drain(0)
vi.advanceTimersByTime(0)
}
})
it('stops after a slow resume lookup before starting the invite and assignment lookups', async () => {
const h = harness()
await activeHost(h)
const store = h.store as typeof h.store & { resolveInviteForMove: ReturnType<typeof vi.fn> }
store.resolveInviteForMove = vi.fn().mockResolvedValue(null)
const slowResume = deferred<null>()
h.store.resolveResume.mockReturnValueOnce(slowResume.promise)
const resolveAssignment = (h.assignments as unknown as { resolve: ReturnType<typeof vi.fn> })
.resolve
resolveAssignment.mockClear()
const client = new FakeSocket()
const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined)
try {
const accepting = h.registry.acceptClient(
client as unknown as WebSocket,
identity.relayHostId,
'credential'
)
await vi.advanceTimersByTimeAsync(0)
client.close(1000, 'client bound')
slowResume.resolve(null)
await accepting
expect(store.resolveInviteForMove).not.toHaveBeenCalled()
expect(resolveAssignment).not.toHaveBeenCalled()
expect(h.store.reserveCredential).not.toHaveBeenCalled()
expect(h.observer.recordClientAcceptAbandoned).toHaveBeenCalledWith(
'assignment',
expect.any(Number)
)
} finally {
warn.mockRestore()
h.registry.drain(0)
vi.advanceTimersByTime(0)
}
})
it('stops after a slow same-cell assignment resolve, before reserving a credential', async () => {
const h = harness()
await activeHost(h)
const resolveAssignment = (h.assignments as unknown as { resolve: ReturnType<typeof vi.fn> })
.resolve
const slowResolve = deferred<{ cellId: string }>()
resolveAssignment.mockReturnValueOnce(slowResolve.promise)
const client = new FakeSocket()
const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined)
try {
const accepting = h.registry.acceptClient(
client as unknown as WebSocket,
identity.relayHostId,
'credential'
)
await vi.advanceTimersByTimeAsync(0)
client.close(1000, 'client bound')
// A correct, same-cell assignment: only the closed socket stops the accept.
slowResolve.resolve({ cellId: config.cellId })
await accepting
// Proves the accept reached the third guard, not the first.
expect(resolveAssignment).toHaveBeenCalled()
expect(h.store.reserveCredential).not.toHaveBeenCalled()
expect(h.observer.recordClientAcceptAbandoned).toHaveBeenCalledWith(
'assignment',
expect.any(Number)
)
} finally {
warn.mockRestore()
h.registry.drain(0)
vi.advanceTimersByTime(0)
}
})
it('still opens the connection when the phone is holding on', async () => {
const h = harness()
const control = await activeHost(h)
const capacity = { bind: vi.fn(), release: vi.fn() }
const client = new FakeSocket()
await h.registry.acceptClient(
client as unknown as WebSocket,
identity.relayHostId,
'credential',
capacity
)
expect(control.send).toHaveBeenCalledWith(expect.stringContaining('"type":"conn-open"'))
expect(capacity.bind).toHaveBeenCalledOnce()
expect(h.observer.recordClientAcceptAbandoned).not.toHaveBeenCalled()
expect(client.close).not.toHaveBeenCalled()
h.registry.drain(0)
vi.advanceTimersByTime(0)
})
})
describe('control lease jitter', () => {
beforeEach(() => vi.useFakeTimers())
afterEach(() => {
vi.clearAllTimers()
vi.useRealTimers()
})
it('grants a lease uniformly around its mean so cohorts drift apart at the same mean rate', async () => {
const now = 1_700_000_000_000
const helloAck = (socket: FakeSocket) =>
JSON.parse(
String(socket.send.mock.calls.find((call) => String(call[0]).includes('host-hello-ack'))![0])
) as { leaseExpiresAt: number }
const shortest = harness({ now: () => now, random: () => 0 })
const shortestAck = helloAck(await activeHost(shortest))
const centered = harness({ now: () => now, random: () => 0.5 })
const centeredAck = helloAck(await activeHost(centered))
const longestRoll = 0.999999
const longest = harness({ now: () => now, random: () => longestRoll })
const longestAck = helloAck(await activeHost(longest))
// Pinned, not bounded: a jitter clamped to one side still satisfies an upper
// bound, so only the exact top of the band proves it is symmetric.
const longestOffset = Math.floor((longestRoll * 2 - 1) * CONTROL_LEASE_JITTER_MS)
expect(shortestAck.leaseExpiresAt).toBe(now + CONTROL_LEASE_MS - CONTROL_LEASE_JITTER_MS)
expect(centeredAck.leaseExpiresAt).toBe(now + CONTROL_LEASE_MS)
expect(longestAck.leaseExpiresAt).toBe(now + CONTROL_LEASE_MS + longestOffset)
shortest.registry.drain(0)
centered.registry.drain(0)
longest.registry.drain(0)
vi.advanceTimersByTime(0)
})
it('rebinds re-roll the jitter instead of pinning the cohort phase', async () => {
const now = 1_700_000_000_000
let roll = 0
const h = harness({ now: () => now, random: () => roll })
const first = await activeHost(h)
const session = h.registry.get({ userId: identity.sub, relayHostId: identity.relayHostId })!
const firstLease = session.leaseExpiresAt
roll = 0.75
const rebind = new FakeSocket()
await (
h.registry as unknown as {
activate: (...args: unknown[]) => Promise<void>
}
).activate(rebind as unknown as WebSocket, identity, session, 1, true, 1, '1.4.197')
expect(session.leaseExpiresAt).toBe(now + CONTROL_LEASE_MS + CONTROL_LEASE_JITTER_MS / 2)
expect(session.leaseExpiresAt).not.toBe(firstLease)
expect(first.close).toHaveBeenCalledWith(RELAY_CLOSE_CODE.PEER_DROPPED, 'control rebound')
h.registry.drain(0)
vi.advanceTimersByTime(0)
})
})
+89 -13
View File
@@ -14,7 +14,8 @@ import {
HostHelloSchema,
InviteCreateSchema,
RELAY_PROTOCOL_LIMITS,
RELAY_CLOSE_CODE
RELAY_CLOSE_CODE,
type RelayHostCloseReason
} from '@orca-cloud/relay-contract'
import nacl from 'tweetnacl'
import type WebSocket from 'ws'
@@ -25,9 +26,10 @@ import {
RelayCredentialStore,
type CredentialReservation
} from './credential-store.js'
import { HostCloseReasonMemory } from './host-close-reason-memory.js'
import { relayHostLogDigest } from './relay-host-log-digest.js'
import type { RelayTokenClaims } from './relay-token-verifier.js'
import type { RelayRuntimeObserver } from './relay-observability.js'
import type { RelayClientAcceptStage, RelayRuntimeObserver } from './relay-observability.js'
import type { PendingHostDataReservation } from './relay-connection-ledger.js'
import { closeRelayWebSocket } from './relay-websocket-close.js'
import { ProcessQueuedByteBudget, wireSplice } from './splice-forwarder.js'
@@ -127,9 +129,23 @@ function send(socket: WebSocket, type: string, message: object): void {
// stalled predecessor only accumulates doomed sockets.
const ACTIVATION_QUEUE_WAIT_MS = 30_000
// Why: this lease bounds how long a host lingers on a cell after a missed drain,
// and rebinding it is the only passive rebalancing we have, so it has to stay
// finite. 6h keeps both properties while cutting control-activation traffic on
// the contended cell-inventory lock ~6x; the relay JWT (5 min, refreshed by the
// desktop) and the 75s silence watchdog are enforced separately, so a longer
// grant authorizes nothing extra. Symmetric jitter walks same-minute reconnect
// cohorts apart across cycles without changing the mean rebind rate.
export const CONTROL_LEASE_MS = 6 * 60 * 60 * 1000
export const CONTROL_LEASE_JITTER_MS = 30 * 60 * 1000
export class HostSessionRegistry {
private readonly sessions = new Map<string, HostSession>()
private readonly activationQueues = new Map<string, Promise<void>>()
// Why it outlives `sessions`: the orphan grace deletes the session within 30s,
// but a signed-out desktop never comes back, so the phone that asks minutes
// later would otherwise find nothing to explain its rejection with.
private readonly hostCloseReasons = new HostCloseReasonMemory(() => this.now())
private draining = false
constructor(
@@ -139,9 +155,16 @@ export class HostSessionRegistry {
private readonly assignments: RelayAssignmentStore,
private readonly queuedByteBudget: ProcessQueuedByteBudget,
private readonly observer: RelayRuntimeObserver,
private readonly now: () => number = Date.now
private readonly now: () => number = Date.now,
private readonly random: () => number = Math.random
) {}
// Uniform over [CONTROL_LEASE_MS - jitter, CONTROL_LEASE_MS + jitter).
private controlLeaseExpiresAt(): number {
const offset = Math.floor((this.random() * 2 - 1) * CONTROL_LEASE_JITTER_MS)
return this.now() + CONTROL_LEASE_MS + offset
}
async acceptClient(
socket: WebSocket,
hostId: string,
@@ -153,10 +176,31 @@ export class HostSessionRegistry {
this.rejectClient(socket, RELAY_CLOSE_CODE.DRAINING)
return
}
// Why: the accept runs several serialized Postgres calls behind the contended
// cell-inventory lock, and phones bound their dial. Finishing the work for a
// phone that already hung up took an activity lease held for the 10s attach
// deadline, then failed at bind with host_data_reservation_already_bound.
const acceptStartedAt = this.now()
const abandonedByClient = (stage: RelayClientAcceptStage, cleanup?: () => void): boolean => {
if (socket.readyState === socket.OPEN) return false
capacityReservation?.release()
cleanup?.()
const elapsedMs = this.now() - acceptStartedAt
this.observer.recordClientAcceptAbandoned?.(stage, elapsedMs)
console.warn(
JSON.stringify({ event: 'orca_relay_client_accept_abandoned', stage, elapsedMs })
)
return true
}
if (this.config.role === 'cell') {
const outerIdentity =
(await this.store.resolveResume(hostId, credential)) ??
(await this.store.resolveInviteForMove(hostId, credential))
// Each lookup is its own pooled round trip; stop between them once the phone
// has left instead of running the rest of the chain for nobody.
let outerIdentity = await this.store.resolveResume(hostId, credential)
if (abandonedByClient('assignment')) return
if (!outerIdentity) {
outerIdentity = await this.store.resolveInviteForMove(hostId, credential)
if (abandonedByClient('assignment')) return
}
const assignment = outerIdentity
? await this.assignments.resolve({ userId: outerIdentity.userId, relayHostId: hostId })
: null
@@ -166,6 +210,7 @@ export class HostSessionRegistry {
this.rejectClient(socket, RELAY_CLOSE_CODE.WRONG_CELL)
return
}
if (abandonedByClient('assignment')) return
}
const reservation = await this.store.reserveCredential(hostId, credential)
if (!reservation) {
@@ -175,7 +220,9 @@ export class HostSessionRegistry {
return
}
this.observer.recordAuth(true)
const session = this.sessions.get(this.key(reservation.userId, hostId))
if (abandonedByClient('credential', () => this.failReservationBestEffort(reservation))) return
const sessionKey = this.key(reservation.userId, hostId)
const session = this.sessions.get(sessionKey)
if (
!session ||
session.state !== 'active' ||
@@ -184,7 +231,13 @@ export class HostSessionRegistry {
) {
capacityReservation?.release()
await this.store.failReservation(reservation)
this.rejectClient(socket, RELAY_CLOSE_CODE.HOST_OFFLINE)
// The only rejection that can name a cause: the host is genuinely absent.
// The attach-deadline 4404 below fires while control is still connected.
this.rejectClient(
socket,
RELAY_CLOSE_CODE.HOST_OFFLINE,
this.hostCloseReasons.read(sessionKey)
)
return
}
if (session.activeConnIds.size + session.pendingConns.size >= 8) {
@@ -214,6 +267,14 @@ export class HostSessionRegistry {
return
}
}
if (
abandonedByClient('activity', () => {
this.failReservationBestEffort(reservation)
if (credentialActivityId) this.releaseActivityBestEffort(identity, credentialActivityId)
})
) {
return
}
const attachTimer = setTimeout(() => {
session.pendingConns.delete(connId)
capacityReservation?.release()
@@ -727,7 +788,7 @@ export class HostSessionRegistry {
existing.socket = socket
existing.state = existing.regionalDrainAttemptId ? 'drain-only' : 'active'
existing.appVersion = appVersion
existing.leaseExpiresAt = this.now() + 55 * 60 * 1000
existing.leaseExpiresAt = this.controlLeaseExpiresAt()
existing.lastPongAt = this.now()
existing.activityRenewalDueAt =
this.now() + RELAY_PROTOCOL_LIMITS.controlPingIntervalMs
@@ -778,7 +839,7 @@ export class HostSessionRegistry {
appVersion,
state: 'active',
socket,
leaseExpiresAt: this.now() + 55 * 60 * 1000,
leaseExpiresAt: this.controlLeaseExpiresAt(),
orphanTimer: null,
heartbeatTimer: null,
lastPongAt: this.now(),
@@ -793,7 +854,10 @@ export class HostSessionRegistry {
regionalDrainTimer: null,
regionalDrainExpiresAt: null
}
this.sessions.set(this.key(identity.sub, identity.relayHostId), session)
const sessionKey = this.key(identity.sub, identity.relayHostId)
// A host that proved itself again is not signed out, whatever it said last.
this.hostCloseReasons.forget(sessionKey)
this.sessions.set(sessionKey, session)
this.wireActiveControl(session)
this.sendHelloAck(session)
}
@@ -813,6 +877,11 @@ export class HostSessionRegistry {
})
socket.once('close', (code, reason) => {
this.observer.recordControlClose?.(code)
// Guarded on identity: a predecessor retired by a rebind must not stamp a
// cause onto the live session that replaced it.
if (session.socket === socket) {
this.hostCloseReasons.record(this.key(session.identity.sub, session.relayHostId), reason)
}
// One line per control close makes reconnect churners attributable by
// host digest without exposing the raw relay host id.
console.warn(
@@ -1187,9 +1256,16 @@ export class HostSessionRegistry {
if (session.socket) send(session.socket, 'control-error', { ...(reqId ? { reqId } : {}), code })
}
private rejectClient(socket: WebSocket, code: number): void {
// hostCloseReason rides the WebSocket close reason, never relay-hello: every
// shipped phone parses relay-hello with a strict schema that rejects an
// unknown key, and none of them read the close reason at all.
private rejectClient(
socket: WebSocket,
code: number,
hostCloseReason?: RelayHostCloseReason | null
): void {
send(socket, 'relay-hello', { ok: false, code })
closeRelayWebSocket(socket, code, 'relay connection rejected')
closeRelayWebSocket(socket, code, hostCloseReason ?? 'relay connection rejected')
}
private releaseControlActivity(session: HostSession): void {
@@ -0,0 +1,206 @@
import { EventEmitter } from 'node:events'
import {
CONTROL_CONTINUITY_LIMITS,
RELAY_CLOSE_CODE,
RELAY_HOST_CLOSE_REASON
} from '@orca-cloud/relay-contract'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import type WebSocket from 'ws'
import type { RelayAssignmentStore } from './assignment-store.js'
import type { RelayConfig } from './config.js'
import type { RelayCredentialStore } from './credential-store.js'
import { HostSessionRegistry } from './host-session-registry.js'
import type { RelayRuntimeObserver } from './relay-observability.js'
import type { RelayTokenClaims } from './relay-token-verifier.js'
import { ProcessQueuedByteBudget } from './splice-forwarder.js'
class FakeSocket extends EventEmitter {
readonly OPEN = 1
readonly CLOSED = 3
readyState = this.OPEN
readonly send = vi.fn()
readonly close = vi.fn((code?: number, reason?: string) => {
this.readyState = this.CLOSED
this.emit('close', code, Buffer.from(reason ?? ''))
})
readonly terminate = vi.fn(() => {
this.readyState = this.CLOSED
this.emit('close', 1006, Buffer.alloc(0))
})
}
const config = {
port: 8080,
publicUrl: 'https://relay-c3.example.com',
cellUrl: 'https://relay-c3.example.com',
authIssuer: 'https://auth.example.com',
authAudience: 'orca-relay',
jwksUrl: 'https://auth.example.com/jwks',
assignmentSigningKey: new Uint8Array(32),
role: 'cell',
cellId: 'production-gce-c3',
cells: []
} as unknown as RelayConfig
const identity = {
sub: 'user-1',
prof: 'profile-1',
org: 'org-1',
relayHostId: 'AbCdEf0123_-xyZ9'
} as unknown as RelayTokenClaims
const reservation = {
userId: identity.sub,
relayHostId: identity.relayHostId,
credentialKind: 'resume',
relayDeviceId: 'device-1',
leaseExpiresAt: Date.now() + 60_000
}
function createRegistry() {
const store = {
resolveResume: vi.fn().mockResolvedValue({ userId: identity.sub }),
reserveCredential: vi.fn().mockResolvedValue(reservation),
failReservation: vi.fn().mockResolvedValue(undefined)
}
const assignments = {
activateControl: vi.fn().mockResolvedValue('control:production-gce-c3:1'),
markMigrationTargetRegistered: vi.fn().mockResolvedValue(undefined),
resolve: vi.fn().mockResolvedValue({ cellId: config.cellId }),
acquireActivity: vi.fn().mockResolvedValue(undefined),
renewControlActivity: vi.fn().mockResolvedValue(undefined),
releaseActivity: vi.fn().mockResolvedValue(true)
} as unknown as RelayAssignmentStore
const observer = {
recordAuth: vi.fn(),
recordForwardedBytes: vi.fn(),
recordHttp: vi.fn(),
recordReconnect: vi.fn(),
recordSql: vi.fn(),
recordControlClose: vi.fn(),
recordSpliceClose: vi.fn()
} satisfies RelayRuntimeObserver
const registry = new HostSessionRegistry(
config,
vi.fn(),
store as unknown as RelayCredentialStore,
assignments,
new ProcessQueuedByteBudget(),
observer
)
const activate = (socket: WebSocket, generation: number): Promise<void> =>
(
registry as unknown as {
activate: (
socket: WebSocket,
identity: RelayTokenClaims,
existing: null,
generation: number,
rebind: boolean,
assignmentEpoch: number,
appVersion: string
) => Promise<void>
}
).activate(socket, identity, null, generation, false, 1, '1.4.173')
return { registry, activate }
}
async function dialPhone(registry: HostSessionRegistry): Promise<FakeSocket> {
const phone = new FakeSocket()
await registry.acceptClient(phone as unknown as WebSocket, identity.relayHostId, 'credential')
return phone
}
// The 4404 hello body is unchanged: every shipped phone parses it with a strict
// schema, so the cause has to ride the close frame instead.
const HOST_OFFLINE_HELLO = JSON.stringify({
type: 'relay-hello',
ok: false,
code: RELAY_CLOSE_CODE.HOST_OFFLINE
})
describe('host sign-out reason on phone rejection', () => {
beforeEach(() => vi.useFakeTimers())
afterEach(() => {
vi.clearAllTimers()
vi.useRealTimers()
})
it('names the sign-out to a phone that arrives after the host is gone', async () => {
const { registry, activate } = createRegistry()
const control = new FakeSocket()
await activate(control as unknown as WebSocket, 1)
control.close(1000, RELAY_HOST_CLOSE_REASON.SIGNED_OUT)
vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1)
const phone = await dialPhone(registry)
expect(phone.send).toHaveBeenCalledWith(HOST_OFFLINE_HELLO)
expect(phone.close).toHaveBeenCalledWith(
RELAY_CLOSE_CODE.HOST_OFFLINE,
RELAY_HOST_CLOSE_REASON.SIGNED_OUT
)
})
it('says nothing when the host died without naming a cause', async () => {
const { registry, activate } = createRegistry()
const control = new FakeSocket()
await activate(control as unknown as WebSocket, 1)
control.terminate()
vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1)
const phone = await dialPhone(registry)
expect(phone.close).toHaveBeenCalledWith(
RELAY_CLOSE_CODE.HOST_OFFLINE,
'relay connection rejected'
)
})
it('ignores a close reason the host invented', async () => {
const { registry, activate } = createRegistry()
const control = new FakeSocket()
await activate(control as unknown as WebSocket, 1)
control.close(1000, 'signed-out-ish')
vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1)
const phone = await dialPhone(registry)
expect(phone.close).toHaveBeenCalledWith(
RELAY_CLOSE_CODE.HOST_OFFLINE,
'relay connection rejected'
)
})
it('forgets the sign-out once the host proves itself again', async () => {
const { registry, activate } = createRegistry()
const control = new FakeSocket()
await activate(control as unknown as WebSocket, 1)
control.close(1000, RELAY_HOST_CLOSE_REASON.SIGNED_OUT)
vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1)
const reconnected = new FakeSocket()
await activate(reconnected as unknown as WebSocket, 2)
// Drop it abruptly, as a network death would, so only the stale memory
// could still name a cause.
reconnected.terminate()
vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1)
const phone = await dialPhone(registry)
expect(phone.close).toHaveBeenCalledWith(
RELAY_CLOSE_CODE.HOST_OFFLINE,
'relay connection rejected'
)
})
// A live host is present: the 4404 there is an attach deadline, not absence.
it('never names a cause while the host control is connected', async () => {
const { registry, activate } = createRegistry()
const control = new FakeSocket()
await activate(control as unknown as WebSocket, 1)
const phone = await dialPhone(registry)
expect(phone.close).not.toHaveBeenCalled()
expect(control.send).toHaveBeenCalledWith(expect.stringContaining('"type":"conn-open"'))
})
})
@@ -503,12 +503,19 @@ describePostgres('PostgreSQL transaction recovery', () => {
const directorLockOrder: string[] = []
const assignmentDatabase = new TransactionProbeDatabase(database, async (phase, sql) => {
if (phase === 'before') {
if (sql.includes('FROM relay_assignments WHERE user_id = ?')) {
// Only locked statements reach this hook, so classifying the pin read
// is what proves it stays unlocked: if it ever grows a FOR UPDATE it
// shows up in the order below instead of silently joining the queue.
if (sql.includes('SELECT cell_id FROM relay_assignments')) {
directorLockOrder.push('pin-read')
} else if (sql.includes('FROM relay_assignments WHERE user_id = ?')) {
directorLockOrder.push('assignment')
} else if (sql.includes('FROM relay_assignment_activity_leases')) {
directorLockOrder.push('activity')
} else if (sql.includes('FROM relay_cells ORDER BY')) {
directorLockOrder.push('cell-inventory')
} else if (sql.includes('FROM relay_cells WHERE cell_id IN')) {
directorLockOrder.push('cell-rows')
} else if (sql.includes('FROM relay_cells WHERE cell_id = ?')) {
directorLockOrder.push('cell')
}
@@ -530,11 +537,14 @@ describePostgres('PostgreSQL transaction recovery', () => {
})
await expect(legacyTransaction).resolves.toBeUndefined()
expect(assignmentDatabase.attempts).toBe(2)
// The retry still takes a cell row before the assignment row — the order
// that avoids the legacy cycle — but only the pinned row, never the
// inventory.
expect(directorLockOrder).toEqual([
'assignment',
'activity',
'cell',
'cell-inventory',
'cell-rows',
'assignment',
'activity'
])
@@ -195,16 +195,23 @@ describe('relay observability', () => {
observability.recordControlClose(4402)
observability.recordSpliceClose('host-oversize-frame')
observability.recordSpliceClose('queue-limit')
observability.recordClientAcceptAbandoned('activity', 14_250.4)
observability.recordClientAcceptAbandoned('activity', 2_000)
observability.recordClientAcceptAbandoned('credential', 3_000)
observability.flush(counts)
observability.flush(counts)
expect(entries[0]).toMatchObject({
controlClosesByCodeDelta: { 1006: 2, 4402: 1 },
spliceClosesByTriggerDelta: { 'host-oversize-frame': 1, 'queue-limit': 1 }
spliceClosesByTriggerDelta: { 'host-oversize-frame': 1, 'queue-limit': 1 },
clientAcceptsAbandonedByStageDelta: { activity: 2, credential: 1 },
clientAcceptAbandonedMsMax: 14_250.4
})
expect(entries[1]).toMatchObject({
controlClosesByCodeDelta: {},
spliceClosesByTriggerDelta: {}
spliceClosesByTriggerDelta: {},
clientAcceptsAbandonedByStageDelta: {},
clientAcceptAbandonedMsMax: 0
})
})
@@ -64,8 +64,12 @@ export interface RelayRuntimeObserver {
}): void
recordControlClose?(code: number): void
recordSpliceClose?(trigger: string): void
recordClientAcceptAbandoned?(stage: RelayClientAcceptStage, elapsedMs: number): void
}
// Which serialized accept step the phone had already hung up behind.
export type RelayClientAcceptStage = 'assignment' | 'credential' | 'activity'
type RelayMetricDeltas = {
forwardedBytes: number
authSuccesses: number
@@ -87,6 +91,8 @@ type RelayMetricDeltas = {
unavailableRegions: Record<string, number>
controlClosesByCode: Record<string, number>
spliceClosesByTrigger: Record<string, number>
clientAcceptsAbandonedByStage: Record<string, number>
clientAcceptAbandonedMsMax: number
controlRenewalLatenciesMs: number[]
controlRenewalsByOutcome: Record<string, number>
controlActivityRecoveries: number
@@ -116,6 +122,8 @@ const emptyDeltas = (): RelayMetricDeltas => ({
unavailableRegions: {},
controlClosesByCode: {},
spliceClosesByTrigger: {},
clientAcceptsAbandonedByStage: {},
clientAcceptAbandonedMsMax: 0,
controlRenewalLatenciesMs: [],
controlRenewalsByOutcome: {},
controlActivityRecoveries: 0,
@@ -228,6 +236,14 @@ export class RelayObservability implements RelayRuntimeObserver {
(this.deltas.spliceClosesByTrigger[trigger] ?? 0) + 1
}
recordClientAcceptAbandoned(stage: RelayClientAcceptStage, elapsedMs: number): void {
increment(this.deltas.clientAcceptsAbandonedByStage, stage)
this.deltas.clientAcceptAbandonedMsMax = Math.max(
this.deltas.clientAcceptAbandonedMsMax,
elapsedMs
)
}
start(readCounts: () => RelayProcessCounts, intervalMs = 30_000): void {
if (this.timer) return
this.eventLoop.enable()
@@ -289,6 +305,8 @@ export class RelayObservability implements RelayRuntimeObserver {
unavailableRegionsDelta: deltas.unavailableRegions,
controlClosesByCodeDelta: deltas.controlClosesByCode,
spliceClosesByTriggerDelta: deltas.spliceClosesByTrigger,
clientAcceptsAbandonedByStageDelta: deltas.clientAcceptsAbandonedByStage,
clientAcceptAbandonedMsMax: Number(deltas.clientAcceptAbandonedMsMax.toFixed(3)),
sqlQueriesDelta: deltas.sqlQueries,
sqlFailuresDelta: deltas.sqlFailures,
sqlLatencyMsMax: Number(deltas.sqlLatencyMsMax.toFixed(3)),
+3 -1
View File
@@ -88,6 +88,7 @@ export function createRelayServer(
database: RelayDatabase,
options: {
now?: () => number
random?: () => number
connectionLedgerLimits?: { hardCap: number; controlReserve: number }
cellIncarnation?: string
} = {}
@@ -123,7 +124,8 @@ export function createRelayServer(
assignments,
queuedBytes,
observability,
options.now
options.now,
options.random
)
const app = createRelayApp(config, {
store,
@@ -133,10 +133,15 @@
"google_logging_metric.relay_snapshot",
"google_monitoring_alert_policy.relay_assignment_5xx",
"google_monitoring_alert_policy.relay_assignment_edge_429",
"google_monitoring_alert_policy.relay_cell_process_exit",
"google_monitoring_alert_policy.relay_cloud_nat_port_drops",
"google_monitoring_alert_policy.relay_cloud_sql_backends",
"google_monitoring_alert_policy.relay_cloud_sql_checkpoint_loop",
"google_monitoring_alert_policy.relay_cloud_sql_disk",
"google_monitoring_alert_policy.relay_custom",
"google_monitoring_alert_policy.relay_gce_connection_headroom",
"google_monitoring_alert_policy.relay_postgres_retry_exhausted",
"google_monitoring_dashboard.relay_incident",
"google_project_iam_custom_role.github_production_relay_capacity_mutation",
"google_project_iam_custom_role.github_relay_asia_topology_mutation",
"google_project_iam_custom_role.github_relay_asia_topology_read",
@@ -1,4 +1,5 @@
import { pathToFileURL } from 'node:url'
import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs'
import { inspectAdmissionSelector } from './relay-admission-selector.mjs'
const DIRECTOR_ORIGIN = 'https://relay.onorca.dev'
@@ -229,15 +230,20 @@ export async function recoverRegionalRehomeEnable(config, post) {
export async function operateRegionalRehome(config, dependencies = {}) {
const fetchImpl = dependencies.fetch ?? fetch
const post = dependencies.post ?? (async (path, body) => await responseJson(
await fetchImpl(`${config.directorOrigin}${path}`, {
method: 'POST',
headers: {
authorization: `Bearer ${config.token}`,
'content-type': 'application/json'
// Generation-guarded writes make a retry a no-op or an explicit mismatch, never a double apply.
await fetchAdminOnceMore(
fetchImpl,
`${config.directorOrigin}${path}`,
{
method: 'POST',
headers: {
authorization: `Bearer ${config.token}`,
'content-type': 'application/json'
},
body: JSON.stringify(body)
},
body: JSON.stringify(body),
signal: AbortSignal.timeout(30_000)
}),
{ wait: dependencies.wait }
),
path
))
if (config.mode === 'recover-enable') {
@@ -263,3 +263,55 @@ test('main executes recovery mode and emits verified disabled control', async ()
control: control(6, false)
})
})
test('retries a transient 503 on the director control endpoint', async () => {
const config = parseRegionalRehomeArguments(
argumentsFor('inspect'),
{ ORCA_RELAY_ADMIN_ID_TOKEN: 'token' }
)
const paths = []
let selectorCalls = 0
const result = await operateRegionalRehome(config, {
wait: async () => {},
fetch: async (url) => {
const path = new URL(url).pathname
paths.push(path)
if (path === '/v1/admin/admission-selector/status') {
selectorCalls += 1
// The first read of each admin path 503s the way a warming instance does.
if (selectorCalls === 1) return new Response('warming up', { status: 503 })
return Response.json({ selector: { generation: 11, membership } })
}
if (paths.filter((value) => value === path).length === 1) {
return new Response('warming up', { status: 503 })
}
return Response.json({ v: 1, control: control(4, false) })
}
})
assert.equal(result.control.generation, 4)
assert.deepEqual(paths, [
'/v1/admin/admission-selector/status',
'/v1/admin/admission-selector/status',
'/v1/admin/regional-rehome-control',
'/v1/admin/regional-rehome-control'
])
})
test('fails when both attempts at the director control endpoint return 503', async () => {
const config = parseRegionalRehomeArguments(
argumentsFor('inspect'),
{ ORCA_RELAY_ADMIN_ID_TOKEN: 'token' }
)
let calls = 0
await assert.rejects(
operateRegionalRehome(config, {
wait: async () => {},
fetch: async () => {
calls += 1
return new Response('warming up', { status: 503 })
}
}),
/returned 503/
)
assert.equal(calls, 2)
})
@@ -1,10 +1,12 @@
import { pathToFileURL } from 'node:url'
import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs'
import {
applyExactAdmissionSelector,
inspectAdmissionSelector,
membershipWithStates,
selectorCellState
} from './relay-admission-selector.mjs'
import { SAME_CAP_CELLS } from './relay-production-same-cap-wave.mjs'
const DIRECTOR_ORIGIN = 'https://relay.onorca.dev'
export const PRODUCTION_CAPACITY_CELL_IDS = [
@@ -30,6 +32,9 @@ function cellOrigin(cellId) {
return `https://${cellId.slice('production-gce-'.length)}.relay.onorca.dev`
}
// The same-cap roll covers the Asia cells the US-only capacity rollout never touches.
const APPROVED_CELL_LISTS = { 'same-cap': SAME_CAP_CELLS }
export function parseProductionCapacityCellArguments(argv) {
const values = {}
for (let index = 0; index < argv.length; index += 2) {
@@ -41,8 +46,15 @@ export function parseProductionCapacityCellArguments(argv) {
if (!['isolate', 'drain', 'activate'].includes(values.mode)) {
throw new Error('--mode must be isolate, drain, or activate')
}
const approvedList = values['approved-cells']
if (approvedList !== undefined && !APPROVED_CELL_LISTS[approvedList]) {
throw new Error('--approved-cells is not a known allowlist')
}
const approvedCellIds = approvedList === undefined
? PRODUCTION_CAPACITY_CELL_IDS
: APPROVED_CELL_LISTS[approvedList]
const cellId = values['cell-id']
if (!PRODUCTION_CAPACITY_CELL_IDS.includes(cellId)) {
if (!approvedCellIds.includes(cellId)) {
throw new Error('production capacity target is not approved')
}
const expectedCellOrigin = cellOrigin(cellId)
@@ -72,12 +84,16 @@ export async function prepareProductionCapacityCell(config, overrides = {}) {
if (!token || token.length > 8_192) throw new Error('admin identity token is unavailable')
const postAt = async (origin, path, body) =>
await responseJson(
await fetchImpl(`${origin}${path}`, {
method: 'POST',
headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' },
body: JSON.stringify(body),
signal: AbortSignal.timeout(30_000)
}),
await fetchAdminOnceMore(
fetchImpl,
`${origin}${path}`,
{
method: 'POST',
headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' },
body: JSON.stringify(body)
},
{ wait: overrides.wait }
),
path
)
const post = async (path, body) => await postAt(config.directorOrigin, path, body)
@@ -104,6 +104,47 @@ describe('production Relay capacity cell admission', () => {
'--cell-id', 'production-gce-c7',
'--mode', 'isolate'
]), /origin is not exact/)
assert.throws(() => parseProductionCapacityCellArguments([
'--director-origin', 'https://relay.onorca.dev',
'--cell-origin', 'https://c27.relay.onorca.dev',
'--cell-id', 'production-gce-c27',
'--mode', 'isolate'
]), /not approved/)
})
it('admits the same-cap Asia cells only under the same-cap allowlist', () => {
for (const cellId of ['production-gce-c27', 'production-gce-c28', 'production-gce-c29']) {
const hostname = cellId.slice('production-gce-'.length)
assert.deepEqual(parseProductionCapacityCellArguments([
'--director-origin', 'https://relay.onorca.dev',
'--cell-origin', `https://${hostname}.relay.onorca.dev`,
'--cell-id', cellId,
'--approved-cells', 'same-cap',
'--mode', 'isolate'
]), {
directorOrigin: 'https://relay.onorca.dev',
cellOrigin: `https://${hostname}.relay.onorca.dev`,
cellId,
mode: 'isolate'
})
}
for (const cellId of ['production-gce-c17', 'production-gce-c18', 'production-gce-c30']) {
const hostname = cellId.slice('production-gce-'.length)
assert.throws(() => parseProductionCapacityCellArguments([
'--director-origin', 'https://relay.onorca.dev',
'--cell-origin', `https://${hostname}.relay.onorca.dev`,
'--cell-id', cellId,
'--approved-cells', 'same-cap',
'--mode', 'isolate'
]), /not approved/)
}
assert.throws(() => parseProductionCapacityCellArguments([
'--director-origin', 'https://relay.onorca.dev',
'--cell-origin', 'https://c27.relay.onorca.dev',
'--cell-id', 'production-gce-c27',
'--approved-cells', 'every-cell',
'--mode', 'isolate'
]), /not a known allowlist/)
})
it('isolates only the selected cell without depending on its runtime', async () => {
@@ -170,4 +211,42 @@ describe('production Relay capacity cell admission', () => {
/irreversible/
)
})
it('retries a transient 503 on the cell drain endpoint', async () => {
let calls = 0
const result = await prepareProductionCapacityCell(
{ ...config, mode: 'drain' },
{
token: 'token',
wait: async () => {},
fetch: async (url) => {
assert.equal(new URL(url).pathname, '/v1/admin/drain')
calls += 1
if (calls === 1) return response({ error: 'warming up' }, 503)
return response({ v: 1, draining: true })
}
}
)
assert.equal(calls, 2)
assert.deepEqual(result, { changed: false, drained: true })
})
it('fails when both drain attempts return a transient 503', async () => {
let calls = 0
await assert.rejects(
prepareProductionCapacityCell(
{ ...config, mode: 'drain' },
{
token: 'token',
wait: async () => {},
fetch: async () => {
calls += 1
return response({ error: 'warming up' }, 503)
}
}
),
/returned 503/
)
assert.equal(calls, 2)
})
})
@@ -1,4 +1,5 @@
import { pathToFileURL } from 'node:url'
import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs'
const PRODUCTION_CELL = /^production-gce-c(?:7|8|9|10|13|14|15|16|19|20|21|22|23|24|25|26)$/
const DIRECTOR_ORIGIN = 'https://relay.onorca.dev'
@@ -35,7 +36,8 @@ export function parseRehomeTrustProbeArguments(argv, environment = process.env)
export async function probeRehomeTrust(config, dependencies = {}) {
const fetchImpl = dependencies.fetch ?? fetch
const response = await fetchImpl(
const response = await fetchAdminOnceMore(
fetchImpl,
`${config.directorOrigin}/v1/admin/regional-rehome-trust-probe`,
{
method: 'POST',
@@ -47,9 +49,9 @@ export async function probeRehomeTrust(config, dependencies = {}) {
v: 1,
sourceCellId: config.cellId,
sourceCellIncarnation: config.cellIncarnation
}),
signal: AbortSignal.timeout(30_000)
}
})
},
{ wait: dependencies.wait }
)
const body = await response.json().catch(() => ({}))
if (!response.ok) {
@@ -68,3 +68,46 @@ test('rejects partial or mismatched proof', async () => {
/incomplete/
)
})
const provenProbe = {
v: 1,
dedicatedIdentity: {
firstOutcome: 'host-not-connected',
secondOutcome: 'host-not-connected',
accepted: true,
idempotent: true
},
sharedRuntimeIdentityRejected: true,
proven: true
}
test('retries a transient 503 on the trust probe and proves on the second answer', async () => {
const config = parseRehomeTrustProbeArguments(argv, environment)
let calls = 0
const result = await probeRehomeTrust(config, {
wait: async () => {},
fetch: async () => {
calls += 1
if (calls === 1) return new Response('warming up', { status: 503 })
return Response.json(provenProbe)
}
})
assert.equal(calls, 2)
assert.equal(result.proven, true)
})
test('fails when both trust-probe attempts return a transient 503', async () => {
const config = parseRehomeTrustProbeArguments(argv, environment)
let calls = 0
await assert.rejects(
probeRehomeTrust(config, {
wait: async () => {},
fetch: async () => {
calls += 1
return new Response('warming up', { status: 503 })
}
}),
/returned 503/
)
assert.equal(calls, 2)
})
@@ -0,0 +1,43 @@
import assert from 'node:assert/strict'
import { readFileSync } from 'node:fs'
import { test } from 'node:test'
import { fileURLToPath } from 'node:url'
import { relayWorkflowUrl } from './relay-repository.mjs'
const WORKFLOWS = [
'deploy-relay-production-same-cap-job.yml',
'operate-relay-production-rehome-job.yml'
]
function workflow(name) {
return readFileSync(fileURLToPath(relayWorkflowUrl(name)), 'utf8')
}
// A single transient 5xx from a warming instance behind the global load balancer
// must not fail a canary, so no admin endpoint may be read by a bare curl.
test('no admin endpoint is reached by a curl without a bounded retry', () => {
for (const name of WORKFLOWS) {
for (const invocation of workflow(name).split(/\bcurl\b/).slice(1)) {
const flags = invocation.split('\n }')[0]
assert.match(flags, /--retry 3 --retry-delay 2 --retry-connrefused/, name)
assert.match(flags, /--max-time 30/, name)
// --retry-all-errors would also retry 401, 403, and 409, which are final.
assert.doesNotMatch(flags, /--retry-all-errors/, name)
}
}
})
test('every retried admin request captures only the final attempt body', () => {
const job = workflow('deploy-relay-production-same-cap-job.yml')
// --fail-with-body writes every failed attempt to stdout, so a retried
// request must land in a file curl truncates per attempt.
assert.match(job, /--output "\$\{out\}"/)
assert.equal(job.split('admin_post() {').length - 1, 2)
for (const call of [
/CURRENT_RUNTIME="\$\(admin_post current-runtime/,
/CURRENT_DIRECTOR_STATUS="\$\(admin_post current-cell-status/,
/TARGET_RUNTIME="\$\(admin_post target-runtime/,
/TARGET_DIRECTOR_STATUS="\$\(admin_post target-cell-status/
]) assert.match(job, call)
assert.doesNotMatch(job, /\$\(curl /)
})
@@ -0,0 +1,29 @@
// A single transient 5xx (load-balancer warm-up behind a fresh instance) must not fail a
// deploy step. 4xx is never retried: auth and generation-mismatch answers are final.
const TRANSIENT_STATUSES = [500, 502, 503, 504]
const RETRY_DELAY_MS = 2_000
const REQUEST_TIMEOUT_MS = 30_000
export function isTransientAdminStatus(status) {
return TRANSIENT_STATUSES.includes(status)
}
// Each attempt gets its own timeout budget, so a reused signal cannot abort the retry.
export async function fetchAdminOnceMore(fetchImpl, url, init, overrides = {}) {
const wait = overrides.wait ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms)))
const timeoutMs = overrides.timeoutMs ?? REQUEST_TIMEOUT_MS
const retryDelayMs = overrides.retryDelayMs ?? RETRY_DELAY_MS
const attempt = async () =>
await fetchImpl(url, { ...init, signal: AbortSignal.timeout(timeoutMs) })
let response
try {
response = await attempt()
} catch {
await wait(retryDelayMs)
return await attempt()
}
if (!isTransientAdminStatus(response.status)) return response
await response.arrayBuffer?.().catch(() => undefined)
await wait(retryDelayMs)
return await attempt()
}
@@ -0,0 +1,130 @@
import assert from 'node:assert/strict'
import { test } from 'node:test'
import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs'
const url = 'https://relay.onorca.dev/v1/admin/cell-status'
const init = { method: 'POST', body: '{"v":1}' }
function recordingWait(waits) {
return async (ms) => { waits.push(ms) }
}
test('a single transient 5xx is retried and the second answer is returned', async () => {
const waits = []
const statuses = [503, 200]
let calls = 0
const response = await fetchAdminOnceMore(
async () => {
calls += 1
const status = statuses.shift()
return new Response(JSON.stringify({ ok: status === 200 }), { status })
},
url,
init,
{ wait: recordingWait(waits) }
)
assert.equal(calls, 2)
assert.equal(response.status, 200)
assert.deepEqual(waits, [2_000])
assert.deepEqual(await response.json(), { ok: true })
})
test('a connection failure is retried and the second answer is returned', async () => {
const waits = []
let calls = 0
const response = await fetchAdminOnceMore(
async () => {
calls += 1
if (calls === 1) throw new TypeError('fetch failed')
return Response.json({ ok: true })
},
url,
init,
{ wait: recordingWait(waits) }
)
assert.equal(calls, 2)
assert.equal(response.status, 200)
assert.deepEqual(waits, [2_000])
})
test('two transient failures surface the second answer without a third attempt', async () => {
let calls = 0
const response = await fetchAdminOnceMore(
async () => {
calls += 1
return new Response('down', { status: 503 })
},
url,
init,
{ wait: async () => {} }
)
assert.equal(calls, 2)
assert.equal(response.status, 503)
})
test('two connection failures rethrow the second error', async () => {
let calls = 0
await assert.rejects(
fetchAdminOnceMore(
async () => {
calls += 1
throw new TypeError(`fetch failed ${calls}`)
},
url,
init,
{ wait: async () => {} }
),
/fetch failed 2/
)
assert.equal(calls, 2)
})
test('4xx is final: auth and generation-mismatch answers are never retried', async () => {
for (const status of [400, 401, 403, 404, 409, 429]) {
let calls = 0
const response = await fetchAdminOnceMore(
async () => {
calls += 1
return new Response('no', { status })
},
url,
init,
{ wait: async () => { throw new Error('must not wait') } }
)
assert.equal(calls, 1, `status ${status} must not be retried`)
assert.equal(response.status, status)
}
})
test('each attempt carries its own unexpired timeout signal', async () => {
const signals = []
await fetchAdminOnceMore(
async (_url, attemptInit) => {
signals.push(attemptInit.signal)
return new Response('down', { status: 502 })
},
url,
init,
{ wait: async () => {}, timeoutMs: 30_000 }
)
assert.equal(signals.length, 2)
assert.notEqual(signals[0], signals[1])
assert.equal(signals[1].aborted, false)
})
test('the caller init is forwarded unchanged apart from the signal', async () => {
let seen
await fetchAdminOnceMore(
async (seenUrl, attemptInit) => {
seen = { seenUrl, attemptInit }
return Response.json({})
},
url,
{ method: 'POST', headers: { authorization: 'Bearer t' }, body: '{"v":1}' },
{ wait: async () => {} }
)
assert.equal(seen.seenUrl, url)
assert.equal(seen.attemptInit.method, 'POST')
assert.deepEqual(seen.attemptInit.headers, { authorization: 'Bearer t' })
assert.equal(seen.attemptInit.body, '{"v":1}')
})
@@ -0,0 +1,94 @@
import { spawnSync } from 'node:child_process'
import { fileURLToPath } from 'node:url'
import {
RELAY_REPOSITORY_ROOT,
relayTreePath,
relayWorkflowPath
} from './relay-repository.mjs'
const SHA = /^[a-f0-9]{40}$/
// Every file that decides how relay evidence is produced, sealed, verified, and then spent against
// production; identical content across two commits is what makes the older commit's verdict binding.
export const TRUSTED_EVIDENCE_CODE_PATHS = [
// Produces and seals the 15-minute dry-run evidence.
relayWorkflowPath('monitor-relay-production.yml'),
relayWorkflowPath('monitor-relay-production-job.yml'),
// Download it, verify its authority, and mutate production on it.
relayWorkflowPath('deploy-relay-production-same-cap.yml'),
relayWorkflowPath('deploy-relay-production-same-cap-job.yml'),
relayWorkflowPath('operate-relay-production-rehome.yml'),
relayWorkflowPath('operate-relay-production-rehome-job.yml'),
// Sealing, verification, the wave/canary authority, and the path constants below.
relayTreePath('dev/scripts/relay-evidence-code-provenance.mjs'),
relayTreePath('dev/scripts/relay-monitor-evidence.mjs'),
relayTreePath('dev/scripts/relay-production-same-cap-wave.mjs'),
relayTreePath('dev/scripts/relay-repository.mjs'),
// Every other script those jobs run against live production.
relayTreePath('dev/scripts/infra.mjs'),
relayTreePath('dev/scripts/operate-relay-regional-rehome.mjs'),
relayTreePath('dev/scripts/prepare-relay-production-capacity-canary.mjs'),
relayTreePath('dev/scripts/probe-relay-rehome-trust.mjs'),
relayTreePath('dev/scripts/validate-relay-capacity-plan.mjs'),
relayTreePath('dev/scripts/verify-relay-capacity-transition.mjs'),
// The monitor itself and the live preflight recheck, plus anything that changes their behaviour.
relayTreePath('apps/relay-ops'),
relayTreePath('package.json'),
relayTreePath('pnpm-lock.yaml'),
relayTreePath('pnpm-workspace.yaml'),
// The Cloud SQL rollout lease every mutation job takes and releases.
'.github/actions/cloud-sql-rollout-lease'
]
function git(root, args) {
const result = spawnSync('git', ['-C', root, ...args], { encoding: 'utf8' })
if (result.error) throw new Error('relay evidence provenance cannot run git')
return result
}
/**
* Accepts evidence sealed at a different commit only when the current commit descends from it and
* every trusted path is byte-identical, so the verdict provably came from this exact code. Anything
* git cannot answer (no checkout, unknown commit, shallow clone) fails closed.
*/
export function requireSameEvidenceCode({
sealedSha,
currentSha,
label,
repositoryRoot = fileURLToPath(RELAY_REPOSITORY_ROOT)
}) {
if (!SHA.test(sealedSha ?? '') || !SHA.test(currentSha ?? '')) {
throw new Error(`${label} commit is invalid`)
}
if (sealedSha === currentSha) return
if (git(repositoryRoot, ['rev-parse', '--git-dir']).status !== 0) {
throw new Error(`${label} commit cannot be compared without a git checkout`)
}
for (const sha of [sealedSha, currentSha]) {
if (git(repositoryRoot, ['rev-parse', '--verify', '--quiet', `${sha}^{commit}`]).status !== 0) {
throw new Error(
`${label} commit ${sha} is unknown to this checkout; check out with fetch-depth: 0`
)
}
}
const ancestry = git(repositoryRoot, ['merge-base', '--is-ancestor', sealedSha, currentSha])
if (ancestry.status === 1) {
throw new Error(`${label} commit ${sealedSha} is not an ancestor of ${currentSha}`)
}
if (ancestry.status !== 0) {
throw new Error(`${label} commit ancestry could not be determined`)
}
const diff = git(repositoryRoot, [
'diff',
'--name-only',
sealedSha,
currentSha,
'--',
...TRUSTED_EVIDENCE_CODE_PATHS
])
if (diff.status !== 0) throw new Error(`${label} commit comparison failed`)
const changed = diff.stdout.split('\n').filter(Boolean)
if (changed.length > 0) {
throw new Error(`${label} code changed after it was sealed: ${changed.join(',')}`)
}
}
+18 -5
View File
@@ -2,6 +2,7 @@ import { createHash } from 'node:crypto'
import { chmod, readFile, readdir, stat, writeFile } from 'node:fs/promises'
import { basename, join, resolve } from 'node:path'
import { pathToFileURL } from 'node:url'
import { requireSameEvidenceCode } from './relay-evidence-code-provenance.mjs'
const SAFE_ID = /^[A-Za-z0-9][A-Za-z0-9._-]{1,127}$/
const SHA = /^[a-f0-9]{40}$/
@@ -102,7 +103,7 @@ export async function createEvidenceManifest(argv) {
return manifest
}
async function readAndVerifyManifest(directory, expected) {
async function readAndVerifyManifest(directory, expected, sameCodeCommit) {
const manifest = JSON.parse(
await readFile(join(directory, 'evidence-manifest.json'), 'utf8')
)
@@ -111,11 +112,23 @@ async function readAndVerifyManifest(directory, expected) {
manifest.incidentId !== expected.incidentId ||
manifest.runId !== expected.runId ||
manifest.runAttempt !== expected.runAttempt ||
manifest.commitSha !== expected.commitSha ||
manifest.mode !== expected.mode
!SHA.test(manifest.commitSha ?? '') ||
manifest.mode !== expected.mode ||
(!sameCodeCommit && manifest.commitSha !== expected.commitSha)
) {
throw new Error('relay monitor evidence provenance does not match')
}
// Unrelated merges land on main every few minutes, so the deployer resolves a newer commit than
// the monitor it must trust; identical monitor and mutation code is the property the SHA stood in
// for. Restore and mutation keep the exact-SHA bind: both run at the commit that sealed them.
if (sameCodeCommit) {
requireSameEvidenceCode({
sealedSha: manifest.commitSha,
currentSha: expected.commitSha,
label: 'relay monitor evidence',
...sameCodeCommit
})
}
const names = Object.keys(manifest.files ?? {})
if (!names.includes(`${expected.incidentId}.state.json`)) {
throw new Error('relay monitor evidence has no durable state')
@@ -209,12 +222,12 @@ function validCompletedDryRunState(state, expected, nowMs, maxAgeMs) {
)
}
export async function verifyDryRunAuthority(argv, now = Date.now) {
export async function verifyDryRunAuthority(argv, now = Date.now, repositoryRoot) {
const values = argumentsByName(argv)
const directory = resolve(values.directory ?? '')
const expected = provenance(values)
if (expected.mode !== 'dry-run') throw new Error('relay mutation requires dry-run evidence')
const manifest = await readAndVerifyManifest(directory, expected)
const manifest = await readAndVerifyManifest(directory, expected, { repositoryRoot })
const state = JSON.parse(
await readFile(join(directory, `${expected.incidentId}.state.json`), 'utf8')
)
@@ -1,9 +1,15 @@
import assert from 'node:assert/strict'
import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises'
import { execFileSync } from 'node:child_process'
import { mkdir, mkdtemp, readFile, rm, stat, writeFile } from 'node:fs/promises'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { dirname, join } from 'node:path'
import test from 'node:test'
import { relayWorkflowPath, relayWorkflowUrl } from './relay-repository.mjs'
import { TRUSTED_EVIDENCE_CODE_PATHS } from './relay-evidence-code-provenance.mjs'
import {
RELAY_REPOSITORY_ROOT,
relayWorkflowPath,
relayWorkflowUrl
} from './relay-repository.mjs'
import {
createEvidenceManifest,
verifyDryRunAuthority,
@@ -12,7 +18,7 @@ import {
} from './relay-monitor-evidence.mjs'
const now = Date.parse('2026-07-28T12:00:00.000Z')
const provenance = [
const provenanceFor = (commitSha) => [
'--incident-id',
'relay-123',
'--run-id',
@@ -20,10 +26,11 @@ const provenance = [
'--run-attempt',
'1',
'--commit-sha',
'a'.repeat(40),
commitSha,
'--mode',
'dry-run'
]
const provenance = provenanceFor('a'.repeat(40))
const selector = {
generation: 2,
membership: {
@@ -513,3 +520,157 @@ test('monitor uses a reusable job so exact job_workflow_ref is present', async (
assert.match(job, /workflow_call:/)
assert.match(job, /environment: production/)
})
function gitIn(root, ...args) {
return execFileSync('git', ['-C', root, ...args], { encoding: 'utf8' }).trim()
}
// A real repository shaped like main under unrelated merge traffic: one sealed commit, a
// descendant that only touched untrusted files, a descendant that touched the monitor, and a
// sibling that never descended from the seal.
async function trustedCodeRepository() {
const root = await mkdtemp(join(tmpdir(), 'relay-evidence-repository-'))
gitIn(root, 'init', '--quiet')
gitIn(root, 'config', 'user.email', 'relay@example.test')
gitIn(root, 'config', 'user.name', 'Relay Evidence Test')
gitIn(root, 'config', 'commit.gpgsign', 'false')
const commit = async (path, body, message) => {
await mkdir(dirname(join(root, path)), { recursive: true })
await writeFile(join(root, path), body)
gitIn(root, 'add', '--all')
gitIn(root, 'commit', '--quiet', '--no-verify', '--message', message)
return gitIn(root, 'rev-parse', 'HEAD')
}
const base = await commit(
'cloud/apps/relay-ops/src/incident-monitor.ts',
'export const v = 1\n',
'monitor'
)
const sealed = await commit('README.md', 'base\n', 'base')
const sameCode = await commit('README.md', 'an unrelated merge\n', 'unrelated')
const changedCode = await commit(
'cloud/apps/relay-ops/src/incident-monitor.ts',
'export const v = 2\n',
'monitor change'
)
// Branches before the seal, so the seal is not in its history even though its code matches.
gitIn(root, 'checkout', '--quiet', '--detach', base)
const sibling = await commit('README.md', 'a divergent line\n', 'divergent')
return { root, sealed, sameCode, changedCode, sibling }
}
const authorityAt = (directory, commitSha, repositoryRoot) => verifyDryRunAuthority(
[
'--directory',
directory,
...provenanceFor(commitSha),
'--required-migration-policy',
'strict'
],
() => now,
repositoryRoot
)
test('accepts dry-run evidence sealed by identical code at an ancestor commit', async () => {
const repository = await trustedCodeRepository()
const directory = await evidenceDirectory()
try {
await createEvidenceManifest([
'--directory',
directory,
...provenanceFor(repository.sealed)
])
// An exact match never consults git: a root with no checkout at all still verifies.
await assert.doesNotReject(authorityAt(directory, repository.sealed, directory))
await assert.doesNotReject(authorityAt(directory, repository.sameCode, repository.root))
} finally {
await rm(repository.root, { recursive: true, force: true })
await rm(directory, { recursive: true, force: true })
}
})
test('rejects dry-run evidence whose monitor code or lineage differs', async () => {
const repository = await trustedCodeRepository()
const directory = await evidenceDirectory()
try {
await createEvidenceManifest([
'--directory',
directory,
...provenanceFor(repository.sealed)
])
await assert.rejects(
authorityAt(directory, repository.changedCode, repository.root),
/code changed after it was sealed: cloud\/apps\/relay-ops\/src\/incident-monitor\.ts/
)
await assert.rejects(
authorityAt(directory, repository.sibling, repository.root),
/is not an ancestor of/
)
// Fails closed: a shallow clone that never fetched the sealed commit proves nothing.
await assert.rejects(
authorityAt(directory, 'f'.repeat(40), repository.root),
/unknown to this checkout/
)
// Fails closed: no checkout to compare against.
await assert.rejects(
authorityAt(directory, repository.sameCode, directory),
/cannot be compared without a git checkout/
)
} finally {
await rm(repository.root, { recursive: true, force: true })
await rm(directory, { recursive: true, force: true })
}
})
test('keeps restore and mutation bound to the exact sealing commit', async () => {
const repository = await trustedCodeRepository()
const directory = await evidenceDirectory()
try {
await createEvidenceManifest([
'--directory',
directory,
...provenanceFor(repository.sealed)
])
await assert.rejects(
verifyRestoredEvidence([
'--directory',
directory,
...provenanceFor(repository.sameCode)
]),
/provenance does not match/
)
await assert.rejects(
verifyMutationEvidence(
[
'--directory',
directory,
...provenanceFor(repository.sameCode),
'--mutation-mode',
'execute',
'--source-cell-id',
'c1',
'--director-origin',
'https://relay.example'
],
{ ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' },
async () => Response.json({ selector }),
() => now
),
/provenance does not match/
)
} finally {
await rm(repository.root, { recursive: true, force: true })
await rm(directory, { recursive: true, force: true })
}
})
// A trusted path that no longer exists silently stops being compared, so the same-code rule would
// pass over code it was written to pin.
test('every trusted provenance path exists in this checkout', async () => {
for (const path of TRUSTED_EVIDENCE_CODE_PATHS) {
await assert.doesNotReject(
stat(new URL(path, RELAY_REPOSITORY_ROOT)),
`${path} is missing`
)
}
})
@@ -1,5 +1,6 @@
import { readFileSync } from 'node:fs'
import { pathToFileURL } from 'node:url'
import { requireSameEvidenceCode } from './relay-evidence-code-provenance.mjs'
export const SAME_CAP_CELLS = [
'production-gce-c7', 'production-gce-c8', 'production-gce-c9', 'production-gce-c10',
@@ -85,10 +86,10 @@ export function canaryAuthority(input) {
}
}
export function verifyCanaryAuthority(authority, expected) {
export function verifyCanaryAuthority(authority, expected, repositoryRoot) {
if (
authority?.v !== 1 ||
authority.commitSha !== expected.commitSha ||
!/^[0-9a-f]{40}$/.test(authority.commitSha ?? '') ||
authority.runId !== expected.runId ||
authority.targetDigest !== expected.targetDigest ||
authority.rollbackDigest !== expected.rollbackDigest ||
@@ -96,6 +97,14 @@ export function verifyCanaryAuthority(authority, expected) {
authority.rehomeGeneration !== Number(expected.rehomeGeneration) ||
!SAME_CAP_CELLS.includes(authority.cellId)
) throw new Error('canary authority does not match this batch')
// The batch dispatch resolves main after the canary sealed, so bind to the same code, not the
// same SHA; every field above still pins this batch to that exact canary.
requireSameEvidenceCode({
sealedSha: authority.commitSha,
currentSha: expected.commitSha,
label: 'relay same-cap canary authority',
repositoryRoot
})
return authority
}
@@ -1,4 +1,8 @@
import assert from 'node:assert/strict'
import { execFileSync } from 'node:child_process'
import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'
import { tmpdir } from 'node:os'
import { dirname, join } from 'node:path'
import { test } from 'node:test'
import {
canaryAuthority,
@@ -104,3 +108,66 @@ test('seals and verifies canary authority for later batches', () => {
rehomeGeneration: '4'
}), /does not match/)
})
function gitIn(root, ...args) {
return execFileSync('git', ['-C', root, ...args], { encoding: 'utf8' }).trim()
}
async function canaryRepository() {
const root = await mkdtemp(join(tmpdir(), 'relay-same-cap-canary-'))
gitIn(root, 'init', '--quiet')
gitIn(root, 'config', 'user.email', 'relay@example.test')
gitIn(root, 'config', 'user.name', 'Relay Wave Test')
gitIn(root, 'config', 'commit.gpgsign', 'false')
const commit = async (path, body, message) => {
await mkdir(dirname(join(root, path)), { recursive: true })
await writeFile(join(root, path), body)
gitIn(root, 'add', '--all')
gitIn(root, 'commit', '--quiet', '--no-verify', '--message', message)
return gitIn(root, 'rev-parse', 'HEAD')
}
const sealed = await commit(
'cloud/dev/scripts/relay-production-same-cap-wave.mjs',
'export const v = 1\n',
'wave'
)
const sameCode = await commit('README.md', 'an unrelated merge\n', 'unrelated')
const changedCode = await commit(
'cloud/dev/scripts/relay-production-same-cap-wave.mjs',
'export const v = 2\n',
'wave change'
)
return { root, sealed, sameCode, changedCode }
}
test('a batch trusts a canary sealed by identical code at an ancestor commit', async () => {
const repository = await canaryRepository()
try {
const authority = canaryAuthority({
cellIds: 'production-gce-c7',
targetDigest,
rollbackDigest,
confirmation: `ROLL_RELAY_SAME_CAP ${targetDigest} production-gce-c7`,
commitSha: repository.sealed,
runId: '42',
selectorGeneration: '11',
rehomeGeneration: '4'
})
const verifyAt = (commitSha, repositoryRoot) => verifyCanaryAuthority(authority, {
commitSha,
runId: '42',
targetDigest,
rollbackDigest,
selectorGeneration: '13',
rehomeGeneration: '4'
}, repositoryRoot)
assert.equal(verifyAt(repository.sameCode, repository.root).cellId, 'production-gce-c7')
assert.throws(
() => verifyAt(repository.changedCode, repository.root),
/code changed after it was sealed/
)
assert.throws(() => verifyAt('f'.repeat(40), repository.root), /unknown to this checkout/)
} finally {
await rm(repository.root, { recursive: true, force: true })
}
})
@@ -73,7 +73,10 @@ test('same-cap wrapper is reusable, canary-bound, and sequential', () => {
job,
/--rollback-image "\$\{DESIRED_IMAGE\}" \\\n {16}--rehome-director-service-account "\$\{DIRECTOR_RUNTIME_SERVICE_ACCOUNT\}"/
)
assert.match(job, /host-drain \\\n {14}\| jq -e '\.changes == 2' >\/dev\/null/)
assert.match(
job,
/host-drain \\\n {16}--regional-rehome-protocol "\$\{DESIRED_REHOME_PROTOCOL\}" \\\n {14}\| jq -e '\.changes == 2' >\/dev\/null/
)
assert.match(job, /resume requires the isolated migration-only cell/)
assert.match(job, /test "\$\{TARGET_INCARNATION\}" = "\$\{SOURCE_INCARNATION\}"/)
assert.match(job, /\(.regionalRehomeProtocol \/\/ 0\) == \$protocol/)
@@ -92,7 +95,11 @@ test('same-cap wrapper is reusable, canary-bound, and sequential', () => {
// age checks must scale by wave or cell_2+ can never pass; the bound's
// per-wave step is the cell job timeout, so the two must move together.
assert.match(job, /--required-migration-policy strict \\\n --wave-index "\$\{WAVE_INDEX\}"/)
assert.match(job, /dry-run\.state\.json" \\\n --wave-index "\$\{WAVE_INDEX\}" "\$\{RETRY_ARGS\[@\]\}"/)
// Wave 0 must retry freshness-only failures too: one Cloud Monitoring publish
// lag at the sample instant is not health evidence, and single-shot wave 0
// failed a whole batch on a series that was fresh again a minute later.
assert.match(job, /dry-run\.state\.json" \\\n --wave-index "\$\{WAVE_INDEX\}" --retry-freshness/)
assert.doesNotMatch(job, /RETRY_ARGS/)
assert.match(job, /timeout-minutes: 75/)
// Both age gates step by the cell job timeout above; the constant is
// duplicated across the two languages, so pin each copy to it.
+15
View File
@@ -1,4 +1,6 @@
import { readFileSync } from 'node:fs'
import { relative } from 'node:path'
import { fileURLToPath } from 'node:url'
// Single place naming the repository the Relay workflows live in and where their files sit. The
// public-repo copy moves this tree under cloud/, prefixes every workflow filename, and changes the
@@ -11,6 +13,19 @@ export const RELAY_WORKFLOW_FILE_PREFIX = 'cloud-'
// this tree moves under cloud/, so the depth changes at the copy even though the layout does not.
export const RELAY_WORKFLOW_DIRECTORY = new URL('../../../.github/workflows/', import.meta.url)
// Repository root, derived from the one directory above that already tracks the copy's depth.
export const RELAY_REPOSITORY_ROOT = new URL('../../', RELAY_WORKFLOW_DIRECTORY)
// Repository-relative path for a file in this tree. The prefix is 'cloud/' here and empty where
// the tree is the repository root, so callers naming git paths never restate the layout.
export function relayTreePath(suffix) {
const prefix = relative(
fileURLToPath(RELAY_REPOSITORY_ROOT),
fileURLToPath(new URL('../../', import.meta.url))
).split(/[\\/]/).filter(Boolean)
return [...prefix, suffix].join('/')
}
export function relayWorkflowFile(name) {
return `${RELAY_WORKFLOW_FILE_PREFIX}${name}`
}
@@ -0,0 +1,217 @@
import assert from 'node:assert/strict'
import { spawnSync } from 'node:child_process'
import { readFileSync } from 'node:fs'
import { describe, it } from 'node:test'
import { parseProductionCapacityCellArguments } from './prepare-relay-production-capacity-canary.mjs'
import { SAME_CAP_CELLS } from './relay-production-same-cap-wave.mjs'
import { readRelayWorkflow } from './relay-repository.mjs'
import { validateCapacityPlan } from './validate-relay-capacity-plan.mjs'
const workflow = readRelayWorkflow('deploy-relay-production-same-cap-job.yml')
const capacityWorkflow = readRelayWorkflow('deploy-relay-production-capacity-job.yml')
const production = readFileSync(
new URL('../../infra/terraform/environments/production.tfvars', import.meta.url),
'utf8'
)
const REHOME_SOURCE_CELLS = rehomeSourceCells()
const DIRECTOR_IDENTITY = 'relay-director@onorca-cloud.iam.gserviceaccount.com'
const AUDIENCE = 'https://relay.onorca.dev/v1/admin/host-drain'
const ROLLBACK_IMAGE = `us-central1-docker.pkg.dev/p/orca-cloud/relay@sha256:${'d'.repeat(64)}`
const TARGET_IMAGE = `us-central1-docker.pkg.dev/p/orca-cloud/relay@sha256:${'e'.repeat(64)}`
// The startup template emits rehome trust only for cells in this list, so it is what decides
// whether a cell's plan may carry those lines at all.
function rehomeSourceCells() {
const start = production.indexOf('relay_region_rehome_source_cell_ids = [')
assert.notEqual(start, -1, 'production.tfvars has no rehome source cell list')
const end = production.indexOf(']', start)
assert.notEqual(end, -1, 'the rehome source cell list is unterminated')
return new Set(
[...production.slice(start, end).matchAll(/"([^"]+)"/g)].map(([, cell]) => cell)
)
}
function startupScript({ cap, image, trusted }) {
return [
` printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '${cap}'`,
` printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '60'`,
...(trusted ? [
` printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '${DIRECTOR_IDENTITY}'`,
` printf 'ORCA_RELAY_REHOME_AUDIENCE=%s\\n' '${AUDIENCE}'`
] : []),
`printf 'ORCA_RELAY_IMAGE_DIGEST=%s\\n' '${image.split('@')[1]}'`,
`docker pull '${image}'`,
'docker run --detach \\',
' --name orca-relay \\',
` '${image}'`
].join('\n')
}
// The exact shape the apply step's plan has: template replaced, MIG rebound to it.
function rollPlan({ cellId, cap, protocol }) {
return {
configuration: {
root_module: {
resources: [{
address: 'google_compute_instance_group_manager.relay_gce_cell',
expressions: {
version: [{
instance_template: {
references: [
'google_compute_instance_template.relay_gce_cell',
'each.key'
]
},
name: { constant_value: 'primary' }
}]
}
}]
}
},
resource_changes: [
{
address: `google_compute_instance_template.relay_gce_cell[${JSON.stringify(cellId)}]`,
change: {
actions: ['create', 'delete'],
before: {
metadata_startup_script: startupScript({
cap,
image: ROLLBACK_IMAGE,
trusted: protocol === 1
})
},
after: {
metadata_startup_script: startupScript({
cap,
image: TARGET_IMAGE,
trusted: protocol === 1
}),
self_link: null
},
after_unknown: { self_link: true }
}
},
{
address: `google_compute_instance_group_manager.relay_gce_cell[${JSON.stringify(cellId)}]`,
change: {
actions: ['update'],
before: { target_size: 1, version: [{ instance_template: 'old' }] },
after: { target_size: 1, version: [{ instance_template: null }] },
after_unknown: { version: [{ instance_template: true }] }
}
}
]
}
}
function hostname(cellId) {
return cellId.slice('production-gce-'.length)
}
// The job resolves cap and region from the cell id before any admin call; run that block alone.
function resolveCellShape(cellId) {
const start = workflow.indexOf(' TARGET_HOSTNAME="${TARGET_CELL_ID#production-gce-}"')
assert.notEqual(start, -1, 'the same-cap cell shape block is missing')
const end = workflow.indexOf('\n esac\n', start)
assert.notEqual(end, -1, 'the same-cap cell shape block has no esac')
const script = workflow.slice(start, end + '\n esac'.length).replace(/^ {10}/gm, '')
return spawnSync('bash', [
'-euo',
'pipefail',
'-c',
`${script}\necho "\${EXPECTED_REGION} \${EXPECTED_HARD_CAP}"`
], { env: { ...process.env, TARGET_CELL_ID: cellId }, encoding: 'utf8' })
}
describe('same-cap roll scripts accept every same-cap cell', () => {
it('parses every wave cell through the same-cap canary allowlist', () => {
for (const cellId of SAME_CAP_CELLS) {
for (const mode of ['isolate', 'drain', 'activate']) {
assert.deepEqual(parseProductionCapacityCellArguments([
'--director-origin', 'https://relay.onorca.dev',
'--cell-origin', `https://${hostname(cellId)}.relay.onorca.dev`,
'--cell-id', cellId,
'--approved-cells', 'same-cap',
'--mode', mode
]), {
directorOrigin: 'https://relay.onorca.dev',
cellOrigin: `https://${hostname(cellId)}.relay.onorca.dev`,
cellId,
mode
})
}
}
})
it('resolves a cap and region for every wave cell and refuses anything else', () => {
for (const cellId of SAME_CAP_CELLS) {
const resolved = resolveCellShape(cellId)
assert.equal(resolved.status, 0, `${cellId}: ${resolved.stderr}`)
assert.match(resolved.stdout.trim(), /^(us-central1 1000|asia-east2 3000)$/)
}
assert.equal(resolveCellShape('production-gce-c17').status, 1)
assert.equal(resolveCellShape('production-gce-c30').status, 1)
})
it('passes the same-cap allowlist on every canary invocation the job runs', () => {
const invocations = workflow.split('prepare-relay-production-capacity-canary.mjs').slice(1)
assert.equal(invocations.length, 4)
for (const invocation of invocations) {
const lines = invocation.split('\n')
const end = lines.findIndex((line) => !line.endsWith('\\'))
const call = lines.slice(0, end + 1).join(' ')
assert.match(call, /--approved-cells same-cap/)
assert.match(call, /--mode (isolate|drain|activate)/)
}
})
it('passes this cell\'s rehome protocol on every plan validation the job runs', () => {
const invocations = workflow.split('validate-relay-capacity-plan.mjs').slice(1)
assert.equal(invocations.length, 2)
for (const invocation of invocations) {
const lines = invocation.split('\n')
const end = lines.findIndex((line) => !line.trimEnd().endsWith('\\'))
const call = lines.slice(0, end + 1).join(' ')
assert.match(call, /--mode same-cap-cell/)
assert.match(call, /--regional-rehome-protocol "\$\{DESIRED_REHOME_PROTOCOL\}"/)
}
})
it('validates a correct plan for every wave cell at that cell\'s rehome protocol', () => {
for (const cellId of SAME_CAP_CELLS) {
const [region, cap] = resolveCellShape(cellId).stdout.trim().split(' ')
const protocol = REHOME_SOURCE_CELLS.has(cellId) ? 1 : 0
assert.equal(protocol, region === 'us-central1' ? 1 : 0, cellId)
const config = {
mode: 'same-cap-cell',
cellId,
hardCap: Number(cap),
unobservedBound: 60,
image: TARGET_IMAGE,
rollbackImage: ROLLBACK_IMAGE,
rehomeDirectorServiceAccount: DIRECTOR_IDENTITY,
rehomeAudience: AUDIENCE,
regionalRehomeProtocol: String(protocol)
}
const plan = rollPlan({ cellId, cap, protocol })
assert.deepEqual(
validateCapacityPlan(plan, config),
{ mode: 'same-cap-cell', changes: 2 },
cellId
)
// The other protocol must reject the same plan, or the flag decides nothing.
assert.throws(
() => validateCapacityPlan(plan, {
...config,
regionalRehomeProtocol: String(1 - protocol)
}),
/reviewed image and capacity/,
cellId
)
}
})
it('leaves the US-only capacity job on the default allowlist', () => {
assert.doesNotMatch(capacityWorkflow, /--approved-cells/)
})
})
@@ -4,7 +4,18 @@ import { pathToFileURL } from 'node:url'
const SERVICE_ACCOUNT_EMAIL =
/^[a-z][a-z0-9-]{4,28}[a-z0-9]@[a-z0-9-]+\.iam\.gserviceaccount\.com$/
function parseArguments(argv) {
const REHOME_CONFIG =
/^ printf 'ORCA_RELAY_REHOME_(?:DIRECTOR_SERVICE_ACCOUNT|AUDIENCE)=%s\\n' '[^'\n]+'$/
// Only cells listed as regional rehome sources get rehome trust lines in their startup script.
function rehomeProtocol({ regionalRehomeProtocol }) {
if (![0, 1, '0', '1'].includes(regionalRehomeProtocol)) {
throw new Error('same-cap Terraform plan has an invalid regional rehome protocol')
}
return Number(regionalRehomeProtocol)
}
export function parseCapacityPlanArguments(argv) {
const values = {}
for (let index = 0; index < argv.length; index += 2) {
const key = argv[index]
@@ -31,8 +42,12 @@ function parseArguments(argv) {
values.mode === 'same-cap-cell' &&
(!values['rollback-image'] ||
!values['rehome-director-service-account'] ||
!values['rehome-audience'])
!values['rehome-audience'] ||
!['0', '1'].includes(values['regional-rehome-protocol']))
) throw new Error('same-cap validation requires rollback image and rehome trust config')
if (values.mode !== 'same-cap-cell' && values['regional-rehome-protocol'] !== undefined) {
throw new Error('--regional-rehome-protocol applies only to same-cap-cell validation')
}
if (values.mode === 'same-cap-image' && !values['rollback-image']) {
throw new Error('same-cap image validation requires a rollback image')
}
@@ -51,7 +66,8 @@ function parseArguments(argv) {
capacityServiceAccount: values['capacity-service-account'],
rollbackImage: values['rollback-image'],
rehomeDirectorServiceAccount: values['rehome-director-service-account'],
rehomeAudience: values['rehome-audience']
rehomeAudience: values['rehome-audience'],
regionalRehomeProtocol: values['regional-rehome-protocol']
}
}
@@ -175,15 +191,13 @@ function normalizedStartupScript(
/^ printf 'ORCA_RELAY_CELL_CONNECTION_(?:HARD_CAP|UNOBSERVED_BOUND)=%s\\n' '[0-9]+'$/
const capacityIdentity =
/^ printf 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT=%s\\n' '[a-z][a-z0-9-]{4,28}[a-z0-9]@[a-z0-9-]+\.iam\.gserviceaccount\.com'$/
const rehomeConfig =
/^ printf 'ORCA_RELAY_REHOME_(?:DIRECTOR_SERVICE_ACCOUNT|AUDIENCE)=%s\\n' '[^'\n]+'$/
return script
.split('\n')
.filter(
(line) =>
(preserveCapacity || !capacityAssignment.test(line)) &&
(!stripCapacityIdentity || !capacityIdentity.test(line)) &&
(!stripRehomeConfig || !rehomeConfig.test(line))
(!stripRehomeConfig || !REHOME_CONFIG.test(line))
)
.join('\n')
.replaceAll(image, '<relay-image>')
@@ -213,7 +227,8 @@ function requireDesiredStartupScript(script, config) {
` printf 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT=%s\\n' '${config.capacityServiceAccount}'`
])
}
if (config.mode === 'same-cap-cell') {
const rehomeTrusted = config.mode === 'same-cap-cell' && rehomeProtocol(config) === 1
if (rehomeTrusted) {
expected.push(
[
/^ printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '[^'\n]+'$/,
@@ -225,9 +240,15 @@ function requireDesiredStartupScript(script, config) {
]
)
}
// A protocol-0 cell is not a rehome source, so gaining any rehome trust line is real drift.
const unexpectedRehome =
config.mode === 'same-cap-cell' &&
!rehomeTrusted &&
lines.some((line) => REHOME_CONFIG.test(line))
if (
typeof script !== 'string' ||
relayImage(script) !== config.image ||
unexpectedRehome ||
expected.some(([pattern, line]) => !hasExactSingleAssignment(lines, pattern, line))
) {
throw new Error('cell plan does not contain the reviewed image and capacity')
@@ -450,6 +471,9 @@ export function validateCapacityPlan(plan, config) {
) {
throw new Error('capacity Terraform plan has an invalid service account')
}
if (config.mode === 'same-cap-cell') {
rehomeProtocol(config)
}
if (
config.mode === 'same-cap-cell' &&
(!SERVICE_ACCOUNT_EMAIL.test(config.rehomeDirectorServiceAccount ?? '') ||
@@ -504,7 +528,7 @@ export function validateCapacityPlan(plan, config) {
}
export function main(argv = process.argv.slice(2)) {
const config = parseArguments(argv)
const config = parseCapacityPlanArguments(argv)
const plan = JSON.parse(readFileSync(0, 'utf8'))
process.stdout.write(`${JSON.stringify({ event: 'relay_capacity_plan_verified', ...validateCapacityPlan(plan, config) })}\n`)
}
@@ -1,6 +1,9 @@
import assert from 'node:assert/strict'
import { test } from 'node:test'
import { validateCapacityPlan as validateCapacityPlanRaw } from './validate-relay-capacity-plan.mjs'
import {
parseCapacityPlanArguments,
validateCapacityPlan as validateCapacityPlanRaw
} from './validate-relay-capacity-plan.mjs'
const config = {
cellId: 'staging-gce-c3',
@@ -466,7 +469,8 @@ test('same-cap mode preserves 1000/60 while adding only the reviewed trust confi
image,
rollbackImage,
rehomeDirectorServiceAccount: directorIdentity,
rehomeAudience: audience
rehomeAudience: audience,
regionalRehomeProtocol: '1'
}
assert.deepEqual(
validateCapacityPlan({ resource_changes: [template, manager] }, sameCapConfig),
@@ -644,3 +648,134 @@ test('same-cap mode preserves 1000/60 while adding only the reviewed trust confi
{ mode: 'same-cap-image', changes: 1, changeKind: 'manager-convergence' }
)
})
test('protocol-0 same-cap cells roll without rehome trust lines', () => {
const rollbackImage = `us-docker.pkg.dev/project/relay/image@sha256:${'d'.repeat(64)}`
const image = `us-docker.pkg.dev/project/relay/image@sha256:${'e'.repeat(64)}`
const directorIdentity = 'relay-director@project.iam.gserviceaccount.com'
const audience = 'https://relay.example.com/v1/admin/host-drain'
const startup = ({ selectedImage, trust = false }) => [
` printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '3000'`,
` printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '60'`,
` printf 'ORCA_RELAY_CELL_REGION=%s\\n' 'asia-east2'`,
...(trust ? [
` printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '${directorIdentity}'`,
` printf 'ORCA_RELAY_REHOME_AUDIENCE=%s\\n' '${audience}'`
] : []),
`printf 'ORCA_RELAY_IMAGE_DIGEST=%s\\n' '${selectedImage.split('@')[1]}'`,
`docker pull '${selectedImage}'`,
'docker run --detach \\',
' --name orca-relay \\',
` '${selectedImage}'`
].join('\n')
const template = {
address: 'google_compute_instance_template.relay_gce_cell["production-gce-c27"]',
change: {
actions: ['create', 'delete'],
before: { metadata_startup_script: startup({ selectedImage: rollbackImage }) },
after: { metadata_startup_script: startup({ selectedImage: image }), self_link: null },
after_unknown: { self_link: true }
}
}
const manager = {
address: 'google_compute_instance_group_manager.relay_gce_cell["production-gce-c27"]',
change: {
actions: ['update'],
before: { target_size: 1, version: [{ instance_template: 'old' }] },
after: { target_size: 1, version: [{ instance_template: null }] },
after_unknown: { version: [{ instance_template: true }] }
}
}
const asiaConfig = {
cellId: 'production-gce-c27',
hardCap: 3_000,
unobservedBound: 60,
mode: 'same-cap-cell',
image,
rollbackImage,
rehomeDirectorServiceAccount: directorIdentity,
rehomeAudience: audience,
regionalRehomeProtocol: '0'
}
assert.deepEqual(
validateCapacityPlan({ resource_changes: [template, manager] }, asiaConfig),
{ mode: 'same-cap-cell', changes: 2 }
)
const gainsTrust = structuredClone(template)
gainsTrust.change.after.metadata_startup_script = startup({
selectedImage: image,
trust: true
})
assert.throws(
() => validateCapacityPlan({ resource_changes: [gainsTrust, manager] }, asiaConfig),
/reviewed image and capacity/
)
// Under protocol 1 that same script is the reviewed roll: trust is added, not drift.
assert.deepEqual(
validateCapacityPlan(
{ resource_changes: [gainsTrust, manager] },
{ ...asiaConfig, regionalRehomeProtocol: '1' }
),
{ mode: 'same-cap-cell', changes: 2 }
)
// A protocol-1 cell whose script has no rehome lines is the pre-existing failure, unchanged.
assert.throws(
() => validateCapacityPlan(
{ resource_changes: [template, manager] },
{ ...asiaConfig, regionalRehomeProtocol: '1' }
),
/reviewed image and capacity/
)
for (const protocol of [undefined, '', '2', 'yes']) {
assert.throws(
() => validateCapacityPlan(
{ resource_changes: [template, manager] },
{ ...asiaConfig, regionalRehomeProtocol: protocol }
),
/invalid regional rehome protocol/
)
}
})
test('the rehome protocol argument is required by same-cap-cell mode alone', () => {
const image = `us-docker.pkg.dev/project/relay/image@sha256:${'e'.repeat(64)}`
const rollbackImage = `us-docker.pkg.dev/project/relay/image@sha256:${'d'.repeat(64)}`
const sameCapArguments = (...extra) => [
'--mode', 'same-cap-cell',
'--cell-id', 'production-gce-c27',
'--hard-cap', '3000',
'--unobserved-bound', '60',
'--image', image,
'--rollback-image', rollbackImage,
'--rehome-director-service-account', 'relay-director@project.iam.gserviceaccount.com',
'--rehome-audience', 'https://relay.onorca.dev/v1/admin/host-drain',
...extra
]
assert.equal(
parseCapacityPlanArguments(sameCapArguments('--regional-rehome-protocol', '0'))
.regionalRehomeProtocol,
'0'
)
assert.throws(
() => parseCapacityPlanArguments(sameCapArguments()),
/requires rollback image and rehome trust config/
)
for (const protocol of ['', '2', 'true']) {
assert.throws(
() => parseCapacityPlanArguments(sameCapArguments('--regional-rehome-protocol', protocol)),
/requires rollback image and rehome trust config/
)
}
assert.throws(
() => parseCapacityPlanArguments([
'--mode', 'bootstrap-cell',
'--cell-id', 'staging-gce-c3',
'--hard-cap', '1000',
'--unobserved-bound', '60',
'--image', image,
'--capacity-service-account', 'orca-cap@onorca-cloud.iam.gserviceaccount.com',
'--regional-rehome-protocol', '0'
]),
/applies only to same-cap-cell validation/
)
})
@@ -1,4 +1,5 @@
import { pathToFileURL } from 'node:url'
import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs'
const CAPACITY_PROTOCOL = 2
@@ -378,9 +379,12 @@ export async function verifyCapacityTransition(config, overrides = {}) {
const token = overrides.token ?? process.env.ORCA_RELAY_ADMIN_ID_TOKEN
if (!token || token.length > 8_192) throw new Error('admin identity token is unavailable')
const health = await responseJson(
await fetchImpl(`${config.directorOrigin}/health`, {
signal: AbortSignal.timeout(15_000)
}),
await fetchAdminOnceMore(
fetchImpl,
`${config.directorOrigin}/health`,
{},
{ wait, timeoutMs: 15_000 }
),
'director health'
)
if (health.ok !== true || health.connectionCapacityProtocol !== CAPACITY_PROTOCOL) {
@@ -394,12 +398,16 @@ export async function verifyCapacityTransition(config, overrides = {}) {
lastObservation = { runtimeAvailable: runtime !== null }
if ((runtime === null) === (config.runtime === 'unavailable')) {
const result = await responseJson(
await fetchImpl(`${config.directorOrigin}/v1/admin/cell-status`, {
method: 'POST',
headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' },
body: JSON.stringify({ v: 1, cellId: config.cellId }),
signal: AbortSignal.timeout(30_000)
}),
await fetchAdminOnceMore(
fetchImpl,
`${config.directorOrigin}/v1/admin/cell-status`,
{
method: 'POST',
headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' },
body: JSON.stringify({ v: 1, cellId: config.cellId })
},
{ wait }
),
'cell status'
)
const status = result.status
@@ -1094,3 +1094,77 @@ test('does not retry a rejected cell admin token', async () => {
)
assert.equal(waits, 0)
})
test('retries a transient 503 on the director cell-status read', async () => {
const base = harness()
const statusCalls = []
const result = await verifyCapacityTransition(config, {
token: 'masked-token',
wait: async () => {},
fetch: async (url, options) => {
const path = new URL(url).pathname
if (path !== '/v1/admin/cell-status') return await base(url, options)
statusCalls.push(path)
if (statusCalls.length === 1) return new Response('warming up', { status: 503 })
return await base(url, options)
}
})
assert.equal(statusCalls.length, 2)
assert.equal(result.cellId, config.cellId)
})
test('fails when both director cell-status attempts return a transient 503', async () => {
const base = harness()
let statusCalls = 0
await assert.rejects(
verifyCapacityTransition(config, {
token: 'masked-token',
wait: async () => {},
fetch: async (url, options) => {
const path = new URL(url).pathname
if (path !== '/v1/admin/cell-status') return await base(url, options)
statusCalls += 1
return new Response('warming up', { status: 503 })
}
}),
/cell status returned 503/
)
assert.equal(statusCalls, 2)
})
test('retries a transient 503 on the director health preflight', async () => {
const base = harness()
let healthCalls = 0
const result = await verifyCapacityTransition(config, {
token: 'masked-token',
wait: async () => {},
fetch: async (url, options) => {
const path = new URL(url).pathname
if (path !== '/health') return await base(url, options)
healthCalls += 1
if (healthCalls === 1) return new Response('warming up', { status: 503 })
return await base(url, options)
}
})
assert.equal(healthCalls, 2)
assert.equal(result.cellId, config.cellId)
})
test('fails when both director health attempts return a transient 503', async () => {
const base = harness()
let healthCalls = 0
await assert.rejects(
verifyCapacityTransition(config, {
token: 'masked-token',
wait: async () => {},
fetch: async (url, options) => {
const path = new URL(url).pathname
if (path !== '/health') return await base(url, options)
healthCalls += 1
return new Response('warming up', { status: 503 })
}
}),
/director health returned 503/
)
assert.equal(healthCalls, 2)
})
@@ -0,0 +1,189 @@
# Relay improvement: implementation checklist, lanes, and disruption
Companion to [`relay-improvement-roadmap-2026-09.md`](./relay-improvement-roadmap-2026-09.md) (item numbers
match). This file answers three questions per item: what are the concrete steps, what can run in parallel,
and will a user notice.
## Status as of 2026-09-04 22:30Z
Three buckets. "Merged" means the code is on `main` and nothing in production has changed yet. "Deployed" means users are already getting it. "Awaiting owner" means I will not touch production without a go.
**Deployed to production**
- Auth instance cap 20 + dead-family audit fix (orca-cloud #474) as revision `orca-cloud-auth-00031-tox`.
- Dynamic NAT ports in both regions (stablyai/orca #18693). Zero drops and zero proxy dial errors since.
- Nine alert policies with log metrics: 4 auth (#475), 3 relay Cloud SQL/NAT (#18693), 1 cell process-exit (#18717), all on the relay Slack channel.
**Merged, ships with the next relay cell image roll (Roll 1 carries `519f4914`; Roll 2 needs a fresh image build)**
- Per-cell inventory locks, delta counters, pool `statement_timeout` (#18722). Roll 2.
- Cells dial Cloud SQL with `--private-ip` when configured (#18720). Inert until 2.1 applies.
- Phone shows a clear "sign in on the desktop again" state when the desktop is signed out (#18698).
**Merged, ships with the next auth deploy**
- Refresh rotation grace window (orca-cloud #478). Startup adds one nullable column (brief exclusive lock on `refresh_tokens`).
- Pruning job code (orca-cloud #476) is in the image; the job itself is Terraform-disabled until 1.2.
**Merged, ships with the next desktop release**
- Never replay a refresh token after a timeout; ±10 % jitter on relay lease renewal (#18719).
- Renderer learns when a cloud session is revoked (#18694).
**Merged, not applied**
- Incident dashboard (#18717) blocked behind the runtime-metric label drift (5.x first item).
- Monitor probe fix (#18723) is live in the workflow; the same-cap roll gate has not yet produced a green dry-run since.
**Awaiting owner go (production mutations)**
1. Roll 1 cell image roll (1.1): dry-run gate, then c8 canary, then batches.
2. Auth deploy carrying #478 (3.1): quiet minute for the column add.
3. orca-cloud #477 private IP (2.1): merge arms an instance restart and a one-way door. Recommendation: hold.
4. Runtime-metric `region` label drift (5.x): intentional replacement of 21 metrics, or drop the label.
5. Enable pruning (1.2): first budget 20k rows; needs a Terraform apply.
6. Paging channel for auth alerts (5.2): needs the destination from you.
**Open code follow-ups (no gate, nobody assigned)**
- Monitor summary Markdown does not render `tolerated: true` continuity events (added by #18798); the state artifact has them, the checkpoint table does not.
- Relay container boot races the `cloud-sql-proxy` sidecar: c13's fresh container exited twice (`applyPostgresSchema` connection timeout, 2 s each) before the proxy was listening. Make schema apply wait for the proxy or order the containers.
- `cloud-deploy-relay-production-capacity-job.yml` (~line 416) has the same wave-0 single-shot preflight carve-out that #18778 removes from the same-cap job; its single-evidence path never retries freshness-only failures.
- `cloud/package.json` `test` names every dev-script test file explicitly; an unregistered `*.test.mjs` is silently never run in CI (found by #18769). Needs a glob or a ratchet that fails on an unlisted test file.
- Same-cap job's verify step uses bare `curl --fail-with-body` against the just-rolled cell; one 503 at the LB warm-up edge failed c8 canary #2 (run 33935407461) after the transition verifier had already passed. Needs a bounded retry, same rule as #18723/#18740.
- `verify-mutation` in `cloud-deploy-relay-production.yml`, the multi-target workflow, and the capacity workflow still binds to an exact commit; same exposure #18754 fixed for the same-cap and rehome paths.
- `incident-live-preflight-cli.ts` reports only `source/code` (`active-probe/threshold_max`) with no signal name or observed value, so a failed mutation preflight (c27 recovery #3, run 33986948522) cannot be attributed to an endpoint without an out-of-band probe. Print the signal and observed/threshold pair. Related: the 2 000 ms `endpointLatencyMs` bar is shared by US and Asia cells while Asia /health round trips from a US runner sit at 0.7–1.3 s idle; consider a per-region bar or the p50 of the gate window instead of one shot. Gates #44 and #45 (2026-09-05) both froze on `cell.production-gce-c27.latency_ms` at 2.6–2.7 s with c28 showing the identical tail under operator probes; the bar is now blocking Asia rolls. **Fix: stablyai/orca #18877** (per-region `cellEndpointLatencyMs`, us-central1 2 000 / asia-east2 4 000, plus signal/observed/threshold in preflight messages). Residual: `probeEndpointHealth` in `resource-inventory.ts` still uses the flat 2 000 bar to decide whether to retry after the 10 s readiness-cache wait, so a healthy Asia cell over 2 s costs one extra probe per sample (latency, not verdict); thread the region bar into the retry decision.
- The root oxlint config ignores `cloud/**`, so `check:code-quality:changed` never inspects relay-ops or the cloud dev scripts; typecheck + vitest is the only gate there.
- Monitor bars that froze on non-health today: `directorInstancesMin: 5` with `latest-sum` (one-minute instance recycle), `endpointLatencyMs: 2000` on a US-runner probe to asia-east2, `cloudDataMaxAgeMs: 180000` vs Cloud Monitoring publish lag up to 255 s. Recalibrate with a week of data.
- `parsed()` in `resource-inventory.ts` still returns null on a 200 with a malformed MIG body; a second path to `runtime_power_unknown`.
- Deploy script strips `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` on every release (3.1 first item).
- `assignOnce` placement lock still global (4.1 remainder).
- Region preference (4.2), retries-bar recalibration after a week of Roll 2 data (4.4), pruner `stopReason` alert (1.5).
- Full apps-root apply for 4 unrelated drifts (1.4), from a host with the 1Password account.
## Uplift ranking (reliability gained per unit of effort)
| Rank | Item | Why it ranks here |
|---|---|---|
| 1 | 1.1 cell image roll | Removes the only crash mode we have seen in production. 22 of 23 cells still have it. One afternoon. |
| 2 | 3.1 refresh rotation grace window | Turns the entire "slow auth → mass sign-out" class into a slowdown. One day. |
| 3 | 4.1 inventory lock contention | The floor under every 503 and slow phone accept, every day, not just incidents. One week. |
| — | 2.2 relay/auth database split | **Deferred 2026-09-04** to ~2026-11-01. Biggest structural fix, but the concrete cause is fixed and alerts now page; see roadmap 2.2 for re-open triggers. |
| 4 | 1.2 + 1.3 pruning and reclaim | Defuses the 63 M-row time bomb. Low effort, mostly waiting. |
| 5 | 5.1 + 5.2 crash alert, page a human | Cheapest detection uplift; today's incident ran 4 h unpaged. |
| 6 | 2.1 private IP | Durable version of a fix that already landed (dynamic NAT ports). Do it on the existing instance. |
| 7 | 4.3 + 3.2 desktop hardening | Small, ride the normal desktop release. |
| 8 | 4.2, 4.4, 5.4, 1.4, 1.5 | Housekeeping and quality-of-life. |
## The shared bottleneck: cell rolls
Every change to what runs on a cell (image, proxy flag, env, relay code) needs a same-cap roll: drain →
recreate → verify, one wave at a time, gated by the 15-minute monitor, about an afternoon. Each wave forces
the desktops on that cell to re-dial (c7 canary: 807 controls re-dialed in ~10 s) and phones on those
desktops reconnect on their normal retry. Users see a few seconds of "reconnecting" per wave.
So batch. Two rolls, not five:
- **Roll 1 (now):** current image only (1.1). Do not wait for anything else.
- **Roll 2 (week 2–3):** proxy `--private-ip` (2.1) + relay pool `statement_timeout` (2.3) + lock-contention
fix (4.1), all in one image/template. Prerequisite: 2.1's peering and private IP exist first.
## Lanes (independent; different people can own them)
```
Lane A data plane 1.1 roll ──────────────────► Roll 2 (2.1 flag + 2.3 + 4.1) ──► 4.4 recalibrate
Lane B auth/DB 1.2 enable pruning ──(10 d)──► 1.3 reclaim 3.1 grace window (any time)
Lane C network 2.1 peering + private IP ─────┐ (feeds Roll 2) (2.2 DB split deferred)
Lane D desktop 3.2 no same-token retry, 4.3 lease jitter (any release; wire-compatible)
Lane E observability 1.5, 5.1, 5.2, 5.4 (Terraform only, any time)
Lane F director 4.2 region preference (Cloud Run deploy, any time)
Misc 1.4 full apps-root apply (any time; see its check)
```
Hard dependencies: Roll 2 waits on 2.1's network work; 1.3 waits on 1.2 finishing. Everything else is
independent. (2.2 deferred; if revived, do it after 2.1 so the new instance is private from day one.)
## Disruption summary
| Item | User-visible? | What they see | Mitigation |
|---|---|---|---|
| 1.1 / Roll 2 | **Yes, transient** | Per wave, desktops on that cell reconnect within seconds; phones follow on retry. | Waves gated by the monitor; run in the US night. Already rehearsed on c7. |
| 1.2 pruning | No | Background deletes, 5k rows per batch. | Small first budget; watch `stopReason` and Cloud SQL write throughput. Stop the scheduler if checkpoint alerts fire. |
| 1.3 reclaim | **Depends on tool** | `VACUUM FULL` takes an exclusive lock on `refresh_tokens`: sign-in and refresh block for its duration (minutes to tens of minutes on 16 GB). `pg_repack` holds only brief locks. | Use `pg_repack`. If VACUUM FULL, announce a maintenance window. |
| 1.4 full apps apply | Should be none, **verify** | Terraform will create a new auth revision (env added). Traffic is pinned to `00031-tox` by name, so the new revision should receive 0 %. | Confirm in the plan that no `traffic` change appears. If it does, stop: the Terraform image variable is not the serving image. |
| 1.5, 5.x alerts | No | | |
| 2.1 private IP | **Yes, certain** | Google: "Configuring an existing Cloud SQL instance to use private IP causes the instance to restart, resulting in downtime." No in-place path, HA does not avoid it. Expect 1–2 min DB unavailability: sign-in fails, relay renewals retry. **One-way door**: private IP cannot be disabled and the VPC link cannot be removed once set. The proxy flag change rides Roll 2. | Off-peak; only after Roll 1 (old image dies on a 2 min DB blip). Owner decision required before the foundation apply. |
| 2.2 DB split (deferred) | **Yes, scheduled** | Relay unavailable for the cutover (drain all cells → copy relay tables → flip `DATABASE_URL` → restart). Minutes if rehearsed. Desktops and phones reconnect automatically after. | Rehearse on staging; do it in the US night; announce. |
| 2.3 statement timeout | No beyond Roll 2 | | |
| 3.1 grace window | No | Auth deploys are no-traffic candidate → smoke → promote. | Security trade-off: a stolen token replayed inside the window is served once instead of revoking. 60 s is the usual choice. |
| 3.2, 4.3 desktop | No | Normal app update. | |
| 4.1 lock fix | No beyond Roll 2 | | Verify against real Postgres on 55440 with concurrent probes before shipping. |
| 4.2 region preference | **Minor, Asia users** | Phones that start being placed in Asia reconnect once to a nearer cell. | Roll out behind the existing region-preference flag. |
| 4.4 | No | | |
## Checklists
### 1.1 Cell image roll (Roll 1)
- [x] Confirm fleet is quiet: 15-min monitor dry-run passes. #19 green 23:07:53Z (run 33927238469). Canary then failed the evidence provenance check because main moved during the gate; re-gating with a same-commit chain.
- [x] Confirm director is on 519f4914 and c7 on 85bf6799 (confirmed 2026-09-04 via instance-template census; 20 serving cells still on `5aedbca5`) (`verify` mode of the same-cap workflow).
- [x] Dispatch `cloud-deploy-relay-production-same-cap` waves per the plan in the findings doc; one wave, verify, next. Done 2026-09-05 01:14Z–22:27Z: c8 canary, US batches c9–c10, c13–c16, c19–c26 at protocol 1, then Asia c27 (recovered via `mode=rollback` re-entry after gate freezes on the flat latency bar, fixed by #18877), c28, c29 as single-cell canaries at protocol 0.
- [x] After each wave: the transition verifier passed at migration-only and again at general on every cell (assignments carried, heartbeat fresh, hard cap 3 000); no `container die` fleet-wide across the whole roll. The 4408/1006 burst per wave was not measured separately; the verifier's assignment count before and after each restart is the recovery evidence recorded.
- [x] Record image census in the findings doc. 2026-09-05 22:27Z: all 19 general cells on `519f4914` except c7 on `85bf6799`; existing-only c1–c6, c11, c12 and migration-only c17, c18 untouched on their older images by design. Selector at gen 148.
### 1.2 Enable pruning
- [x] `auth_token_pruner_image` = digest of `orca-cloud-auth-00031-tox` (`343a0915…`; it contains the entrypoint). orca-cloud #479 merged.
- [x] `auth_token_pruner_enabled = true`, `auth_token_pruner_max_rows_per_run = 20000` for the first day (orca-cloud #479).
- [x] Targeted plan asserted 9 create / 0 change / 0 destroy. Applied 2026-09-05 02:06Z.
- [x] Trigger one run by hand; read the summary event. 02:18Z: `time-budget`, 73 batches, 365k scanned, 1 040 deleted (1 021 revoked, 19 expired), no errors. Scan-bound.
- [ ] Raise the budget to the default 200k after a clean day; watch Cloud SQL write MB/s and the checkpoint alert.
- [ ] 1.5: log metric + policy on `stopReason != complete`.
### 1.3 Reclaim
- [ ] Wait for steady-state runs deleting ~0 rows.
- [ ] `pg_repack -t refresh_tokens` off-peak (needs the extension; check `pg_available_extensions`). Not `VACUUM FULL` without a window.
- [ ] Confirm table + index size and `disk/utilization` dropped.
### 1.4 Full apps-root apply
- [ ] Run from CI or a host with the 1Password account (local plan fails on the Cloudflare data source).
- [ ] Plan shows exactly the four known drifts and **no traffic change** on `google_cloud_run_v2_service.auth`.
- [ ] Apply; confirm `status.traffic` still pins `00031-tox` at 100 %.
### 2.1 Private IP (PRs open: orca-cloud #477 foundation, stablyai/orca #18720 relay flag)
- [ ] **Owner decision**: the foundation apply restarts the instance and is irreversible on Google's side. Merging #477 arms the next foundation apply; hold the merge until the window is chosen.
- [ ] Director is out of scope: it uses the Cloud Run built-in connector (managed Google path, not the relay VPC NAT), so it consumed none of the exhausted ports; moving it needs Direct VPC egress + a separate DSN secret. Own PR if ever wanted.
- [ ] Step 7 (`ipv4_enabled=false`) is blocked until humans have IAP/bastion access and the director is moved; it breaks both today.
- [ ] Allocate a `/24` private services range on the relay VPC; `google_service_networking_connection`.
- [ ] Add `ip_configuration.private_network` to `google_sql_database_instance.auth` (foundation root). Plan must show update, not replace.
- [ ] Apply off-peak; expect a possible restart. Watch auth 5xx alert and relay `sqlFailures`.
- [ ] Cell template: proxy args add `--private-ip` (code merged #18720; flag not set). Director: Direct VPC egress or connector, then the same flag. Both ride Roll 2.
- [ ] After Roll 2: NAT `port_usage` for relay gateways drops to ~0; then consider `ipv4_enabled = false` (removes the public IP; breaks the local `cloud-sql-proxy --token` workflow unless it also goes private).
### 2.2 Database split (deferred to ~2026-11-01; checklist kept for when it is revived)
- [ ] New `google_sql_database_instance.relay` (private IP from day one, its own size and flags). Staging first.
- [ ] Relay schema applies cleanly to an empty instance (it does at startup).
- [ ] Rehearsal on staging: drain → `pg_dump` relay tables → restore → flip `relay_database_url` secret → restart director + cells → phones/desktops reconnect. Time it.
- [ ] Production: announce a window; same steps; verify `orca_relay_runtime_metrics` controls recover to pre-cutover count.
- [ ] Update `production-cloud-sql-app-consumers` budget test and both alert policies' `database_id`.
### 2.3 Relay pool statement timeout (merged stablyai/orca #18722; ships Roll 2)
- [x] `statement_timeout` on the relay `pg.Pool` (5 s, env-configurable; schema pool untimed; `57014` retryable), below the control-renewal deadline; DDL on an untimed connection (same pattern as auth #476).
- [x] Postgres test on 55440: a held lock fails the query fast and the bounded retry takes over.
### 3.1 Refresh rotation grace window (orca-cloud #478 merged 2026-09-04; deploy pending owner go)
- [ ] Fix the deploy-script env strip for `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` (pre-existing; found by #478).
- [x] `rotateRefreshToken`: if `rotated_at` within 60 s and not revoked, return the existing successor (idempotent), no revoke, no audit.
- [x] Outside the window or a third presentation: unchanged (revoke + audit).
- [x] Tests: replay inside window returns same successor; outside revokes; concurrent double-present yields one successor.
- [x] Deploy via `deploy-auth-production` (candidate → smoke → promote). Deployed 2026-09-04 23:15Z as `orca-cloud-auth-00035-gos`, cap 20 kept, 0 5xx; `successor_material` column present; sealed successors being written. (candidate → smoke → promote).
### 3.2 / 4.3 Desktop (merged stablyai/orca #18719; ships next desktop release)
- [x] 3.2: on refresh timeout, re-read stored session before retrying; do not re-send a token already rotated locally.
- [x] 4.3: ±10 % jitter on control lease renewal; unit test on the distribution; wire-compatible (server accepts early renewals already).
### 4.1 Lock contention (partial: stablyai/orca #18722 merged; ships Roll 2)
- [x] Replace the global `FOR UPDATE` over `relay_cells` with per-cell row locks; counters delta-only. Remaining: `assignOnce` placement lock is still global (optimistic snapshot follow-up). with per-cell row locks or `pg_advisory_xact_lock(cell)`; counters delta-only.
- [x] Postgres tests on 55440 with concurrent probes (in #18722). Staging load run still owed; `postgres_retries` per hour drops in staging load run.
- [ ] Ships in Roll 2; then 4.4 recalibrates the retries bar from a week of data.
### 4.2 Region preference
- [ ] Director: honor requested region when the preferred region has headroom, else sticky. Behind the existing flag.
- [ ] Measure with `orca_relay_runtime_metrics` region counters before/after.
### 5.x Observability
- [x] **Relay-root runtime-metric drift**: resolved by dropping the `region` label to match live state (stablyai/orca #18734). Applied 2026-09-04 23:11Z: 8 never-applied `control_*` renewal metrics + the incident dashboard created, 0 destroyed, 21 live metrics untouched.
- [x] 5.1 `container die` log metric per cell (`relay_cell_process_exit`, applied 2026-09-04 via #18717), > 3 / 15 min, relay channel.
- [ ] 5.2 Add a paging channel (**needs owner input**: destination) to `auth_alert_notification_channels` for refresh rejections + latency.
- [x] 5.4 One dashboard (applied 2026-09-04 23:11Z): `orca_relay_cloud_sql_wal_checkpoint`, NAT drops, `orca_auth_refresh_401`, summed `controls`.
@@ -0,0 +1,67 @@
# Relay improvement roadmap (written 2026-09-04, after the auth/relay outage)
Owner-facing list of what is left to make the relay more robust, in priority order. Evidence and history
for every item is in [`relay-reconnect-2026-09-findings.md`](./relay-reconnect-2026-09-findings.md)
(Findings 1–13). Everything already landed on 2026-09-04 is listed at the end so this file is complete on
its own.
## 1. Finish what 2026-09-04 started (this week)
| # | Item | Why | How | Size |
|---|---|---|---|---|
| 1.1 | **Roll all 23 cells onto the current relay image** | Every cell still runs the image that exits the whole process on a Postgres connect timeout (Finding 6). The fixed image runs only on the director and c7. Any future DB stall repeats the 200-crashes-in-48h pattern. | `cloud-deploy-relay-production-same-cap` waves, gated by the 15-min monitor. Roll inputs and canary results are in the findings doc ("Roll inputs", "Canary blast radius"). | one afternoon |
| 1.2 | **Enable the refresh_tokens pruning job** (orca-cloud #476, merged, off) | `refresh_tokens` is 63 M rows / 26 GB and grows forever; its size is what turned a slow disk into a sign-out storm (Finding 13). | Build an auth image from main (the 21:04Z deploy already contains the entrypoint: `orca-cloud-auth-00031-tox`, digest `343a0915…`), set `auth_token_pruner_enabled = true` and the image digest in `infra/terraform-apps/environments/production.tfvars`, apply targeted. First run with a small `auth_token_pruner_max_deleted_rows`. Watch the run summary's `stopReason`, not the exit code. ~48 M rows drain in ~10 days at 200k/hour. | 1 hour + 10 days of watching |
| 1.3 | **Reclaim the disk after pruning** | Deletes leave dead tuples; the 16 GB table does not shrink on its own. | `pg_repack` (or `VACUUM FULL` in a maintenance window; it takes an exclusive lock) on `refresh_tokens` off-peak, after 1.2 finishes. | 1 evening |
| 1.4 | **Full Terraform apply of the orca-cloud apps root** | The production plan carries four drifts from other merged work: `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` env on the auth service (#476), a skill-share log exclusion filter change, skill pressure threshold 16→8, an artifacts bucket lifecycle rule. Locally it also fails on the 1Password Cloudflare data source. | Run from CI or a machine with the 1Password account; review the four drifts as ordinary changes. | 30 min |
| 1.5 | **Alert on the pruning job** | A run that only ever times out exits 0 and reads as green. | Log metric on the job's summary event where `stopReason != "complete"`, policy on the relay channel. | 1 hour |
## 2. Remove the shared fate between auth and relay (2.1 and 2.3 this quarter; 2.2 deferred)
| # | Item | Why | How | Size |
|---|---|---|---|---|
| 2.1 | **Private IP for Cloud SQL, `--private-ip` on the cell proxies** (do this on the existing shared instance; do not wait for 2.2) | Cells reach the database's public IP through Cloud NAT. Dynamic port allocation (landed) raised the ceiling from 64 to 4096 ports per VM, but the NAT is still in the path and its logs are still the only place port exhaustion shows up (Finding 11). | Add a private IP to `orca-cloud-auth-db` (foundation root, orca-cloud), peer the relay VPC, switch the proxy flag in the cell template, roll. | 1–2 days |
| 2.2 | **Split the relay database from the auth database** — *DEFERRED 2026-09-04 (owner decision): revisit ~2026-11-01 once pruning is done and there is a month of alert history* | One Cloud SQL instance serves `orca_auth`, `orca_relay`, `orca_push`, `orca_skills`. The auth table's growth stalled the relay for a day (Findings 10, 13). Deferral rationale: the concrete cause is fixed (disk 250 GB, WAL 16 GB, index, pruning), 2.3 + 1.1 turn a future stall into retries, and the checkpoint/disk/headroom alerts now page. Re-open if the checkpoint-loop or connection-headroom alert fires, or a large new auth-side table is planned. | New instance for `orca_relay`; migrate with a short relay drain. Relay state is small so the cutover is minutes. | 1–2 weeks incl. rehearsal on staging |
| 2.3 | **Statement timeouts on the relay pool** (the auth pool got one in #476) | A relay query stuck behind a checkpoint fsync should fail fast and let the bounded retry take over rather than hold a pool slot for seconds. | `statement_timeout` on the relay `pg.Pool` in `cloud/apps/relay`, tuned under the lease renewal deadline. | half a day |
## 3. Make the desktop refresh path forgiving (next 2 weeks)
| # | Item | Why | How | Size |
|---|---|---|---|---|
| 3.1 | **Refresh-token rotation grace window** | The server revokes the whole family the first time a just-rotated token is presented again. On 2026-09-04 that turned a 30 s server slowdown into 21,605 sign-outs. A short window (e.g. 60 s) where the immediately-previous token is still accepted, returning the same new token, is standard practice. | In `apps/auth/src/tokens/refresh-tokens.ts`: accept `rotated_at` within the window, return the successor instead of revoking. Keep true reuse (outside the window, or a third presentation) as revocation. | 1 day incl. tests |
| 3.2 | **Do not retry `/refresh` with the same token on timeout** | Desktop's 30 s `CLOUD_REQUEST_TIMEOUT_MS` expiring is treated like a network error and retried with a token the server may already have rotated. | In `src/main/orca-profiles/profile-cloud-session-refresh.ts`: on timeout, re-read the stored session first, and prefer a longer single attempt for the refresh call specifically. | half a day |
| 3.3 | **Un-revoke is impossible; make sign-out recovery obvious instead** | Server-side un-revoke does not help because the desktop deletes its local token on the 401. Landed: desktop notices immediately (#18694) and the phone says "desktop signed out" (#18698). | Nothing more unless we want a re-auth deep link from the phone to the desktop. | — |
## 4. Chronic relay issues already characterised
| # | Item | Why | How | Size |
|---|---|---|---|---|
| 4.1 | **Cell-inventory lock contention** (partial: PR #18722 narrowed the remaining non-placement sites; `assignOnce` placement lock is the follow-up) | `postgres_retries` is a global `FOR UPDATE` over the 23-row `relay_cells` table with a 1 s `lock_timeout`; it is the floor under every 503 and every slow phone accept (Findings 2, 5; memory `relay-cell-inventory-lock-contention`). | Per-cell row locks or an advisory lock keyed by cell; move capacity counters to delta writes. Verify against real Postgres on 55440. | 1 week |
| 4.2 | **Region preference is mostly inert** | Phones request an Asia cell on ~19 % of attempts and get one ~6 % of the time; the sticky lane wins silently, so Asia users ride the US path more than intended (memory `relay-region-preference-mostly-inert`). | Let a region preference override stickiness when the preferred region has headroom; measure with `orca_relay_runtime_metrics` region counters. | 2–3 days |
| 4.3 | **Desktop lease-rotation waves** | A cell recreate seeds a fleet-wide 1006/4408 reconnect burst ~54 min later, every ~54 min (Finding 3). | Jitter the desktop control lease renewal by ±10 % so the cohort spreads out. | half a day, desktop + wire-compatible |
| 4.4 | **Raise `postgres_retries` gate calibration** | The 300 bar was recalibrated (PR #18580) but should track the post-lock-fix baseline once 4.1 lands. | Re-derive from a week of `orca_relay_postgres_transaction_retry` counts. | 1 hour |
## 5. Observability still missing
| # | Item | Why | How |
|---|---|---|---|
| 5.1 | **Cell crash-rate alert** | 201 process exits in 48 h with no page (Finding 6). | Log metric on `container die` for `resource.type="gce_instance"` relay cells, > 3 per 15 min per cell. In `cloud/infra/terraform/relay-observability.tf`. |
| 5.2 | **Page a person for auth alerts** | Today's four auth policies (orca-cloud #475) route to the relay Slack channel only. A repeat of 2026-09-04 deserves a page. | Add a PagerDuty/phone notification channel to `auth_alert_notification_channels` for refresh rejections and latency. |
| 5.3 | **Pruning job alert** | See 1.5. | |
| 5.4 | **Dashboard that puts the four signals side by side** | Diagnosis took hours because checkpoint state, NAT drops, auth 401 rate, and fleet controls live in four consoles. | One Cloud Monitoring dashboard: `orca_relay_cloud_sql_wal_checkpoint`, NAT `dropped_sent_packets_count`, `orca_auth_refresh_401`, summed `controls`. |
## Landed on 2026-09-04 (for completeness)
- Auth service cap 2 → 20 (service-level manual scaling removed); Cloud SQL disk 49 → 250 GB PD-SSD;
`max_wal_size` 16384; partial index `refresh_tokens_family_unrevoked` built concurrently by hand.
- orca-cloud #474: the above in Terraform + deploy workflow; replayed dead token answers 401 without
re-revoking or re-auditing. Deployed as `orca-cloud-auth-00031-tox` 21:04Z.
- orca-cloud #475: auth alerts (refresh 401 > 100/5 min, 429 > 20/5 min, 5xx > 10/5 min, p99 > 10 s). Applied.
- orca-cloud #476: batched `refresh_tokens` pruner (disabled), auth pool `statement_timeout` 10 s, schema
DDL on an untimed connection.
- stablyai/orca #18693: both relay NATs on dynamic port allocation 64..4096 (applied US 21:01Z, Asia 21:05Z);
alerts for Cloud SQL WAL-checkpoint loop, disk > 70 %, NAT `OUT_OF_RESOURCES` drops. Applied.
- stablyai/orca #18694: desktop learns of a revoked session immediately, panes re-fetch on mount, pairing
notice says "Sign in again to use Orca Relay".
- stablyai/orca #18698: phone shows "Desktop signed out — sign in to Orca on your desktop to reconnect" via
the WebSocket close reason (only additive slot old phones tolerate).
- Director on image 519f4914; c7 on 85bf6799; other 22 cells still on the old image (see 1.1).
+28 -1
View File
@@ -73,6 +73,13 @@ for a committed forward-recovery gate. Durable files default to
gap resets the active window at the next fresh sample and preserves the prior
window evidence. A threshold freeze never clears automatically.
A signal that reads missing or stale may miss up to two consecutive samples
without restarting the window. The sample still counts and is still checked
against every threshold it can read, and each tolerated gap is recorded in
`continuityEvents` with `tolerated: true`. A third consecutive miss of the same
signal, a failed collector, a runner gap, or any threshold breach restarts or
freezes as before.
A production candidate or multi-target mutation must download the exact
dry-run artifact by workflow run ID and attempt. It verifies the artifact
hashes and provenance, requires a green completed 15-minute state no older
@@ -89,7 +96,8 @@ durably marked consumed before mutation and cannot authorize another run.
| Signal | Freeze condition |
| --- | ---: |
| Active probe age | over 60 seconds |
| Cloud/log data age | over 180 seconds |
| Cloud Monitoring data age | over 330 seconds |
| Relay log and director admin data age | over 180 seconds |
| Cell heartbeat age | over 45 seconds |
| Endpoint latency | over 2,000 ms |
| Cloud SQL CPU | over 80% |
@@ -155,6 +163,25 @@ heartbeats, and matching live admission.
separate it from today's baseline; the exhausted-retry bar (incident peak
467 vs bar 300), director concurrency, and the pool bars carry that role.
Re-tighten after the fleet is on the 500 ms lock wait.
- Raised the Cloud Monitoring freshness bar from 180 s to 330 s and let a
freshness-only failure miss up to two consecutive samples without restarting
the window (2026-09-05). Basis: Google's metric list documents Cloud Run
`request_count`, `container/instance_count`, `container/cpu/utilizations`,
`container/memory/utilizations` and `container/max_request_concurrencies` as
"Sampled every 60 seconds. After sampling, data is not visible for up to 120
seconds", and Cloud SQL `database/cpu/utilization`,
`database/memory/utilization`, `database/postgresql/num_backends`,
`database/postgresql/backends_in_wait` and `database/postgresql/deadlock_count`
as "up to 165 seconds", so the newest visible point is up to 180 s and 225 s
old respectively. Window-sum signals age further: `observedAt` is the newest
point in the 5-minute query window, so a label series that stops emitting
reads as 300 s old while its summed value is complete. The old bar sat under
all three. Production on 2026-09-04/05 restarted healthy 15-minute windows at
181 s and 255 s (`auth.errors`, run 33928912676) and at 189 s
(`cloud_sql.lock_waits`, run 33944873727), and the last of those then blew the
25-minute lineage cap at 1 500 004 ms, so a green fleet produced no verdict.
The director admin bar stays at 180 s and the nonzero lock-wait carry window
stays at 180 s; both publish on our own cadence.
- Recalibrated the exhausted-PostgreSQL-retry freeze from 0 to 300 per five
minutes (2026-09-04). Basis: #18521 cut the request-path cell-inventory
lock wait from the 1 s pool `lock_timeout` to 500 ms, so contended waiters
File diff suppressed because it is too large Load Diff
+154
View File
@@ -0,0 +1,154 @@
# Relay Roll 2 and close-out plan (2026-09-05)
Owner-approved scope 2026-09-05: finish the relay reliability work with one more cell image roll,
deferring the Cloud SQL private-IP move (2.1, orca-cloud #477) to a separate owner decision. Roll 1
is complete (see `relay-reconnect-2026-09-findings.md`, "Roll 1 complete"); every serving cell runs
`519f4914` except c7 on `85bf6799`.
Estimate: about two working days of effort over one week of calendar time. The cell roll itself is
6 to 7 hours of mostly unattended wall clock, run in the US night.
## Phase 0. Land the code (half a day, no production change)
### 0a. Split PR #18565
The branch mixes three relay/mobile/desktop fixes with the operator record. Split so the record
lands regardless of how the code review goes.
- **Docs PR** (new branch off main): `relay-reconnect-2026-09-findings.md`,
`relay-improvement-checklist-2026-09.md`, `relay-improvement-roadmap-2026-09.md`, this file.
Docs only, merge on CI green.
- **Code PR** (rebase #18565 onto main, resolve two conflicts):
- `cloud/apps/relay/src/host-session-registry.ts`: conflict with #18698 (signed-out signal).
Keep both; the accept-abandonment and lease changes are orthogonal to the signed-out path.
- `src/main/runtime/relay/relay-origin-pool.ts`: **drop this branch's version**. #18719 already
merged the desktop early-window jitter (1 to 6 min). Also drop
`relay-session-broker.test.ts` additions that only exercise the dropped change.
- Keep: relay accept abandonment (`orca_relay_client_accept_abandoned` event), relay-side lease
jitter, mobile direct-probe fail-fast, and their tests.
### 0b. Lengthen the control lease (same code PR)
In `cloud/apps/relay/src/host-session-registry.ts`:
```
CONTROL_LEASE_MS = 6 * 60 * 60 * 1000 // was 55 min
CONTROL_LEASE_JITTER_MS = 30 * 60 * 1000 // was 5 min
```
Why 6 h: the lease bounds how long a host stays on a cell after a missed drain and is the only
passive rebalancing; 6 h keeps both and cuts control-activation traffic on the inventory lock by
about 6x. Nothing else depends on it: the relay JWT (5 min) is refreshed by the desktop on its own
schedule and liveness is the 75 s silence watchdog. Wire-safe: the relay sends `leaseExpiresAt` in
the hello ack and old desktops schedule from that value.
Update the comment above the constants and the three assertions in
`host-session-client-accept.test.ts` that pin the lease arithmetic. Check that nothing in
`cloud/apps/relay-ops` or the monitor thresholds assumes a 55 min rotation period (grep
`55`, `CONTROL_LEASE`, `rotation`).
### 0c. Review and merge
Review rounds per the standing process (Opus review, then Codex pass). Merge order: docs PR first
(no dependency), then the code PR. Record the merge SHA of the code PR; that is the Roll 2 image
source.
## Phase 1. Build and stage the image (half a day)
Roll 2 image = code PR merge SHA. It carries, relative to `519f4914`:
| Change | PR | Effect |
|---|---|---|
| Per-cell inventory locks, delta counters | #18722 | Removes the global `relay_cells FOR UPDATE` behind the phone accept hang |
| Relay pool `statement_timeout` 5 s | #18722 | A relay query can no longer hang a cell |
| Accept abandonment | #18565 | Cell stops finishing accepts for phones that already closed |
| Control lease 6 h ± 30 min | #18565 | Fewer, spread-out rebinds |
| `--private-ip` proxy flag support | #18720 | Code only; flag stays unset until 2.1 |
Steps, in order (from the findings doc's post-merge dispatch plan):
1. `gh workflow run cloud-publish-relay-production.yml --ref main -f mode=publish`. Resolve the
digest by tag, not from the log:
`gcloud artifacts docker images describe us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay:sha-<merge-sha> --format='value(image_summary.digest)'`.
2. Staging: `cloud-deploy-relay-staging.yml` with the new digest; paired phone plus desktop smoke
(connect, background, reconnect). Confirm `orca_relay_client_accept_abandoned` appears only when
a client closes early, and that `sqlLatencyMsMax` no longer pins at the lock timeout.
3. Director: `cloud-deploy-relay-production-director.yml -f image-digest=<new>
-f regional-placement-mode=preserve -f prune-incompatible-revisions=false
-f expected-rehome-generation=12 -f bootstrap-runtime-identity=false
-f predecessor-image-digest=<serving digest>`. Blue/green; prior revision stays as rollback.
Watch director `orca_relay_postgres_transaction_retry` per minute before and after. The director
goes first so the per-cell locks are live before any cell restart burst.
4. Same-cap `verify` mode against c7 with target=<new>, rollback=`519f4914`. Read-only.
Go/no-go for Phase 2: director serving the new image for at least 30 min, retries per minute at or
below the pre-deploy baseline, no `container die`, no auth 5xx.
## Phase 2. Roll the cells (one US night, mostly unattended)
Same machinery as Roll 1: `cloud-monitor-relay-production.yml` dry-run gate, then
`cloud-deploy-relay-production-same-cap.yml`. Cells roll one at a time by design (exact selector
assertions, single Terraform state, and one cell's ~1.2k-host reconnect burst per restart). Do not
add parallelism for this roll.
Inputs: target=<new digest>, rollback=`519f4914` (c7: rollback=`85bf6799`). Selector membership is
unchanged from the end of Roll 1 (gen 148; existing-only c1–c6, c11, c12; migration-only c17, c18).
Order:
1. **c7 canary** (`canary-apply`, protocol 1). c7 is the rehearsal cell and the only one not on
`519f4914`.
2. **c8 canary**, then **batch c9, c10, c13, c14**.
3. **c15 canary**, then **batch c16, c19, c20, c21**.
4. **c22 canary**, then **batch c23, c24, c25, c26**.
5. **Asia c27, c28, c29** as three single canaries at protocol 0 (`PROTO=0`). Batch mode cannot
take Asia cells yet and needs at least two cells.
Each batch needs a same-commit canary authority; each wave needs a fresh 15 min gate. Use the
chain script pattern from Roll 1 (wait gate green, check trusted-path ancestry, dispatch within 5 min,
log `CANARY <run> <status>`) under `caffeinate -i`. Budget: 11 to 13 min per cell plus 15 min per gate,
about 6 to 7 h total.
Per wave checks (same as Roll 1): transition verifier passes at migration-only and again at general
with assignments carried; no `container die` fleet-wide; selector generation advances by exactly 2
per cell. After the Asia cells: image census from MIG templates; every general cell on the new digest.
Failure handling: a failed canary re-enters through `mode=rollback` with rollback-digest = desired
image (Roll 1 c27 pattern). A gate freeze on an Asia latency probe despite the 4 000 ms bar is a
stop-and-investigate, not a retry. Monitor-side freezes (freshness, continuity deadline) re-gate
after a 2 min back-off; the chain does this on its own.
Record every gate and wave in the findings doc as in Roll 1.
## Phase 3. After the roll (spread over the following week)
- **4.4 Recalibrate the retries bar.** After one week of `orca_relay_postgres_transaction_retry`
on the new image, re-derive the `postgres_retries` monitor threshold from the new baseline
(PR against `cloud/apps/relay-ops/src/incident-monitor.ts` thresholds). About 2 h.
- **1.2 Pruner budget.** Raise `auth_token_pruner_max_rows_per_run` to the default 200k after a
clean day; watch Cloud SQL write MB/s and the checkpoint alert. Then **1.5** log metric plus
policy on `stopReason != complete`.
- **1.3 Reclaim.** Once pruner runs delete ~0 rows: `pg_repack -t refresh_tokens` off-peak (check
`pg_available_extensions` first; not `VACUUM FULL`). Confirm table, index, and `disk/utilization`
dropped.
- **Monitor residuals** already in the checklist: `probeEndpointHealth` retry decision still uses the
flat 2 000 ms bar; operator protocol unbound for Asia; `probe-relay-rehome-trust` regex.
- Update the checklist status header; tick 2.3, 4.1, 4.3 relay-side as deployed.
## Deferred, owner decision required
- **2.1 Private IP** (orca-cloud #477). One-way door with a Cloud SQL restart. When chosen: apply the
foundation off-peak, then a template-only change that sets the `--private-ip` proxy flag. That is
another cell roll unless bundled with a future image.
- **5.2 Paging channel** for auth alerts: needs a destination.
- **Parallel cell rolls** (2 or 3 at a time): about 1.5 days (relax exact-selector assertions to
"exact except in-flight", single coordinator Terraform apply, parallel job shape, tests). Only
worth building if more image rolls are planned after Roll 2, and only once the per-cell locks are
live so a multi-cell reconnect burst is safe.
- **2.2 Database split**: deferred to ~2026-11-01.
## Not in this plan
Desktop and mobile changes already merged (#18719 desktop early-window jitter and no same-token
refresh retry; #18565 mobile fail-fast once merged) ship with the next desktop and mobile releases
on their own schedules. No relay action needed.
+1
View File
@@ -242,6 +242,7 @@ resource "google_compute_instance_template" "relay_gce_cell" {
artifact_registry_host = "${var.region}-docker.pkg.dev"
relay_image = each.value.image
cloud_sql_proxy_image = var.relay_gce_cloud_sql_proxy_image
cloud_sql_private_ip = var.relay_cloud_sql_private_ip
# Keep cell-only plans independent from unrelated database configuration drift.
cloud_sql_connection_name = local.relay_database_connection_name
})
@@ -42,6 +42,12 @@ resource "google_compute_router_nat" "relay_gce" {
router = google_compute_router.relay_gce[0].name
nat_ip_allocate_option = "AUTO_ONLY"
source_subnetwork_ip_ranges_to_nat = "LIST_OF_SUBNETWORKS"
# Cells reach Cloud SQL's public IP through this NAT. The static default of 64 ports per VM
# filled during the 2026-09-04 incident and every cell's proxy dial timed out at once.
enable_dynamic_port_allocation = true
enable_endpoint_independent_mapping = false
min_ports_per_vm = 64
max_ports_per_vm = 4096
subnetwork {
name = google_compute_subnetwork.relay_gce[0].id
@@ -85,6 +91,12 @@ resource "google_compute_router_nat" "relay_gce_additional" {
router = google_compute_router.relay_gce_additional[each.key].name
nat_ip_allocate_option = "AUTO_ONLY"
source_subnetwork_ip_ranges_to_nat = "LIST_OF_SUBNETWORKS"
# Cells reach Cloud SQL's public IP through this NAT. The static default of 64 ports per VM
# filled during the 2026-09-04 incident and every cell's proxy dial timed out at once.
enable_dynamic_port_allocation = true
enable_endpoint_independent_mapping = false
min_ports_per_vm = 64
max_ports_per_vm = 4096
subnetwork {
name = google_compute_subnetwork.relay_gce_additional[each.key].id
@@ -109,6 +109,9 @@ docker run --detach \
--user 0:0 \
--volume "$${cloudsql_dir}:/cloudsql" \
'${cloud_sql_proxy_image}' \
%{ if cloud_sql_private_ip ~}
--private-ip \
%{ endif ~}
--unix-socket=/cloudsql \
'${cloud_sql_connection_name}'
+289 -7
View File
@@ -37,6 +37,16 @@ locals {
description = "Relay PostgreSQL transactions that exhausted bounded retry."
filter = "((resource.type=\"cloud_run_revision\" AND (${local.relay_service_log_filter})) OR resource.type=\"gce_instance\") AND jsonPayload.event=\"orca_relay_postgres_transaction_exhausted\""
}
cell_process_exit = {
# The docker event stream is the only per-exit line: the relay's own crash footer only
# appears for unhandled rejections, and `container start` also counts healthy first boots.
description = "Relay cell container exits, one Docker `container die` event per process exit."
filter = "resource.type=\"gce_instance\" AND logName=\"projects/${var.project_id}/logs/cos_system\" AND jsonPayload.SYSLOG_IDENTIFIER=\"docker\" AND jsonPayload.MESSAGE:\"container die\" AND jsonPayload.MESSAGE:\"name=orca-relay)\""
}
cloud_sql_wal_checkpoint = {
description = "Cloud SQL checkpoints triggered by WAL volume instead of the timed schedule; a sustained run is the fsync loop that stalled every relay process at once on 2026-09-04."
filter = "resource.type=\"cloudsql_database\" AND resource.labels.database_id=\"${var.project_id}:${local.relay_database_instance_name}\" AND textPayload:\"checkpoint starting: wal\""
}
}
relay_runtime_metrics = {
@@ -201,7 +211,8 @@ resource "google_logging_metric" "relay_snapshot" {
label_extractors = {
role = "EXTRACT(jsonPayload.role)"
cell_id = "EXTRACT(jsonPayload.cellId)"
region = "EXTRACT(jsonPayload.region)"
# No region label: adding one replaces all 21 live metrics (label change = delete+create),
# which resets history and blanks the relay alert policies during the swap.
}
metric_descriptor {
@@ -220,12 +231,6 @@ resource "google_logging_metric" "relay_snapshot" {
value_type = "STRING"
description = "Durable relay cell identifier."
}
labels {
key = "region"
value_type = "STRING"
description = "Coarse Relay region."
}
}
bucket_options {
@@ -523,3 +528,280 @@ resource "google_monitoring_alert_policy" "relay_cloud_sql_backends" {
mime_type = "text/markdown"
}
}
resource "google_monitoring_alert_policy" "relay_cloud_sql_checkpoint_loop" {
project = var.project_id
display_name = "Orca Relay: Cloud SQL checkpoint loop"
combiner = "OR"
enabled = true
notification_channels = var.relay_alert_notification_channels
conditions {
display_name = "WAL-triggered checkpoints above 3 in 5 minutes"
condition_threshold {
filter = "resource.type=\"cloudsql_database\" AND metric.type=\"logging.googleapis.com/user/orca_relay_cloud_sql_wal_checkpoint\""
comparison = "COMPARISON_GT"
threshold_value = 3
duration = "300s"
aggregations {
alignment_period = "300s"
per_series_aligner = "ALIGN_SUM"
cross_series_reducer = "REDUCE_SUM"
}
trigger {
count = 1
}
}
}
documentation {
content = "Healthy operation is one timed checkpoint every 5 minutes. Repeated `checkpoint starting: wal` lines mean WAL is outrunning `max_wal_size` and every checkpoint fsync stalls all relay SQL for seconds. Check `checkpoint complete` sync= times and disk write throughput against the PD-SSD ceiling; the fix is disk size and `max_wal_size` in the Terraform root that owns the instance (orca-cloud `infra/terraform-foundation`)."
mime_type = "text/markdown"
}
depends_on = [google_logging_metric.relay_incident]
}
resource "google_monitoring_alert_policy" "relay_cloud_sql_disk" {
project = var.project_id
display_name = "Orca Relay: Cloud SQL disk utilization"
combiner = "OR"
enabled = true
notification_channels = var.relay_alert_notification_channels
conditions {
display_name = "Cloud SQL disk above 70%"
condition_threshold {
filter = "resource.type=\"cloudsql_database\" AND resource.label.\"database_id\"=\"${var.project_id}:${local.relay_database_instance_name}\" AND metric.type=\"cloudsql.googleapis.com/database/disk/utilization\""
comparison = "COMPARISON_GT"
threshold_value = 0.7
duration = "600s"
aggregations {
alignment_period = "300s"
per_series_aligner = "ALIGN_MAX"
}
trigger {
count = 1
}
}
}
documentation {
content = "The shared auth/relay Cloud SQL disk is filling. `refresh_tokens` is the largest table and grows without pruning; grow the disk (IOPS scale with size) before it reaches the WAL checkpoint loop, and prune revoked token rows."
mime_type = "text/markdown"
}
}
resource "google_monitoring_alert_policy" "relay_cloud_nat_port_drops" {
count = local.relay_gce_configured ? 1 : 0
project = var.project_id
display_name = "Orca Relay: Cloud NAT port exhaustion"
combiner = "OR"
enabled = true
notification_channels = var.relay_alert_notification_channels
conditions {
display_name = "NAT packets dropped for lack of ports"
condition_threshold {
filter = "resource.type=\"nat_gateway\" AND resource.label.\"gateway_name\"=monitoring.regex.full_match(\"${local.relay_gce_name}(-.*)?\") AND metric.type=\"router.googleapis.com/nat/dropped_sent_packets_count\" AND metric.label.\"reason\"=\"OUT_OF_RESOURCES\""
comparison = "COMPARISON_GT"
threshold_value = 0
duration = "120s"
aggregations {
alignment_period = "60s"
per_series_aligner = "ALIGN_SUM"
cross_series_reducer = "REDUCE_SUM"
group_by_fields = ["resource.label.\"gateway_name\""]
}
trigger {
count = 1
}
}
}
documentation {
content = "Relay cells reach Cloud SQL's public IP through this NAT. Port exhaustion makes every cell's Cloud SQL Auth Proxy dial time out at once, which reads as a fleet-wide SQL stall with a healthy database. Check `nat/port_usage` per VM and raise `max_ports_per_vm` in `relay-gce-foundation.tf`, or move the database to a private IP."
mime_type = "text/markdown"
}
}
resource "google_monitoring_alert_policy" "relay_cell_process_exit" {
project = var.project_id
display_name = "Orca Relay: cell process exits"
combiner = "OR"
enabled = true
notification_channels = var.relay_alert_notification_channels
conditions {
display_name = "Cell container exits above 3 in 15 minutes"
condition_threshold {
filter = "resource.type=\"gce_instance\" AND metric.type=\"logging.googleapis.com/user/orca_relay_cell_process_exit\""
comparison = "COMPARISON_GT"
threshold_value = 3
duration = "0s"
aggregations {
alignment_period = "900s"
per_series_aligner = "ALIGN_SUM"
cross_series_reducer = "REDUCE_SUM"
group_by_fields = ["resource.label.\"instance_id\""]
}
trigger {
count = 1
}
}
}
documentation {
content = "A Relay GCE cell restarted its container more than three times in 15 minutes. Each exit drops every host and phone on that cell, and 201 exits went unpaged over 48 h on 2026-09-04. The instance hostname is `relay-<cell>-<suffix>`; read `jsonPayload.MESSAGE` on `cos_system` for the exit code and the container's own stderr for the stack before blaming MIG autoheal or load. A same-capacity roll is the remedy when the running image is behind."
mime_type = "text/markdown"
}
depends_on = [google_logging_metric.relay_incident]
}
# Why: the four signals that had to be assembled by hand during the 2026-09-04 incident.
resource "google_monitoring_dashboard" "relay_incident" {
project = var.project_id
dashboard_json = jsonencode({
displayName = "Orca Relay: incident overview"
mosaicLayout = {
columns = 12
tiles = [
{
xPos = 0
yPos = 0
width = 3
height = 4
widget = {
title = "Cloud SQL WAL checkpoints"
xyChart = {
dataSets = [{
plotType = "LINE"
targetAxis = "Y1"
timeSeriesQuery = {
timeSeriesFilter = {
filter = "metric.type=\"logging.googleapis.com/user/orca_relay_cloud_sql_wal_checkpoint\" AND resource.type=\"cloudsql_database\""
aggregation = {
alignmentPeriod = "300s"
perSeriesAligner = "ALIGN_SUM"
crossSeriesReducer = "REDUCE_SUM"
}
}
}
}]
yAxis = {
label = "checkpoints"
scale = "LINEAR"
}
}
}
},
{
xPos = 3
yPos = 0
width = 3
height = 4
widget = {
title = "Cloud NAT dropped packets"
xyChart = {
dataSets = [{
plotType = "LINE"
targetAxis = "Y1"
timeSeriesQuery = {
timeSeriesFilter = {
filter = "metric.type=\"router.googleapis.com/nat/dropped_sent_packets_count\" AND resource.type=\"nat_gateway\" AND resource.label.\"gateway_name\"=monitoring.regex.full_match(\"${local.relay_gce_name}(-.*)?\")"
aggregation = {
alignmentPeriod = "60s"
perSeriesAligner = "ALIGN_SUM"
crossSeriesReducer = "REDUCE_SUM"
groupByFields = ["resource.label.\"gateway_name\"", "metric.label.\"reason\""]
}
}
}
}]
yAxis = {
label = "packets"
scale = "LINEAR"
}
}
}
},
{
xPos = 6
yPos = 0
width = 3
height = 4
widget = {
title = "Auth refresh 401s"
xyChart = {
dataSets = [{
plotType = "LINE"
targetAxis = "Y1"
timeSeriesQuery = {
timeSeriesFilter = {
filter = "metric.type=\"logging.googleapis.com/user/orca_auth_refresh_401\""
aggregation = {
alignmentPeriod = "300s"
perSeriesAligner = "ALIGN_SUM"
crossSeriesReducer = "REDUCE_SUM"
}
}
}
}]
yAxis = {
label = "rejections"
scale = "LINEAR"
}
}
}
},
{
xPos = 9
yPos = 0
width = 3
height = 4
widget = {
title = "Standing desktop controls (fleet sum)"
xyChart = {
dataSets = [{
plotType = "LINE"
targetAxis = "Y1"
timeSeriesQuery = {
timeSeriesFilter = {
# ALIGN_MEAN, not ALIGN_SUM: each process reports its standing control count once per interval.
filter = "metric.type=\"logging.googleapis.com/user/orca_relay_controls\""
aggregation = {
alignmentPeriod = "300s"
perSeriesAligner = "ALIGN_MEAN"
crossSeriesReducer = "REDUCE_SUM"
}
}
}
}]
yAxis = {
label = "controls"
scale = "LINEAR"
}
}
}
}
]
}
})
depends_on = [google_logging_metric.relay_incident, google_logging_metric.relay_snapshot]
}
+6
View File
@@ -468,6 +468,12 @@ variable "relay_gce_fenced_cells" {
default = []
}
variable "relay_cloud_sql_private_ip" {
type = bool
description = "Dial Cloud SQL over its private IP inside this VPC instead of its public IP through Cloud NAT. Requires the foundation root's private services access peering to be applied first; a cell that cannot reach the private IP never becomes ready."
default = false
}
variable "relay_gce_cloud_sql_proxy_image" {
type = string
description = "Digest-pinned Cloud SQL Auth Proxy image used by private relay workers."
+1 -1
View File
@@ -21,7 +21,7 @@
"load:relay:recovery-gate": "node dev/scripts/run-relay-recovery-wave-gate.mjs",
"ops:relay": "pnpm --filter @orca-cloud/relay-ops dev",
"pretest": "node --test dev/scripts/capture-terraform-plan-baseline.test.mjs dev/scripts/operate-relay-asia-admission.test.mjs dev/scripts/prepare-relay-asia-director-cells.test.mjs dev/scripts/prepare-relay-asia-topology-input.test.mjs dev/scripts/production-cloud-sql-rollout-lock.test.mjs dev/scripts/read-relay-serving-regional-placement-version.test.mjs dev/scripts/relay-asia-admission-workflow.test.mjs dev/scripts/relay-asia-rollout-evidence.test.mjs dev/scripts/relay-asia-topology-workflow.test.mjs dev/scripts/relay-cloud-sql-connection-budget.test.mjs dev/scripts/relay-load-reader-evidence.test.mjs dev/scripts/relay-staging-deploy-identity.test.mjs dev/scripts/sanitize-relay-asia-admission-result.test.mjs dev/scripts/terraform-root-partition.test.mjs dev/scripts/validate-relay-asia-topology-plan.test.mjs ../.github/actions/cloud-sql-rollout-lease/action-contract.test.mjs ../.github/actions/cloud-sql-rollout-lease/storage-lease.test.mjs",
"test": "pnpm -r test && node --test dev/scripts/classify-relay-production-capacity-director.test.mjs dev/scripts/classify-relay-staging-bootstrap.test.mjs dev/scripts/deploy-relay-blue-green.test.mjs dev/scripts/deploy-relay-gce-candidate.test.mjs dev/scripts/deploy-relay-gce-multi-target.test.mjs dev/scripts/github-smoke-token.test.mjs dev/scripts/infra.test.mjs dev/scripts/operate-relay-regional-rehome.test.mjs dev/scripts/power-staging-relay.test.mjs dev/scripts/prepare-relay-capacity-canary.test.mjs dev/scripts/prepare-relay-production-capacity-canary.test.mjs dev/scripts/probe-relay-legacy-admission.test.mjs dev/scripts/probe-relay-rehome-trust.test.mjs dev/scripts/production-cell-image-digest-consistency.test.mjs dev/scripts/read-relay-production-capacity-identity.test.mjs dev/scripts/relay-admission-selector.test.mjs dev/scripts/relay-gce-terraform-fence.test.mjs dev/scripts/relay-load-connection-failure.test.mjs dev/scripts/relay-load-control-peer.test.mjs dev/scripts/relay-load-director-capacity-gate.test.mjs dev/scripts/relay-load-model.test.mjs dev/scripts/relay-load-phase-barrier.test.mjs dev/scripts/relay-load-placement-boundary.test.mjs dev/scripts/relay-load-profile.test.mjs dev/scripts/relay-load-rebind-boundary.test.mjs dev/scripts/relay-load-region-behavior.test.mjs dev/scripts/relay-load-request-unit-boundary.test.mjs dev/scripts/relay-load-run-lifecycle.test.mjs dev/scripts/relay-monitor-evidence.test.mjs dev/scripts/relay-production-capacity-wave.test.mjs dev/scripts/relay-production-capacity-workflow.test.mjs dev/scripts/relay-production-identity-boundaries.test.mjs dev/scripts/relay-production-same-cap-wave.test.mjs dev/scripts/relay-public-workflow-contract.test.mjs dev/scripts/relay-recovery-wave-gate.test.mjs dev/scripts/relay-region-observation-evidence.test.mjs dev/scripts/relay-regional-rehome-workflow.test.mjs dev/scripts/relay-rehome-aggregate-evidence.test.mjs dev/scripts/relay-repository.test.mjs dev/scripts/relay-staging-c4-refresh-workflow.test.mjs dev/scripts/relay-staging-capacity-identity.test.mjs dev/scripts/staging-relay-apply-guard.test.mjs dev/scripts/validate-relay-capacity-plan.test.mjs dev/scripts/verify-relay-capacity-transition.test.mjs dev/scripts/verify-relay-legacy-bootstrap.test.mjs dev/scripts/workload-identity-attribute-conditions.test.mjs",
"test": "pnpm -r test && node --test dev/scripts/classify-relay-production-capacity-director.test.mjs dev/scripts/classify-relay-staging-bootstrap.test.mjs dev/scripts/deploy-relay-blue-green.test.mjs dev/scripts/deploy-relay-gce-candidate.test.mjs dev/scripts/deploy-relay-gce-multi-target.test.mjs dev/scripts/github-smoke-token.test.mjs dev/scripts/infra.test.mjs dev/scripts/operate-relay-regional-rehome.test.mjs dev/scripts/power-staging-relay.test.mjs dev/scripts/prepare-relay-capacity-canary.test.mjs dev/scripts/prepare-relay-production-capacity-canary.test.mjs dev/scripts/probe-relay-legacy-admission.test.mjs dev/scripts/probe-relay-rehome-trust.test.mjs dev/scripts/production-cell-image-digest-consistency.test.mjs dev/scripts/read-relay-production-capacity-identity.test.mjs dev/scripts/relay-admin-endpoint-retry-workflow.test.mjs dev/scripts/relay-admin-transient-retry.test.mjs dev/scripts/relay-admission-selector.test.mjs dev/scripts/relay-gce-terraform-fence.test.mjs dev/scripts/relay-load-connection-failure.test.mjs dev/scripts/relay-load-control-peer.test.mjs dev/scripts/relay-load-director-capacity-gate.test.mjs dev/scripts/relay-load-model.test.mjs dev/scripts/relay-load-phase-barrier.test.mjs dev/scripts/relay-load-placement-boundary.test.mjs dev/scripts/relay-load-profile.test.mjs dev/scripts/relay-load-rebind-boundary.test.mjs dev/scripts/relay-load-region-behavior.test.mjs dev/scripts/relay-load-request-unit-boundary.test.mjs dev/scripts/relay-load-run-lifecycle.test.mjs dev/scripts/relay-monitor-evidence.test.mjs dev/scripts/relay-production-capacity-wave.test.mjs dev/scripts/relay-production-capacity-workflow.test.mjs dev/scripts/relay-production-identity-boundaries.test.mjs dev/scripts/relay-production-same-cap-wave.test.mjs dev/scripts/relay-public-workflow-contract.test.mjs dev/scripts/relay-recovery-wave-gate.test.mjs dev/scripts/relay-region-observation-evidence.test.mjs dev/scripts/relay-regional-rehome-workflow.test.mjs dev/scripts/relay-rehome-aggregate-evidence.test.mjs dev/scripts/relay-repository.test.mjs dev/scripts/relay-same-cap-script-census.test.mjs dev/scripts/relay-staging-c4-refresh-workflow.test.mjs dev/scripts/relay-staging-capacity-identity.test.mjs dev/scripts/staging-relay-apply-guard.test.mjs dev/scripts/validate-relay-capacity-plan.test.mjs dev/scripts/verify-relay-capacity-transition.test.mjs dev/scripts/verify-relay-legacy-bootstrap.test.mjs dev/scripts/workload-identity-attribute-conditions.test.mjs",
"typecheck": "pnpm -r typecheck"
},
"devDependencies": {
@@ -0,0 +1,18 @@
// Mirror of src/shared/relay-host-close-reason.ts in the Orca app repo half.
// A host control socket may close with one of these as its WebSocket close
// reason; the cell records it so a later phone rejection can name the cause.
// Anything else (including the empty reason of an abrupt 1006) means "unknown",
// which is what every peer that predates this file sends.
export const RELAY_HOST_CLOSE_REASON = {
SIGNED_OUT: 'signed-out'
} as const
export type RelayHostCloseReason =
(typeof RELAY_HOST_CLOSE_REASON)[keyof typeof RELAY_HOST_CLOSE_REASON]
const REASONS: readonly string[] = Object.values(RELAY_HOST_CLOSE_REASON)
export function relayHostCloseReasonFrom(value: unknown): RelayHostCloseReason | null {
const text = typeof value === 'string' ? value : (value?.toString() ?? '')
return REASONS.includes(text) ? (text as RelayHostCloseReason) : null
}
@@ -5,6 +5,7 @@ export * from './control-messages.js'
export * from './control-continuity.js'
export * from './credential-messages.js'
export * from './director-messages.js'
export * from './host-close-reason.js'
export * from './host-proof-transcript.js'
export * from './persistence-invariants.js'
export * from './protocol-limits.js'
+7 -2
View File
@@ -6,8 +6,13 @@ ARG LIBASOUND_PACKAGE=libasound2t64
ENV DEBIAN_FRONTEND=noninteractive
# Install Electron's link-time libraries without adding a display server or FUSE.
RUN apt-get update \
&& apt-get install -y --no-install-recommends \
# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts.
RUN for attempt in 1 2 3 4 5; do \
apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \
if [ "$attempt" = 5 ]; then exit 100; fi; \
rm -rf /var/lib/apt/lists/*; sleep 20; \
done \
&& apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \
bash \
ca-certificates \
coreutils \
+7 -2
View File
@@ -5,8 +5,13 @@ ARG LIBASOUND_PACKAGE=libasound2t64
ENV DEBIAN_FRONTEND=noninteractive
RUN apt-get update \
&& apt-get install -y --no-install-recommends \
# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts.
RUN for attempt in 1 2 3 4 5; do \
apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \
if [ "$attempt" = 5 ]; then exit 100; fi; \
rm -rf /var/lib/apt/lists/*; sleep 20; \
done \
&& apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \
bash \
ca-certificates \
dbus-x11 \
@@ -2,8 +2,13 @@ FROM ubuntu@sha256:678c6550cc43645e08669028bc177f50be4e7c5b8cca677067b1914d4afc7
ENV DEBIAN_FRONTEND=noninteractive
RUN apt-get update \
&& apt-get install -y --no-install-recommends \
# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts.
RUN for attempt in 1 2 3 4 5; do \
apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \
if [ "$attempt" = 5 ]; then exit 100; fi; \
rm -rf /var/lib/apt/lists/*; sleep 20; \
done \
&& apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \
bash \
ca-certificates \
dbus-x11 \
+12 -3
View File
@@ -19,6 +19,7 @@ const {
} = require('./scripts/verify-packaged-node-pty-job-ownership.cjs')
const { verifySkillsCliRuntime } = require('./scripts/verify-skills-cli-runtime.cjs')
const { verifyStaticAppImagePackage } = require('./scripts/static-appimage-package-contract.cjs')
const { signWindowsUninstallerViaSignPath } = require('./scripts/windows-uninstaller-signing.cjs')
// Why: dev-channel builds must carry the *release* identity — same bundle id,
// Developer ID signature, and notarization ticket — or Squirrel.Mac refuses to
@@ -401,9 +402,17 @@ module.exports = {
// name is absent. An unsigned build that still claimed 'SignPath Foundation'
// would therefore reject its own channel's next build — and its way back to
// stable with it. Dropping it is what makes dev→dev and dev→stable work.
...(isWinDevChannel
? { verifyUpdateCodeSignature: false }
: { signtoolOptions: { publisherName: 'SignPath Foundation' } }),
// Why a sign hook on a build that does not sign: it is the only moment
// electron-builder exposes the NSIS uninstaller (built in its own makensis
// pass, embedded, then deleted). The hook signs nothing — it relays the file
// to and from the CI SignPath request, and is inert when the relay env vars
// are unset, so local and dev builds are unaffected. publisherName stays on
// its existing channel split above.
signtoolOptions: {
sign: signWindowsUninstallerViaSignPath,
...(isWinDevChannel ? {} : { publisherName: 'SignPath Foundation' })
},
...(isWinDevChannel ? { verifyUpdateCodeSignature: false } : {}),
extraResources: [
...commonExtraResources,
...createPackagedRuntimeNodeModuleResources('win32'),
+35 -9
View File
@@ -49,22 +49,48 @@
; ---------------------------------------------------------------------------
; Clean up the relocated terminal daemon on a REAL uninstall.
;
; Why: the daemon host is deliberately copied to a distinct image name
; (orca-terminal-daemon.exe) under %LOCALAPPDATA%\Orca\daemon-host so that app
; UPDATES cannot kill it — that relocation is what keeps terminals alive across
; updates. The same design means a normal uninstall's process sweep and file
; removal both miss it, leaving an orphaned daemon plus its runtime copy behind.
; Why: the daemon host is deliberately copied OUT of the install dir into
; %LOCALAPPDATA%\Orca\daemon-host so that app UPDATES cannot kill it —
; electron-builder's kill sweep selects processes whose image path is under
; $INSTDIR, and that relocation is what keeps terminals alive across updates.
; The same design means a normal uninstall's process sweep and file removal both
; miss it, leaving an orphaned daemon plus its runtime copy behind.
;
; The ${isUpdated} guard is essential: electron-builder runs this uninstaller as
; part of uninstallOldVersion on EVERY update, and killing the daemon there would
; defeat the whole feature. Only clean up on a genuine uninstall.
;
; The image name and the LOCALAPPDATA folder name must stay in sync with
; DAEMON_HOST_EXE_NAME and LOCAL_HOST_ROOT_NAME in
; src/main/daemon/daemon-host-relocation.ts.
; The LOCALAPPDATA folder name must stay in sync with LOCAL_HOST_ROOT_NAME in
; src/main/daemon/daemon-host-relocation.ts. See
; docs/reference/windows-daemon-host-relocation.md.
!macro customUnInstall
${ifNot} ${isUpdated}
nsExec::Exec 'taskkill /F /IM orca-terminal-daemon.exe'
Push $0
Push $1
Push $2
; The host exe is a verbatim copy of the app exe, so the app's own image name
; reaches it; the second name covers hosts left by builds that renamed the copy.
; Filtered to the current user like upstream's per-user KILL_PROCESS, so an
; elevated machine-wide uninstall cannot reach another logged-on user's session.
; NSIS expands USERNAME itself: routing through cmd.exe only to get %USERNAME%
; would add two interpreter spawns to the uninstall path for nothing.
ReadEnvStr $1 USERNAME
${if} $1 == ""
; Measured: taskkill rejects an empty filter value outright ("The search filter
; cannot be recognized") and kills nothing, so with no USERNAME to scope by,
; kill unfiltered rather than not at all. USERNAME is set in every session an
; uninstaller runs in, so this is a backstop, not the expected path.
StrCpy $2 ""
${else}
StrCpy $2 '/FI "USERNAME eq $1"'
${endIf}
nsExec::Exec 'taskkill /F /IM "${APP_EXECUTABLE_FILENAME}" $2'
Pop $0
nsExec::Exec 'taskkill /F /IM "orca-terminal-daemon.exe" $2'
Pop $0
Pop $2
Pop $1
Pop $0
; Give the OS a moment to release the image lock before removing the tree.
Sleep 500
RMDir /r "$LOCALAPPDATA\Orca\daemon-host"
+35
View File
@@ -0,0 +1,35 @@
{
"$schema": "../node_modules/oxlint/configuration_schema.json",
"plugins": [],
"categories": {
"correctness": "off",
"suspicious": "off",
"pedantic": "off",
"perf": "off",
"style": "off",
"restriction": "off",
"nursery": "off"
},
"jsPlugins": [
{
"name": "app-store-performance",
"specifier": "../config/oxlint-plugins/app-store-performance.mjs"
},
{
"name": "quadratic-buffer-concat",
"specifier": "../config/oxlint-plugins/quadratic-buffer-concat.mjs"
},
{
"name": "sort-comparator-performance",
"specifier": "../config/oxlint-plugins/sort-comparator-performance.mjs"
}
],
"rules": {
"app-store-performance/require-selector": "warn",
"app-store-performance/no-identity-selector": "warn",
"app-store-performance/no-fresh-selector-result": "warn",
"quadratic-buffer-concat/no-loop-carried-concat": "warn",
"sort-comparator-performance/no-repeated-collator": "warn"
},
"ignorePatterns": ["**/node_modules", "**/dist", "**/out", "**/*.test.*", "**/*.spec.*"]
}
@@ -0,0 +1,60 @@
const FUNCTION_TYPES = new Set([
'ArrowFunctionExpression',
'FunctionExpression',
'FunctionDeclaration'
])
function propertyName(node) {
if (node?.type !== 'MemberExpression') {
return null
}
if (!node.computed && node.property.type === 'Identifier') {
return node.property.name
}
return node.property.type === 'Literal' ? node.property.value : null
}
function isInlineSortComparator(node) {
for (let parent = node.parent; parent; parent = parent.parent) {
if (!FUNCTION_TYPES.has(parent.type)) {
continue
}
const call = parent.parent
return (
call?.type === 'CallExpression' &&
call.arguments[0] === parent &&
['sort', 'toSorted'].includes(propertyName(call.callee))
)
}
return false
}
function isCollatorConstruction(node) {
return (
node.callee?.object?.type === 'Identifier' &&
node.callee.object.name === 'Intl' &&
propertyName(node.callee) === 'Collator'
)
}
function createRule(context) {
function inspect(node) {
const optionedComparison =
node.type === 'CallExpression' &&
propertyName(node.callee) === 'localeCompare' &&
node.arguments.length >= 3
if ((optionedComparison || isCollatorConstruction(node)) && isInlineSortComparator(node)) {
context.report({
node,
message:
'Create one Intl.Collator before sorting and reuse its compare method; resolving collation options inside the comparator repeats setup for every comparison. Preserve the locale, options, and tie-breaker.'
})
}
}
return { CallExpression: inspect, NewExpression: inspect }
}
export default {
meta: { name: 'sort-comparator-performance' },
rules: { 'no-repeated-collator': { create: createRule } }
}
@@ -27,15 +27,424 @@ index 855bd4b86f0a3c18c7594212c0e42b6e35bc4001..33774e7ae296f0de39dd94156673c9e7
"/guard:cf",
"/sdl",
diff --git a/src/process.cc b/src/process.cc
index 3eea92077c4d1d433119361d5c432881859131e9..1998f4addd4d7e9aba946ea6f7f7a4a5d13291bc 100644
index 3eea92077c4d1d433119361d5c432881859131e9..738775f6fcdfb676054386fe34c0380327ed1863 100644
--- a/src/process.cc
+++ b/src/process.cc
@@ -37,7 +37,7 @@ uint32_t GetRawProcessList(std::vector<ProcessInfo>& process_info,
process_info.push_back(std::move(pinfo));
process_count++;
}
@@ -1,108 +1,112 @@
-/*---------------------------------------------------------------------------------------------
- * Copyright (c) Microsoft Corporation. All rights reserved.
- * Licensed under the MIT License. See License.txt in the project root for license information.
- *--------------------------------------------------------------------------------------------*/
-
-#include "process.h"
-#include "process_commandline.h"
-
-#include <tlhelp32.h>
-#include <psapi.h>
-#include <limits>
-
-uint32_t GetRawProcessList(std::vector<ProcessInfo>& process_info,
- DWORD process_data_flags) {
- // Fetch the PID and PPIDs
- PROCESSENTRY32 process_entry = { 0 };
- DWORD parent_pid = 0;
- uint32_t process_count = 0;
- HANDLE snapshot_handle = CreateToolhelp32Snapshot(TH32CS_SNAPPROCESS, 0);
- process_entry.dwSize = sizeof(PROCESSENTRY32);
- if (Process32First(snapshot_handle, &process_entry)) {
- do {
- if (process_entry.th32ProcessID != 0) {
- ProcessInfo pinfo;
- pinfo.pid = process_entry.th32ProcessID;
- pinfo.ppid = process_entry.th32ParentProcessID;
-
- if (MEMORY & process_data_flags) {
- GetProcessMemoryUsage(pinfo);
- }
-
- if (COMMANDLINE & process_data_flags) {
- GetProcessCommandLine(pinfo);
- }
-
- strcpy(pinfo.name, process_entry.szExeFile);
- process_info.push_back(std::move(pinfo));
- process_count++;
- }
- } while (process_count < 1024 && Process32Next(snapshot_handle, &process_entry));
- }
-
- CloseHandle(snapshot_handle);
- return process_count;
-}
-
-void GetProcessMemoryUsage(ProcessInfo& process_info) {
- DWORD pid = process_info.pid;
- HANDLE hProcess;
- PROCESS_MEMORY_COUNTERS pmc;
-
- hProcess = OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, false, pid);
-
- if (hProcess == NULL) {
- return;
- }
-
- if (GetProcessMemoryInfo(hProcess, &pmc, sizeof(pmc))) {
- process_info.memory = (DWORD)pmc.WorkingSetSize;
- }
-
- CloseHandle(hProcess);
-}
-
-// Per documentation, it is not recommended to add or subtract values from the FILETIME
-// structure, or to cast it to ULARGE_INTEGER as this can cause alignment faults on 64-bit Windows.
-// Copy the high and low part to a ULARGE_INTEGER and peform arithmetic on that instead.
-// See https://msdn.microsoft.com/en-us/library/windows/desktop/ms724284(v=vs.85).aspx
-ULONGLONG GetTotalTime(const FILETIME* kernelTime, const FILETIME* userTime) {
- ULARGE_INTEGER kt, ut;
- kt.LowPart = (*kernelTime).dwLowDateTime;
- kt.HighPart = (*kernelTime).dwHighDateTime;
-
- ut.LowPart = (*userTime).dwLowDateTime;
- ut.HighPart = (*userTime).dwHighDateTime;
-
- return kt.QuadPart + ut.QuadPart;
-}
-
-void GetCpuUsage(Cpu& cpu_info, bool first_pass) {
- DWORD pid = cpu_info.pid;
- HANDLE hProcess;
-
- hProcess = OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, false, pid);
-
- if (hProcess == NULL) {
- return;
- }
-
- FILETIME creationTime, exitTime, kernelTime, userTime;
- FILETIME sysIdleTime, sysKernelTime, sysUserTime;
- if (GetProcessTimes(hProcess, &creationTime, &exitTime, &kernelTime, &userTime)
- && GetSystemTimes(&sysIdleTime, &sysKernelTime, &sysUserTime)) {
- if (first_pass) {
- cpu_info.initialProcRunTime = GetTotalTime(&kernelTime, &userTime);
- cpu_info.initialSystemTime = GetTotalTime(&sysKernelTime, &sysUserTime);
- } else {
- ULONGLONG endProcTime = GetTotalTime(&kernelTime, &userTime);
- ULONGLONG endSysTime = GetTotalTime(&sysKernelTime, &sysUserTime);
-
- cpu_info.cpu = 100.0 * (endProcTime - cpu_info.initialProcRunTime) / (endSysTime - cpu_info.initialSystemTime);
- }
- } else {
- cpu_info.cpu = std::numeric_limits<double>::quiet_NaN();
- }
-
- CloseHandle(hProcess);
+/*---------------------------------------------------------------------------------------------
+ * Copyright (c) Microsoft Corporation. All rights reserved.
+ * Licensed under the MIT License. See License.txt in the project root for license information.
+ *--------------------------------------------------------------------------------------------*/
+
+#include "process.h"
+#include "process_commandline.h"
+
+#include <tlhelp32.h>
+#include <psapi.h>
+#include <limits>
+
+uint32_t GetRawProcessList(std::vector<ProcessInfo>& process_info,
+ DWORD process_data_flags) {
+ // Fetch the PID and PPIDs
+ PROCESSENTRY32 process_entry = { 0 };
+ DWORD parent_pid = 0;
+ uint32_t process_count = 0;
+ HANDLE snapshot_handle = CreateToolhelp32Snapshot(TH32CS_SNAPPROCESS, 0);
+ process_entry.dwSize = sizeof(PROCESSENTRY32);
+ if (Process32First(snapshot_handle, &process_entry)) {
+ do {
+ if (process_entry.th32ProcessID != 0) {
+ // Value-initialize: `memory` is otherwise stack garbage when the flag is unset.
+ ProcessInfo pinfo{};
+ pinfo.pid = process_entry.th32ProcessID;
+ pinfo.ppid = process_entry.th32ParentProcessID;
+
+ if (MEMORY & process_data_flags) {
+ GetProcessMemoryUsage(pinfo);
+ }
+
+ if (COMMANDLINE & process_data_flags) {
+ GetProcessCommandLine(pinfo);
+ }
+
+ strcpy(pinfo.name, process_entry.szExeFile);
+ process_info.push_back(std::move(pinfo));
+ process_count++;
+ }
+ } while (Process32Next(snapshot_handle, &process_entry));
}
CloseHandle(snapshot_handle);
+ }
+
+ CloseHandle(snapshot_handle);
+ return process_count;
+}
+
+void GetProcessMemoryUsage(ProcessInfo& process_info) {
+ DWORD pid = process_info.pid;
+ HANDLE hProcess;
+ PROCESS_MEMORY_COUNTERS pmc;
+
+ // PROCESS_VM_READ is never used here -- GetProcessMemoryInfo reads counters the
+ // kernel keeps, not the address space -- and acquiring it is what EDR scores.
+ hProcess = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, false, pid);
+
+ if (hProcess == NULL) {
+ return;
+ }
+
+ if (GetProcessMemoryInfo(hProcess, &pmc, sizeof(pmc))) {
+ process_info.memory = (DWORD)pmc.WorkingSetSize;
+ }
+
+ CloseHandle(hProcess);
+}
+
+// Per documentation, it is not recommended to add or subtract values from the FILETIME
+// structure, or to cast it to ULARGE_INTEGER as this can cause alignment faults on 64-bit Windows.
+// Copy the high and low part to a ULARGE_INTEGER and peform arithmetic on that instead.
+// See https://msdn.microsoft.com/en-us/library/windows/desktop/ms724284(v=vs.85).aspx
+ULONGLONG GetTotalTime(const FILETIME* kernelTime, const FILETIME* userTime) {
+ ULARGE_INTEGER kt, ut;
+ kt.LowPart = (*kernelTime).dwLowDateTime;
+ kt.HighPart = (*kernelTime).dwHighDateTime;
+
+ ut.LowPart = (*userTime).dwLowDateTime;
+ ut.HighPart = (*userTime).dwHighDateTime;
+
+ return kt.QuadPart + ut.QuadPart;
+}
+
+void GetCpuUsage(Cpu& cpu_info, bool first_pass) {
+ DWORD pid = cpu_info.pid;
+ HANDLE hProcess;
+
+ // GetProcessTimes needs no more than PROCESS_QUERY_LIMITED_INFORMATION.
+ hProcess = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, false, pid);
+
+ if (hProcess == NULL) {
+ return;
+ }
+
+ FILETIME creationTime, exitTime, kernelTime, userTime;
+ FILETIME sysIdleTime, sysKernelTime, sysUserTime;
+ if (GetProcessTimes(hProcess, &creationTime, &exitTime, &kernelTime, &userTime)
+ && GetSystemTimes(&sysIdleTime, &sysKernelTime, &sysUserTime)) {
+ if (first_pass) {
+ cpu_info.initialProcRunTime = GetTotalTime(&kernelTime, &userTime);
+ cpu_info.initialSystemTime = GetTotalTime(&sysKernelTime, &sysUserTime);
+ } else {
+ ULONGLONG endProcTime = GetTotalTime(&kernelTime, &userTime);
+ ULONGLONG endSysTime = GetTotalTime(&sysKernelTime, &sysUserTime);
+
+ cpu_info.cpu = 100.0 * (endProcTime - cpu_info.initialProcRunTime) / (endSysTime - cpu_info.initialSystemTime);
+ }
+ } else {
+ cpu_info.cpu = std::numeric_limits<double>::quiet_NaN();
+ }
+
+ CloseHandle(hProcess);
}
\ No newline at end of file
diff --git a/src/process_commandline.cc b/src/process_commandline.cc
index ea822b120e8038a4803e34647042f08f4aaf5ca1..25907c0bf542bed6c72b1b462b19bcf3210c3cfd 100644
--- a/src/process_commandline.cc
+++ b/src/process_commandline.cc
@@ -1,67 +1,125 @@
-/*---------------------------------------------------------------------------------------------
- * Copyright (c) Microsoft Corporation. All rights reserved.
- * Licensed under the MIT License. See License.txt in the project root for license information.
- *--------------------------------------------------------------------------------------------*/
-
-#include "process.h"
-#include "process_commandline.h"
-#include <windows.h>
-#include <winternl.h>
-#include <iostream>
-
-bool GetProcessCommandLine(ProcessInfo& process_info) {
- HINSTANCE ntdll = GetModuleHandleW(L"ntdll.dll");
- if (!ntdll) {
- return false;
- }
-
- decltype(NtQueryInformationProcess)* nt_query_information_process =
- reinterpret_cast<decltype(NtQueryInformationProcess)*>(
- GetProcAddress(ntdll, "NtQueryInformationProcess"));
-
- if (!nt_query_information_process) {
- return false;
- }
-
- PROCESS_BASIC_INFORMATION pbi{};
- PEB peb = {NULL};
- RTL_USER_PROCESS_PARAMETERS process_parameters = {NULL};
-
- // Get process handle
- DWORD pid = process_info.pid;
- HANDLE hProcess = OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, FALSE, pid);
- if (hProcess == INVALID_HANDLE_VALUE) {
- return false;
- }
-
- // Get Process Environment Block (PEB)
- NTSTATUS status = nt_query_information_process(hProcess, ProcessBasicInformation, &pbi, sizeof(pbi), nullptr);
- if (NT_SUCCESS(status) && pbi.PebBaseAddress) {
- // Read PEB
- if (ReadProcessMemory(hProcess, pbi.PebBaseAddress, &peb, sizeof(peb), nullptr)) {
- // Read the processs parameters
- if (ReadProcessMemory(hProcess, peb.ProcessParameters, &process_parameters, sizeof(RTL_USER_PROCESS_PARAMETERS), nullptr)) {
- if (process_parameters.CommandLine.Length > 0) {
- std::wstring buffer;
- buffer.resize(process_parameters.CommandLine.Length / sizeof(wchar_t));
- if (ReadProcessMemory(hProcess, process_parameters.CommandLine.Buffer, &buffer[0], process_parameters.CommandLine.Length, nullptr)) {
- int wide_length = static_cast<int>(buffer.length());
- int charcount = WideCharToMultiByte(CP_UTF8, 0, buffer.data(), wide_length,
- NULL, 0, NULL, NULL);
- if (charcount) {
- process_info.commandLine.resize(static_cast<size_t>(charcount));
- WideCharToMultiByte(CP_UTF8, 0, buffer.data(), wide_length,
- &process_info.commandLine[0], charcount,
- NULL, NULL);
- }
- CloseHandle(hProcess);
- return true;
- }
- }
- }
- }
- }
-
- CloseHandle(hProcess);
- return false;
-}
+/*---------------------------------------------------------------------------------------------
+ * Copyright (c) Microsoft Corporation. All rights reserved.
+ * Licensed under the MIT License. See License.txt in the project root for license information.
+ *--------------------------------------------------------------------------------------------*/
+
+#include "process.h"
+#include "process_commandline.h"
+#include <windows.h>
+#include <winternl.h>
+#include <vector>
+
+namespace {
+
+// Windows 8.1 and later hand back a process's command line as a UNICODE_STRING
+// the kernel builds, needing only PROCESS_QUERY_LIMITED_INFORMATION.
+//
+// There is deliberately no PEB fallback. Reading the command line out of the
+// target's address space -- opening it for VM reads and then chaining
+// memory reads across every pid on a timer -- is the credential-dumping
+// primitive this reader exists to not perform, so it is absent from the binary
+// rather than one anomalous NTSTATUS away. Electron's floor is Windows 10, so
+// every OS Orca supports has this class; if a hooked ntdll refuses it anyway,
+// the command line comes back empty, which callers already handle, instead of
+// silently reinstating the primitive on exactly the instrumented machines this
+// reader was written for.
+const ULONG kProcessCommandLineInformation = 60;
+
+const NTSTATUS kStatusInfoLengthMismatch = static_cast<NTSTATUS>(0xC0000004L);
+const NTSTATUS kStatusBufferTooSmall = static_cast<NTSTATUS>(0xC0000023L);
+
+// A command line is a UNICODE_STRING, whose Length is a USHORT, so the kernel
+// can never need more than the header plus 64 KiB. Refusing anything larger
+// keeps a bogus size from throwing bad_alloc out of a scan that has already
+// walked most of the table.
+const ULONG kMaxCommandLineBytes = sizeof(UNICODE_STRING) + 0xFFFF + sizeof(wchar_t);
+
+// winternl.h's PROCESSINFOCLASS does not name class 60 and its enumerator range
+// stops far short of it, so the class travels as a ULONG rather than a cast enum.
+typedef NTSTATUS(NTAPI* NtQueryInformationProcessFn)(HANDLE, ULONG, PVOID, ULONG, PULONG);
+
+// ntdll ships no import library for this entry point; it has to be resolved.
+NtQueryInformationProcessFn ResolveNtQueryInformationProcess() {
+ HMODULE ntdll = GetModuleHandleW(L"ntdll.dll");
+ if (!ntdll) {
+ return nullptr;
+ }
+ return reinterpret_cast<NtQueryInformationProcessFn>(
+ GetProcAddress(ntdll, "NtQueryInformationProcess"));
+}
+
+NtQueryInformationProcessFn NtQueryInformationProcessEntry() {
+ static NtQueryInformationProcessFn entry = ResolveNtQueryInformationProcess();
+ return entry;
+}
+
+bool StoreCommandLineUtf8(ProcessInfo& process_info, const wchar_t* data, size_t wide_length) {
+ if (wide_length == 0) {
+ return false;
+ }
+ int length = static_cast<int>(wide_length);
+ int charcount = WideCharToMultiByte(CP_UTF8, 0, data, length, NULL, 0, NULL, NULL);
+ if (!charcount) {
+ return false;
+ }
+ process_info.commandLine.resize(static_cast<size_t>(charcount));
+ WideCharToMultiByte(CP_UTF8, 0, data, length, &process_info.commandLine[0], charcount, NULL,
+ NULL);
+ return true;
+}
+
+} // namespace
+
+bool GetProcessCommandLine(ProcessInfo& process_info) {
+ NtQueryInformationProcessFn query = NtQueryInformationProcessEntry();
+ if (!query) {
+ return false;
+ }
+
+ HANDLE process = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, FALSE, process_info.pid);
+ if (process == NULL) {
+ return false;
+ }
+
+ ULONG size = 0;
+ NTSTATUS status = query(process, kProcessCommandLineInformation, nullptr, 0, &size);
+ if (NT_SUCCESS(status)) {
+ // Nothing was written, so there is no command line to read.
+ CloseHandle(process);
+ return false;
+ }
+ if (status != kStatusInfoLengthMismatch && status != kStatusBufferTooSmall) {
+ CloseHandle(process);
+ return false;
+ }
+ if (size < sizeof(UNICODE_STRING) || size > kMaxCommandLineBytes) {
+ CloseHandle(process);
+ return false;
+ }
+
+ std::vector<unsigned char> buffer(size);
+ status = query(process, kProcessCommandLineInformation, &buffer[0], size, &size);
+ CloseHandle(process);
+ if (!NT_SUCCESS(status)) {
+ return false;
+ }
+
+ // Header and characters arrive in one allocation, but treat the header as
+ // untrusted: a hooked ntdll is the case this reader is written for, and an
+ // unchecked Buffer/Length here would be an over-read encoded straight into JS.
+ // Bound against buffer.size(), never `size` -- the second query overwrote it.
+ const UNICODE_STRING* command_line = reinterpret_cast<const UNICODE_STRING*>(&buffer[0]);
+ const unsigned char* begin = &buffer[0];
+ const unsigned char* end = begin + buffer.size();
+ const unsigned char* chars = reinterpret_cast<const unsigned char*>(command_line->Buffer);
+ if (chars == nullptr || chars < begin + sizeof(UNICODE_STRING) || chars > end ||
+ command_line->Length > static_cast<ULONG>(end - chars)) {
+ return false;
+ }
+
+ // True only when a command line was actually stored, so "empty" and "not
+ // recovered" stay the same answer they were before this reader replaced the
+ // PEB read. `src/process.cc` discards the result either way.
+ return StoreCommandLineUtf8(process_info, command_line->Buffer,
+ command_line->Length / sizeof(wchar_t));
+}
+149 -21
View File
@@ -603,7 +603,7 @@ index 7b4b9e1f990fbf95b51528bb56dc9717f5b87532..2ae787c5bd4f3eba470584dc658a01a5
}
#endif
diff --git a/src/win/conpty.cc b/src/win/conpty.cc
index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a6a4082ce 100644
index 7b286d3d644c26141df516929703aa6e129df4b2..4aed260dd68e6a171dcfd349e9a7c5c97209248e 100644
--- a/src/win/conpty.cc
+++ b/src/win/conpty.cc
@@ -18,6 +18,7 @@
@@ -614,7 +614,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a
#include <vector>
#include <Windows.h>
#include <strsafe.h>
@@ -44,12 +45,29 @@ struct pty_baton {
@@ -44,12 +45,39 @@ struct pty_baton {
HANDLE hOut;
HPCON hpc;
@@ -630,22 +630,32 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a
+ // refused to create or assign one (an outer job without breakaway rights),
+ // in which case callers fall back to their pre-job behaviour.
+ HANDLE hJob = nullptr;
+
+ // Orca: teardown needs BOTH the shell's death and an explicit kill() before
+ // the baton can be freed, so each side records that it has run. Whichever
+ // arrives second frees it. Freeing on the shell's death alone -- what this
+ // file did before -- destroyed the only record of `hpc` while
+ // ClosePseudoConsole was still owed, which is why a self-exiting shell
+ // leaked its pseudoconsole and the console host it reaps (#18601 / F24).
+ bool shellExited = false;
+ bool consoleClosed = false;
pty_baton(int _id, HANDLE _hIn, HANDLE _hOut, HPCON _hpc) : id(_id), hIn(_hIn), hOut(_hOut), hpc(_hpc) {};
};
static std::vector<std::unique_ptr<pty_baton>> ptyHandles;
+// Orca: guards the job accessors below against the exit watcher thread. It does
+// NOT make the whole table safe -- PtyResize/PtyClear/PtyKill read it unlocked,
+// as they always have -- but it closes the window this patch opened, where the
+// watcher can close hShell/hJob and free the baton between a lookup and its use.
+// Orca: guards the job accessors below, and PtyKill, against the exit watcher
+// thread. It does NOT make the whole table safe -- PtyResize and PtyClear still
+// read it unlocked, as they always have -- but it closes the window this patch
+// opened, where the watcher can close hShell/hJob and free the baton between a
+// lookup and its use.
+// Handle VALUES are recycled aggressively, so an unguarded read could pass the
+// shell-pid check against an unrelated process and terminate the wrong job.
+static std::mutex ptyJobMutex;
static volatile LONG ptyCounter;
static pty_baton* get_pty_baton(int id) {
@@ -102,8 +120,27 @@ void SetupExitCallback(Napi::Env env, Napi::Function cb, pty_baton* baton) {
@@ -102,8 +130,31 @@ void SetupExitCallback(Napi::Env env, Napi::Function cb, pty_baton* baton) {
// Get process exit code.
GetExitCodeProcess(baton->hShell, (LPDWORD)(&exit_event->exit_code));
// Clean up handles
@@ -665,9 +675,13 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a
+ // Why inside the lock: erasing frees the baton the job accessors hold a
+ // pointer to. Note remove_pty_baton must not be an assert() argument --
+ // NDEBUG would compile the call away and leak every baton.
+ const bool removed = remove_pty_baton(baton->id);
+ assert(removed);
+ (void)removed;
+ baton->shellExited = true;
+ if (baton->consoleClosed) {
+ const bool removed = remove_pty_baton(baton->id);
+ assert(removed);
+ (void)removed;
+ }
+ // Else PtyKill has not run yet and still owns hpc. It frees the baton.
+ }
+ // Why the lock ends here: BlockingCall below waits on the JS thread, and the
+ // JS thread can be waiting on ptyJobMutex inside PtyTerminateJob. Holding
@@ -675,7 +689,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a
auto status = tsfn.BlockingCall(exit_event, callback); // In main thread
switch (status) {
@@ -409,6 +446,15 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) {
@@ -409,6 +460,15 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) {
throw errorWithCode(info, "UpdateProcThreadAttribute failed");
}
@@ -691,7 +705,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a
PROCESS_INFORMATION piClient{};
fSuccess = !!CreateProcessW(
nullptr,
@@ -416,7 +462,10 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) {
@@ -416,7 +476,10 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) {
nullptr, // lpProcessAttributes
nullptr, // lpThreadAttributes
false, // bInheritHandles VERY IMPORTANT that this is false
@@ -703,7 +717,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a
envArg, // lpEnvironment
mutableCwd.get(), // lpCurrentDirectory
&siEx.StartupInfo, // lpStartupInfo
@@ -426,8 +475,47 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) {
@@ -426,8 +489,47 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) {
throw errorWithCode(info, "Cannot create process");
}
@@ -753,7 +767,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a
if (useConptyDll && fLoadedDll)
{
PFNRELEASEPSEUDOCONSOLE const pfnReleasePseudoConsole = (PFNRELEASEPSEUDOCONSOLE)GetProcAddress(
@@ -440,6 +528,8 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) {
@@ -440,6 +542,8 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) {
// Update handle
handle->hShell = piClient.hProcess;
@@ -762,7 +776,91 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a
// Close the thread handle to avoid resource leak
CloseHandle(piClient.hThread);
@@ -567,6 +657,143 @@ static Napi::Value PtyKill(const Napi::CallbackInfo& info) {
@@ -544,29 +648,215 @@ static Napi::Value PtyKill(const Napi::CallbackInfo& info) {
int id = info[0].As<Napi::Number>().Int32Value();
const bool useConptyDll = info[1].As<Napi::Boolean>().Value();
- const pty_baton* handle = get_pty_baton(id);
+ // Orca: resolve the DLL BEFORE touching any baton state, for the same reason
+ // PtyConnect does it before creating anything. LoadConptyDll throws when
+ // conpty.dll is missing, and a throw after consoleClosed was set would strand
+ // the pseudoconsole permanently: the retry would find the work already
+ // claimed and do nothing. Only the useConptyDll path can throw here; the
+ // other returns kernel32.
+ HANDLE hLibrary = LoadConptyDll(info, useConptyDll);
+ PFNCLOSEPSEUDOCONSOLE pfnClosePseudoConsole = nullptr;
+ if (hLibrary != nullptr) {
+ pfnClosePseudoConsole = (PFNCLOSEPSEUDOCONSOLE)GetProcAddress(
+ (HMODULE)hLibrary,
+ useConptyDll ? "ConptyClosePseudoConsole" : "ClosePseudoConsole");
+ }
- if (handle != nullptr) {
- HANDLE hLibrary = LoadConptyDll(info, useConptyDll);
- bool fLoadedDll = hLibrary != nullptr;
- if (fLoadedDll)
- {
- PFNCLOSEPSEUDOCONSOLE const pfnClosePseudoConsole = (PFNCLOSEPSEUDOCONSOLE)GetProcAddress(
- (HMODULE)hLibrary,
- useConptyDll ? "ConptyClosePseudoConsole" : "ClosePseudoConsole");
- if (pfnClosePseudoConsole)
- {
- pfnClosePseudoConsole(handle->hpc);
+ // Orca: the baton now outlives the shell, so this runs on a self-exited pty
+ // too -- that is the whole point. Take what we need under the lock: the
+ // watcher thread nulls hShell the moment the shell dies, and TerminateProcess
+ // on a handle it just closed is an invalid-handle operation. Duplicating
+ // rather than reordering keeps upstream's close-then-terminate sequence.
+ HPCON hpc = nullptr;
+ HANDLE hShellDup = nullptr;
+ bool owed = false;
+ {
+ std::lock_guard<std::mutex> guard(ptyJobMutex);
+ pty_baton* handle = get_pty_baton(id);
+ // Why the consoleClosed check: a second kill() would otherwise close the
+ // same pseudoconsole twice. Upstream relied on the baton being gone.
+ if (handle != nullptr && !handle->consoleClosed) {
+ hpc = handle->hpc;
+ owed = true;
+ handle->consoleClosed = true;
+ // Null hShell means a self-exited pty, where there is nothing to kill.
+ if (useConptyDll && handle->hShell != nullptr) {
+ if (!DuplicateHandle(GetCurrentProcess(), handle->hShell, GetCurrentProcess(),
+ &hShellDup, 0, FALSE, DUPLICATE_SAME_ACCESS)) {
+ // Why terminate here instead of skipping: a failed duplication leaves
+ // hShellDup null, which is indistinguishable from the self-exit case,
+ // and skipping would leave the shell RUNNING after its pane closed --
+ // a worse outcome than the leak this all exists to fix. hShell is
+ // valid under this lock and TerminateProcess does not block, so the
+ // only cost is that this rare path kills before the console closes.
+ hShellDup = nullptr;
+ TerminateProcess(handle->hShell, 1);
+ }
+ }
+ if (handle->shellExited) {
+ const bool removed = remove_pty_baton(id);
+ assert(removed);
+ (void)removed;
}
+ // Else the shell is still running and the watcher frees the baton.
}
- if (useConptyDll) {
- TerminateProcess(handle->hShell, 1);
+ }
+
+ // Why outside the lock: ClosePseudoConsole blocks until the conout side has
+ // drained, and the watcher must be able to take the lock while it does.
+ if (owed) {
+ if (pfnClosePseudoConsole)
+ {
+ pfnClosePseudoConsole(hpc);
+ }
+ if (hShellDup != nullptr) {
+ TerminateProcess(hShellDup, 1);
+ CloseHandle(hShellDup);
}
}
return env.Undefined();
}
@@ -808,9 +906,11 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a
+ * Orca: the pids still alive in this pty's tree, straight from the kernel.
+ *
+ * Descendant liveness for a tree that is still tracked, including children that
+ * detached from the console. Once the shell exits the baton is gone, so this
+ * returns null rather than an empty list -- null means "no answer", never
+ * "they died". Also returns null when no job was assigned.
+ * detached from the console. Once the shell exits the watcher nulls hJob, which
+ * ownsShell rejects, so this returns null rather than an empty list -- null
+ * means "no answer", never "they died". (The baton itself now outlives the
+ * shell, until kill() runs; hJob is what makes the answer null.) Also returns
+ * null when no job was assigned.
+ *
+ * Does not include the ConPTY console host: CreatePseudoConsole spawns it
+ * before this job exists, so it is not a member and ClosePseudoConsole is what
@@ -906,7 +1006,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a
/**
* Init
*/
@@ -577,6 +804,9 @@ Napi::Object init(Napi::Env env, Napi::Object exports) {
@@ -577,6 +867,9 @@ Napi::Object init(Napi::Env env, Napi::Object exports) {
exports.Set("resize", Napi::Function::New(env, PtyResize));
exports.Set("clear", Napi::Function::New(env, PtyClear));
exports.Set("kill", Napi::Function::New(env, PtyKill));
@@ -917,7 +1017,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a
};
diff --git a/lib/windowsPtyAgent.js b/lib/windowsPtyAgent.js
index a358ffb..fb3a96f 100644
index a358ffb177357e177661033c1b092f9c9d0e5f5a..26c2a4c58799ce649f5113131e4c52f7ed2d87ad 100644
--- a/lib/windowsPtyAgent.js
+++ b/lib/windowsPtyAgent.js
@@ -136,6 +136,9 @@ var WindowsPtyAgent = /** @class */ (function () {
@@ -930,6 +1030,20 @@ index a358ffb..fb3a96f 100644
this._outSocket.readable = false;
this._getConsoleProcessList().then(function (consoleProcessList) {
consoleProcessList.forEach(function (pid) {
@@ -154,9 +157,10 @@ var WindowsPtyAgent = /** @class */ (function () {
// Close the input write handle to signal the end of session.
this._inSocket.destroy();
this._ptyNative.kill(this._pty, this._useConptyDll);
- this._outSocket.on('data', function () {
- _this._conoutSocketWorker.dispose();
- });
+ // Orca: dispose unconditionally, as the non-DLL branch above does.
+ // Waiting for another 'data' event leaks the conout worker on every
+ // self-exiting shell, because no more data ever arrives (F24).
+ this._conoutSocketWorker.dispose();
}
}
else {
diff --git a/lib/windowsTerminal.js b/lib/windowsTerminal.js
index 3c38f89..e20b3e6 100644
--- a/lib/windowsTerminal.js
@@ -1015,7 +1129,7 @@ index 3c38f89..e20b3e6 100644
\ No newline at end of file
+//# sourceMappingURL=windowsTerminal.js.map
diff --git a/src/windowsPtyAgent.ts b/src/windowsPtyAgent.ts
index d705444..ce611b8 100644
index d7054449516f0c9a62af351c2caa17331206d530..0c28a32e2e1db2b3f208ddde8443cd4e67bb1ad6 100644
--- a/src/windowsPtyAgent.ts
+++ b/src/windowsPtyAgent.ts
@@ -143,6 +143,9 @@ export class WindowsPtyAgent {
@@ -1028,6 +1142,20 @@ index d705444..ce611b8 100644
this._outSocket.readable = false;
this._getConsoleProcessList().then(consoleProcessList => {
consoleProcessList.forEach((pid: number) => {
@@ -159,9 +162,10 @@ export class WindowsPtyAgent {
// Close the input write handle to signal the end of session.
this._inSocket.destroy();
(this._ptyNative as IConptyNative).kill(this._pty, this._useConptyDll);
- this._outSocket.on('data', () => {
- this._conoutSocketWorker.dispose();
- });
+ // Orca: dispose unconditionally, as the non-DLL branch above does.
+ // Waiting for another 'data' event leaks the conout worker on every
+ // self-exiting shell, because no more data ever arrives (F24).
+ this._conoutSocketWorker.dispose();
}
} else {
// Because pty.kill closes the handle, it will kill most processes by itself.
diff --git a/src/windowsTerminal.ts b/src/windowsTerminal.ts
index 13f6c6d..eda63c8 100644
--- a/src/windowsTerminal.ts
+38
View File
@@ -0,0 +1,38 @@
# Performance regression checks
`pnpm --silent audit:perf > performance-audit.json` scans production `src/` with
the existing app-store and buffer-concatenation rules plus the sort-comparator
rule. Warnings are advisory in this full inventory; tool/parser failures fail.
New warning findings on changed lines fail `pnpm check:code-quality:changed`.
Tests, generated files, `mobile/` and `cloud/` are outside this source audit.
The sort rule detects optioned `localeCompare` and `Intl.Collator` construction
inside inline `sort`/`toSorted` callbacks. Construct one collator outside the
callback, preserving locale, options and tie-breakers. If the locale changes at
runtime, reconstruct at the next sort or key the cache by locale. Bare comparisons
and standalone equality checks are allowed. There is no autofix or interprocedural
analysis: named comparators, aliases, custom methods and deferred callbacks need
manual review. A warning identifies repeated setup, not proof of visible lag.
`pnpm test:perf:contracts` runs the explicit selection in
`vitest.performance.config.ts`: SQLite statement reuse and schema parity, relay
filesystem concurrency, tokenizer rejection, highlighting cache, queued
cancellation, terminal backing-memory retention and detector fixtures. Missing
listed files fail configuration loading. Tests run serially, without retries,
and inherit the full suite's setup and forced-GC support. This makes existing
regression coverage easy to run and attribute; it does not create new workload
coverage by itself.
`.github/workflows/performance-contracts.yml` runs daily and manually on Linux,
macOS and Windows, and on PRs changing this tooling or any listed contract file.
It uploads per-OS JSON test results, plus the source inventory once from Linux
because that scan is OS-independent. Its schedule starts after merge. Run the existing
`test:e2e:terminal-perf:scale:report` for rendered typing/frame budgets and
`test:e2e:ssh-docker-perf` for real transport behavior. Relay unit tests do not
measure SSH RTT, WSL scheduling or a packaged Electron renderer.
To extend coverage, select a production-path regression with an operation-count,
identity, queue-admission or retained-memory oracle. Confirm it fails with the
old behavior. Use controlled, counterbalanced benchmark samples for timings;
avoid new machine-dependent millisecond gates in the normal unit suite. A green
source scan and these contracts cannot establish that the whole app is fast.
@@ -0,0 +1,205 @@
const { createHash } = require('node:crypto')
const { readFileSync, renameSync, rmSync, writeFileSync } = require('node:fs')
const { join, resolve } = require('node:path')
/**
* Release the ConPTY teardown handles a relay's npm-installed node-pty never releases.
*
* Two files, and the ORDER of one of the edits is the whole fix.
*
* `windowsPtyAgent.js` -- `kill()` flips `readable` on both sockets and destroys neither.
* `_cleanUpProcess` destroys `_outSocket`, so the conout handle comes back; nothing ever destroys
* `_inSocket`, and it wraps a real Windows named-pipe handle from `fs.openSync(term.conin, 'w')`.
* Every terminal leaks one File handle for the life of the host process.
*
* The obvious fix -- and the placement `config/patches/node-pty@1.1.0.patch` uses -- releases it at
* the TOP of the branch, before `_getConsoleProcessList()` forks and before the native kill. That is
* measurably worse than leaving the leak alone: teardown aborts partway, the forked console-list
* agent is never reaped, and both pipe handles stay alive instead of one. This asset releases it at
* the END of the branch instead, after the fork and the kill have already happened.
*
* Measured on a Windows SSH host, 20 spawn/kill cycles, handles bucketed by NT object type
* (identical numbers standalone and through a real relay). Every row is the NON-DLL branch, which
* is the branch a relay runs -- see the divergence note below for why that matters:
*
* published node-pty File +1/terminal, Process flat
* desktop patch placement File +2/terminal, Process +1/terminal <-- 3x WORSE
* released last (here) File flat, Process flat
*
* `windowsTerminal.js` carries the desktop's error-listener hunks verbatim. The conin listener is
* what keeps a pipe error retiring one terminal instead of the host -- its own comment names the
* failure mode: "Without a listener, Node promotes errors such as write EAGAIN to uncaughtException".
* It is not what fixes the leak (adding it changed nothing on its own), but it is the guard that
* makes destroying conin safe at all.
*
* Why this ships as a relay asset rather than only in config/patches/node-pty@1.1.0.patch: pnpm
* patches do not cross the SSH boundary -- a relay host runs the tree `npm install` put there.
*
* DELIBERATE DIVERGENCE FROM THE DESKTOP, AND WHY IT IS NOT A DESKTOP-TERMINAL BUG: the two hosts
* do not run the same branch of `kill()`. node-pty defaults `_useConptyDll` to false
* (`windowsPtyAgent.js`). Every desktop site that opens a terminal pane sets it true --
* `local-pty-utils.ts` (two) and `native-pty-spawn.ts` -- as does the `windows-conpty-warmup.ts`
* warm-up, so all of those take the `else` branch, where UPSTREAM ALREADY destroys the input
* socket. The relay passes no such option (`src/relay/pty-handler.ts`), so it takes the
* `!useConptyDll` branch -- the one this asset and the desktop patch both edit.
*
* THE DESKTOP IS NOT ENTIRELY OFF THAT BRANCH. Two desktop sites omit the option and so run it
* too: the hidden rate-limit probes in `src/main/rate-limits/claude-pty.ts` and
* `codex-pty-rate-limit-probe.ts`. Both recur -- their fetchers poll -- and both tear down through
* `kill()`, so this hunk is live on the desktop, just never for a pane a user can see. Do not
* restate this as "the desktop never executes that branch": that sentence stood here for two
* revisions and is false.
*
* What the numbers above therefore do NOT cover: they were measured on relay-style spawn/kill
* cycles. Whether the early placement costs the same +2 File / +1 Process across a probe's
* lifecycle is UNMEASURED -- plausible, not established, and worth measuring before anyone quotes
* a desktop figure. What IS settled is the claim this comment replaced: that the desktop patch made
* every Windows user worse off ON EVERY TERMINAL. Terminals take the DLL branch, and the harness
* that produced that claim defaulted into the branch it was not trying to measure.
*
* The divergence is therefore about which branch each host runs for the workload that matters, not
* about a regression in the terminals users open. The test still pins it, because a future "sync
* the patches" would put the early placement onto the relay's branch, where it does cost +2 File
* and +1 Process per terminal.
*
* If you extend this enumeration, grep for `node-pty` rather than for a static import: those two
* probes were missed three times because they use `await import('node-pty')`.
*
* THE SELF-EXIT LEAK: FIXED FOR THE DESKTOP BY #18635, STILL LIVE ON A RELAY. A terminal that exits
* on its own is also torn down through `kill()` -- both hosts call `destroy()` on natural exit and
* `WindowsTerminal.destroy()` is `kill()` -- but the shell is already gone by then, so the ordering
* this asset relies on does not hold. Measured over 20 self-exit cycles on the NON-DLL branch:
* published +3 File/+1 Process per terminal, desktop patch placement +2/+1, this tree +2/+1. This
* asset does not close it.
*
* #18635 does, in `config/patches/node-pty@1.1.0.patch`: the baton outlives the shell so `PtyKill`
* still reaches `ClosePseudoConsole`, plus an unconditional conout dispose on the DLL branch. That
* fix does not reach a Windows relay, and no hunk in THIS file can carry it, because it is mostly
* NATIVE (`src/win/conpty.cc`) and this asset only rewrites `lib/*.js`. Three delivery paths exist
* and none currently covers Windows:
*
* - the pnpm patch does not cross the SSH boundary -- the remote `npm install` yields upstream's
* unpatched node-pty;
* - the orcad prebuild matrix has no win32 entry (`MATRIX_SLOTS`,
* `config/scripts/build-orcad-prebuilds.mjs`), so no Windows binary is ever compiled from
* patched source to ship;
* - a relay asset CAN patch native source and rebuild on the host -- that is exactly what
* `node-pty-1.1.0-master-cloexec-patch.cjs` does -- but it returns
* `skipped:unsupported-platform` for anything but linux/darwin. Extending it to win32 means
* requiring an MSVC toolchain on the relay host, a far heavier precondition than on Linux,
* where node-gyp already runs at install time.
*
* So a Windows SSH relay still leaks a pseudoconsole per self-exiting terminal, and closing it is a
* DELIVERY problem, not another hunk here. Do not read #18635's flat self-exit relay numbers as
* covering deployed relays: they were measured against a locally rebuilt binary, so they describe
* the relay CODE PATH on a patched tree, not the tree a relay host actually installs.
*/
const EXPECTED_NODE_PTY_VERSION = '1.1.0'
/** Each entry is one published file, its patched form, and the edits between them. */
const PATCH_TARGETS = [
{
relativePath: ['lib', 'windowsPtyAgent.js'],
originalSha256: '8636d16b38266112204061a22b135734177c242837982fd3a4055be726efa64a',
patchedSha256: '1e23ef480569e73706e3ab4f5482c7e553c76f51414ae8e7b0bdcc2fd75f7280',
replacements: [
[
' this._ptyNative.kill(this._pty, this._useConptyDll);\n this._conoutSocketWorker.dispose();\n',
' this._ptyNative.kill(this._pty, this._useConptyDll);\n this._conoutSocketWorker.dispose();\n // Orca: released AFTER the console-list fork and the native kill, not before them.\n // Destroying conin first aborts teardown partway -- measured on a Windows SSH relay\n // as +2 File and +1 Process handles per terminal, against +1 File unpatched.\n this._inSocket.destroy();\n'
]
]
},
{
relativePath: ['lib', 'windowsTerminal.js'],
originalSha256: 'c3a65716f53fed0135a8a633373d5f9c2ab092544d651f27ef0a67096dd3bcd9',
patchedSha256: '8247ecd69be8b18257050fb026b290024612c5ffc6d492ff1d46f81e613be2cf',
replacements: [
[
' _this._agent = new windowsPtyAgent_1.WindowsPtyAgent(file, args, parsedEnv, cwd, _this._cols, _this._rows, false, opt.useConpty, opt.useConptyDll, opt.conptyInheritCursor);\n _this._socket = _this._agent.outSocket;\n // Not available until `ready` event emitted.\n _this._pid = _this._agent.innerPid;',
" _this._agent = new windowsPtyAgent_1.WindowsPtyAgent(file, args, parsedEnv, cwd, _this._cols, _this._rows, false, opt.useConpty, opt.useConptyDll, opt.conptyInheritCursor);\n _this._socket = _this._agent.outSocket;\n // Attach before readiness so a broken ConPTY output pipe cannot be unhandled.\n _this._socket.on('error', function (err) {\n var code = err && err.code;\n // PTY output can report EPIPE before `_close()` wins the race.\n _this._close();\n if (code === 'EPIPE' || code === 'ERR_STREAM_PUSH_AFTER_EOF' || code === 'ERR_STREAM_DESTROYED') {\n return;\n }\n // EIO, happens when someone closes our child process: the only process\n // in the terminal.\n // node < 0.6.14: errno 5\n // node >= 0.6.14: read EIO\n if (typeof code === 'string') {\n if (~code.indexOf('errno 5') || ~code.indexOf('EIO'))\n return;\n }\n // Throw anything else.\n if (_this.listeners('error').length < 2) {\n throw err;\n }\n });\n // Not available until `ready` event emitted.\n _this._pid = _this._agent.innerPid;"
],
[
" }\n });\n // Shutdown if `error` event is emitted.\n _this._socket.on('error', function (err) {\n // Close terminal session.\n _this._close();\n // EIO, happens when someone closes our child process: the only process\n // in the terminal.\n // node < 0.6.14: errno 5\n // node >= 0.6.14: read EIO\n if (err.code) {\n if (~err.code.indexOf('errno 5') || ~err.code.indexOf('EIO'))\n return;\n }\n // Throw anything else.\n if (_this.listeners('error').length < 2) {\n throw err;\n }\n });\n // Cleanup after the socket is closed.\n _this._socket.on('close', function () {",
" }\n });\n // Cleanup after the socket is closed.\n _this._socket.on('close', function () {"
],
[
' _this._readable = true;\n _this._writable = true;\n _this._forwardEvents();\n return _this;',
" _this._readable = true;\n _this._writable = true;\n // A ConPTY input-pipe error must retire only this terminal. Without a listener, Node promotes\n // errors such as write EAGAIN to uncaughtException and kills every PTY in the daemon.\n _this._agent.inSocket.on('error', function () {\n if (!_this._writable) {\n return;\n }\n _this._close();\n try {\n _this._agent.kill();\n }\n catch (_a) {\n // The failing pipe may have raced process exit; the terminal is already unwritable.\n }\n });\n _this._forwardEvents();\n return _this;"
],
[
'exports.WindowsTerminal = WindowsTerminal;\n//# sourceMappingURL=windowsTerminal.js.map',
'exports.WindowsTerminal = WindowsTerminal;\n//# sourceMappingURL=windowsTerminal.js.map\n'
]
]
}
]
function inspectTarget(relayDir, target) {
const nodePtyDir = resolve(relayDir, 'node_modules', 'node-pty')
const packageJson = JSON.parse(readFileSync(join(nodePtyDir, 'package.json'), 'utf8'))
if (packageJson.version !== EXPECTED_NODE_PTY_VERSION) {
throw new Error(
`Refusing to patch node-pty ${packageJson.version}; expected ${EXPECTED_NODE_PTY_VERSION}`
)
}
const filePath = join(nodePtyDir, ...target.relativePath)
return { filePath, source: readFileSync(filePath, 'utf8') }
}
function assertPatchedNodePtyWindowsTeardown(relayDir = process.cwd()) {
for (const target of PATCH_TARGETS) {
const inspected = inspectTarget(relayDir, target)
if (sourceSha256(inspected.source) !== target.patchedSha256) {
throw new Error(
`node-pty ConPTY teardown release is not installed in ${target.relativePath.join('/')}`
)
}
}
}
function patchNodePtyWindowsTeardown(relayDir = process.cwd()) {
for (const target of PATCH_TARGETS) {
const inspected = inspectTarget(relayDir, target)
const sourceHash = sourceSha256(inspected.source)
if (sourceHash === target.patchedSha256) {
continue
}
if (sourceHash !== target.originalSha256) {
throw new Error(
`Refusing to patch unexpected node-pty source in ${target.relativePath.join('/')}`
)
}
let patchedSource = inspected.source
for (const [from, to] of target.replacements) {
// Why the count check: an anchor that matched twice would patch the wrong site silently, and
// the hash below would then reject a tree this script had already rewritten.
if (patchedSource.split(from).length - 1 !== 1) {
throw new Error(`Refusing to patch ${target.relativePath.join('/')}; anchor is not unique`)
}
patchedSource = patchedSource.replace(from, to)
}
const temporaryPath = `${inspected.filePath}.orca-patch-${process.pid}`
// Why: a terminated remote install must leave either known source version recoverable on reconnect.
try {
writeFileSync(temporaryPath, patchedSource)
renameSync(temporaryPath, inspected.filePath)
} finally {
rmSync(temporaryPath, { force: true })
}
}
assertPatchedNodePtyWindowsTeardown(relayDir)
}
function sourceSha256(source) {
return createHash('sha256').update(source).digest('hex')
}
if (require.main === module) {
patchNodePtyWindowsTeardown()
}
module.exports = {
assertPatchedNodePtyWindowsTeardown,
patchNodePtyWindowsTeardown
}
File diff suppressed because one or more lines are too long

Some files were not shown because too many files have changed in this diff Show More