mirror of
https://github.com/stablyai/orca.git
synced 2026-10-01 00:02:10 +00:00
Merge origin/main (PR #18902 typed dispatch refusals) into orchestration-v3
Ported the typed refusals onto v3's file layout: runs/dispatch-methods.ts, gates/gates.ts, the dispatch store, and worker-dispatch-start. Dropped v3's inject-rejection-message module in favour of the shared refusal contract. The receipt-code table lives in references/recovery-and-cleanup.md since the kernel routes refused starts there.
This commit is contained in:
@@ -15,8 +15,13 @@ on:
|
||||
# Why: this job holds the only checks that load the Fastfile, so edits to
|
||||
# it or to the release workflow it guards must re-run them.
|
||||
- '.github/workflows/mobile.yml'
|
||||
- '.github/actions/install-node-dependencies/**'
|
||||
- '.github/workflows/mobile-ios-release.yml'
|
||||
|
||||
concurrency:
|
||||
group: mobile-${{ github.event.pull_request.number || github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
verify:
|
||||
runs-on: ubuntu-latest
|
||||
@@ -35,10 +40,7 @@ jobs:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v6
|
||||
with:
|
||||
node-version-file: package.json
|
||||
- uses: ./.github/actions/install-node-dependencies
|
||||
|
||||
# bundler-cache installs mobile/Gemfile.lock, so this job is also what
|
||||
# proves the pinned fastlane the release workflow depends on still
|
||||
@@ -50,23 +52,6 @@ jobs:
|
||||
bundler-cache: true
|
||||
working-directory: mobile
|
||||
|
||||
- name: Setup pnpm
|
||||
uses: pnpm/setup@v2
|
||||
with:
|
||||
install: false
|
||||
|
||||
# Why: the mobile typecheck imports shared types from ../src/shared, and
|
||||
# some of those files import runtime deps (tweetnacl, ws) resolved from
|
||||
# the repo-root node_modules. Without a root install, tsc fails with
|
||||
# "Cannot find module 'tweetnacl'/'ws'". Mobile is a separate pnpm project
|
||||
# (not in the root workspace), so this is a distinct install.
|
||||
# --ignore-scripts skips the root postinstall (Electron native-module
|
||||
# rebuild) which is irrelevant to a type-only check and would only add
|
||||
# time and failure surface on this ubuntu mobile runner.
|
||||
- name: Install root dependencies
|
||||
working-directory: .
|
||||
run: pnpm install --frozen-lockfile --ignore-scripts
|
||||
|
||||
- name: Install dependencies
|
||||
run: pnpm install --frozen-lockfile
|
||||
|
||||
|
||||
+46
-62
@@ -41,6 +41,10 @@ jobs:
|
||||
managed_hook_node18: ${{ steps.filter.outputs.managed_hook_node18 }}
|
||||
package: ${{ steps.filter.outputs.package }}
|
||||
package_windows: ${{ steps.filter.outputs.package_windows }}
|
||||
e2e_should_run: ${{ steps.e2e_filter.outputs.should_run }}
|
||||
test_files: ${{ steps.e2e_filter.outputs.test_files }}
|
||||
ssh_source_changed: ${{ steps.e2e_filter.outputs.ssh_source_changed }}
|
||||
native_ime_source_changed: ${{ steps.e2e_filter.outputs.native_ime_source_changed }}
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v6
|
||||
@@ -66,6 +70,37 @@ jobs:
|
||||
printf '%s\n' "$CHANGED"
|
||||
printf '%s\n' "$CHANGED" | node config/scripts/pr-code-change-scope.mjs | tee -a "$GITHUB_OUTPUT"
|
||||
|
||||
# Reuse the path-detector checkout instead of queuing another runner.
|
||||
- name: Filter changed E2E specs
|
||||
id: e2e_filter
|
||||
if: github.event.pull_request.draft != true && steps.filter.outputs.should_run == 'true'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
BASE="${{ github.event.pull_request.base.sha }}"
|
||||
HEAD="${{ github.event.pull_request.head.sha }}"
|
||||
CHANGED="$(git diff --name-only --diff-filter=AMCR --merge-base "$BASE" "$HEAD")"
|
||||
# Source routes are executable contracts so a test can prove exact
|
||||
# authorities, exclusions, and sentinels without evaluating workflow shell.
|
||||
TEST_FILES_JSON="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs)"
|
||||
echo "test_files=$TEST_FILES_JSON" >> "$GITHUB_OUTPUT"
|
||||
# Why a separate signal: the Docker-SSH lane must trigger on SSH source, not on a
|
||||
# spec name surviving in a route's list. Same routes, so the two cannot drift.
|
||||
SSH_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --ssh-source)"
|
||||
echo "ssh_source_changed=$SSH_SOURCE_CHANGED" >> "$GITHUB_OUTPUT"
|
||||
echo "SSH source changed: $SSH_SOURCE_CHANGED"
|
||||
# Why its own signal: the real-IME lane is a whole ibus session, not a spec, so it must
|
||||
# trigger on IME source rather than on a spec name in some route's list.
|
||||
NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)"
|
||||
echo "native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED" >> "$GITHUB_OUTPUT"
|
||||
echo "Native IME source changed: $NATIVE_IME_SOURCE_CHANGED"
|
||||
if [ "$TEST_FILES_JSON" != '[]' ]; then
|
||||
echo "should_run=true" >> "$GITHUB_OUTPUT"
|
||||
echo "Changed E2E specs: $TEST_FILES_JSON"
|
||||
else
|
||||
echo "should_run=false" >> "$GITHUB_OUTPUT"
|
||||
echo "No changed E2E specs"
|
||||
fi
|
||||
|
||||
static_analysis:
|
||||
name: static analysis
|
||||
needs: [code_paths]
|
||||
@@ -712,7 +747,11 @@ jobs:
|
||||
- name: Package unpacked app
|
||||
env:
|
||||
ORCA_REUSE_PREPARED_NATIVE_RUNTIME: '1'
|
||||
run: pnpm exec electron-builder --config config/electron-builder.config.cjs --linux AppImage deb rpm --x64 --publish never
|
||||
# PR artifacts are only inspected locally; gzip avoids release-size xz compression.
|
||||
run: >-
|
||||
pnpm exec electron-builder --config config/electron-builder.config.cjs
|
||||
--linux AppImage deb rpm --x64 --publish never
|
||||
--config.deb.compression=gz --config.rpm.compression=gzip
|
||||
|
||||
- name: Verify root-package marker payloads
|
||||
run: |
|
||||
@@ -861,65 +900,10 @@ jobs:
|
||||
- name: Smoke packaged CLI
|
||||
run: node config/scripts/smoke-packaged-cli.mjs --app-dir=dist/win-unpacked
|
||||
|
||||
# Why: PR E2E is advisory and only validates changed specs; scheduled and
|
||||
# release runs retain full-suite coverage.
|
||||
e2e-paths:
|
||||
name: detect changed e2e specs
|
||||
needs: [code_paths]
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event.pull_request.draft != true && needs.code_paths.outputs.should_run == 'true'
|
||||
# Why: detector only needs to read the checkout; do not inherit repo defaults.
|
||||
permissions:
|
||||
contents: read
|
||||
outputs:
|
||||
should_run: ${{ steps.filter.outputs.should_run }}
|
||||
test_files: ${{ steps.filter.outputs.test_files }}
|
||||
ssh_source_changed: ${{ steps.filter.outputs.ssh_source_changed }}
|
||||
native_ime_source_changed: ${{ steps.filter.outputs.native_ime_source_changed }}
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
# Why blob:none: full history is needed for the merge-base diff, but historical
|
||||
# file contents are not. Blobs are ~89% of this repo's pack, and Git fetches the
|
||||
# few this job actually reads on demand.
|
||||
fetch-depth: 0
|
||||
filter: blob:none
|
||||
persist-credentials: false
|
||||
|
||||
- name: Filter changed E2E specs
|
||||
id: filter
|
||||
run: |
|
||||
set -euo pipefail
|
||||
BASE="${{ github.event.pull_request.base.sha }}"
|
||||
HEAD="${{ github.event.pull_request.head.sha }}"
|
||||
CHANGED="$(git diff --name-only --diff-filter=AMCR --merge-base "$BASE" "$HEAD")"
|
||||
# Source routes are executable contracts so a test can prove exact
|
||||
# authorities, exclusions, and sentinels without evaluating workflow shell.
|
||||
TEST_FILES_JSON="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs)"
|
||||
echo "test_files=$TEST_FILES_JSON" >> "$GITHUB_OUTPUT"
|
||||
# Why a separate signal: the Docker-SSH lane must trigger on SSH source, not on a
|
||||
# spec name surviving in a route's list. Same routes, so the two cannot drift.
|
||||
SSH_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --ssh-source)"
|
||||
echo "ssh_source_changed=$SSH_SOURCE_CHANGED" >> "$GITHUB_OUTPUT"
|
||||
echo "SSH source changed: $SSH_SOURCE_CHANGED"
|
||||
# Why its own signal: the real-IME lane is a whole ibus session, not a spec, so it must
|
||||
# trigger on IME source rather than on a spec name in some route's list.
|
||||
NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)"
|
||||
echo "native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED" >> "$GITHUB_OUTPUT"
|
||||
echo "Native IME source changed: $NATIVE_IME_SOURCE_CHANGED"
|
||||
if [ "$TEST_FILES_JSON" != '[]' ]; then
|
||||
echo "should_run=true" >> "$GITHUB_OUTPUT"
|
||||
echo "Changed E2E specs: $TEST_FILES_JSON"
|
||||
else
|
||||
echo "should_run=false" >> "$GITHUB_OUTPUT"
|
||||
echo "No changed E2E specs"
|
||||
fi
|
||||
|
||||
e2e:
|
||||
name: e2e
|
||||
needs: e2e-paths
|
||||
if: needs.e2e-paths.outputs.should_run == 'true'
|
||||
needs: code_paths
|
||||
if: needs.code_paths.outputs.e2e_should_run == 'true'
|
||||
# Why: reusable e2e.yml only checkouts, builds, and uploads artifacts.
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -928,8 +912,8 @@ jobs:
|
||||
# The synthetic pull-request merge ref can disappear while this reusable
|
||||
# workflow is queued. The head SHA is immutable and works for every PR.
|
||||
ref: ${{ github.event.pull_request.head.sha }}
|
||||
test_files: ${{ needs.e2e-paths.outputs.test_files }}
|
||||
ssh_source_changed: ${{ needs.e2e-paths.outputs.ssh_source_changed }}
|
||||
test_files: ${{ needs.code_paths.outputs.test_files }}
|
||||
ssh_source_changed: ${{ needs.code_paths.outputs.ssh_source_changed }}
|
||||
|
||||
# Why this is not in verify's needs: it is the first PR-gate run of a harness whose reliability
|
||||
# is only known from nightly main runs (20/20 green, 2026-08-09..2026-08-29, p50 3m25s). It
|
||||
@@ -939,8 +923,8 @@ jobs:
|
||||
# require `success || skipped` outside the strict loop — see the note on `e2e`.
|
||||
terminal_ime_native:
|
||||
name: real IME
|
||||
needs: e2e-paths
|
||||
if: needs.e2e-paths.outputs.native_ime_source_changed == 'true'
|
||||
needs: code_paths
|
||||
if: needs.code_paths.outputs.native_ime_source_changed == 'true'
|
||||
# Why: the reusable workflow only checks out, builds, and uploads artifacts.
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -22,6 +22,10 @@ on:
|
||||
- main
|
||||
paths: *skill-roundtrip-paths
|
||||
|
||||
concurrency:
|
||||
group: skill-roundtrip-${{ github.event_name }}-${{ github.event.pull_request.number || github.ref }}
|
||||
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||
|
||||
jobs:
|
||||
roundtrip:
|
||||
strategy:
|
||||
|
||||
@@ -0,0 +1,189 @@
|
||||
# Relay improvement: implementation checklist, lanes, and disruption
|
||||
|
||||
Companion to [`relay-improvement-roadmap-2026-09.md`](./relay-improvement-roadmap-2026-09.md) (item numbers
|
||||
match). This file answers three questions per item: what are the concrete steps, what can run in parallel,
|
||||
and will a user notice.
|
||||
|
||||
## Status as of 2026-09-04 22:30Z
|
||||
|
||||
Three buckets. "Merged" means the code is on `main` and nothing in production has changed yet. "Deployed" means users are already getting it. "Awaiting owner" means I will not touch production without a go.
|
||||
|
||||
**Deployed to production**
|
||||
- Auth instance cap 20 + dead-family audit fix (orca-cloud #474) as revision `orca-cloud-auth-00031-tox`.
|
||||
- Dynamic NAT ports in both regions (stablyai/orca #18693). Zero drops and zero proxy dial errors since.
|
||||
- Nine alert policies with log metrics: 4 auth (#475), 3 relay Cloud SQL/NAT (#18693), 1 cell process-exit (#18717), all on the relay Slack channel.
|
||||
|
||||
**Merged, ships with the next relay cell image roll (Roll 1 carries `519f4914`; Roll 2 needs a fresh image build)**
|
||||
- Per-cell inventory locks, delta counters, pool `statement_timeout` (#18722). Roll 2.
|
||||
- Cells dial Cloud SQL with `--private-ip` when configured (#18720). Inert until 2.1 applies.
|
||||
- Phone shows a clear "sign in on the desktop again" state when the desktop is signed out (#18698).
|
||||
|
||||
**Merged, ships with the next auth deploy**
|
||||
- Refresh rotation grace window (orca-cloud #478). Startup adds one nullable column (brief exclusive lock on `refresh_tokens`).
|
||||
- Pruning job code (orca-cloud #476) is in the image; the job itself is Terraform-disabled until 1.2.
|
||||
|
||||
**Merged, ships with the next desktop release**
|
||||
- Never replay a refresh token after a timeout; ±10 % jitter on relay lease renewal (#18719).
|
||||
- Renderer learns when a cloud session is revoked (#18694).
|
||||
|
||||
**Merged, not applied**
|
||||
- Incident dashboard (#18717) blocked behind the runtime-metric label drift (5.x first item).
|
||||
- Monitor probe fix (#18723) is live in the workflow; the same-cap roll gate has not yet produced a green dry-run since.
|
||||
|
||||
**Awaiting owner go (production mutations)**
|
||||
1. Roll 1 cell image roll (1.1): dry-run gate, then c8 canary, then batches.
|
||||
2. Auth deploy carrying #478 (3.1): quiet minute for the column add.
|
||||
3. orca-cloud #477 private IP (2.1): merge arms an instance restart and a one-way door. Recommendation: hold.
|
||||
4. Runtime-metric `region` label drift (5.x): intentional replacement of 21 metrics, or drop the label.
|
||||
5. Enable pruning (1.2): first budget 20k rows; needs a Terraform apply.
|
||||
6. Paging channel for auth alerts (5.2): needs the destination from you.
|
||||
|
||||
**Open code follow-ups (no gate, nobody assigned)**
|
||||
- Monitor summary Markdown does not render `tolerated: true` continuity events (added by #18798); the state artifact has them, the checkpoint table does not.
|
||||
- Relay container boot races the `cloud-sql-proxy` sidecar: c13's fresh container exited twice (`applyPostgresSchema` connection timeout, 2 s each) before the proxy was listening. Make schema apply wait for the proxy or order the containers.
|
||||
- `cloud-deploy-relay-production-capacity-job.yml` (~line 416) has the same wave-0 single-shot preflight carve-out that #18778 removes from the same-cap job; its single-evidence path never retries freshness-only failures.
|
||||
- `cloud/package.json` `test` names every dev-script test file explicitly; an unregistered `*.test.mjs` is silently never run in CI (found by #18769). Needs a glob or a ratchet that fails on an unlisted test file.
|
||||
- Same-cap job's verify step uses bare `curl --fail-with-body` against the just-rolled cell; one 503 at the LB warm-up edge failed c8 canary #2 (run 33935407461) after the transition verifier had already passed. Needs a bounded retry, same rule as #18723/#18740.
|
||||
- `verify-mutation` in `cloud-deploy-relay-production.yml`, the multi-target workflow, and the capacity workflow still binds to an exact commit; same exposure #18754 fixed for the same-cap and rehome paths.
|
||||
- `incident-live-preflight-cli.ts` reports only `source/code` (`active-probe/threshold_max`) with no signal name or observed value, so a failed mutation preflight (c27 recovery #3, run 33986948522) cannot be attributed to an endpoint without an out-of-band probe. Print the signal and observed/threshold pair. Related: the 2 000 ms `endpointLatencyMs` bar is shared by US and Asia cells while Asia /health round trips from a US runner sit at 0.7–1.3 s idle; consider a per-region bar or the p50 of the gate window instead of one shot. Gates #44 and #45 (2026-09-05) both froze on `cell.production-gce-c27.latency_ms` at 2.6–2.7 s with c28 showing the identical tail under operator probes; the bar is now blocking Asia rolls. **Fix: stablyai/orca #18877** (per-region `cellEndpointLatencyMs`, us-central1 2 000 / asia-east2 4 000, plus signal/observed/threshold in preflight messages). Residual: `probeEndpointHealth` in `resource-inventory.ts` still uses the flat 2 000 bar to decide whether to retry after the 10 s readiness-cache wait, so a healthy Asia cell over 2 s costs one extra probe per sample (latency, not verdict); thread the region bar into the retry decision.
|
||||
- The root oxlint config ignores `cloud/**`, so `check:code-quality:changed` never inspects relay-ops or the cloud dev scripts; typecheck + vitest is the only gate there.
|
||||
- Monitor bars that froze on non-health today: `directorInstancesMin: 5` with `latest-sum` (one-minute instance recycle), `endpointLatencyMs: 2000` on a US-runner probe to asia-east2, `cloudDataMaxAgeMs: 180000` vs Cloud Monitoring publish lag up to 255 s. Recalibrate with a week of data.
|
||||
- `parsed()` in `resource-inventory.ts` still returns null on a 200 with a malformed MIG body; a second path to `runtime_power_unknown`.
|
||||
- Deploy script strips `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` on every release (3.1 first item).
|
||||
- `assignOnce` placement lock still global (4.1 remainder).
|
||||
- Region preference (4.2), retries-bar recalibration after a week of Roll 2 data (4.4), pruner `stopReason` alert (1.5).
|
||||
- Full apps-root apply for 4 unrelated drifts (1.4), from a host with the 1Password account.
|
||||
|
||||
## Uplift ranking (reliability gained per unit of effort)
|
||||
|
||||
| Rank | Item | Why it ranks here |
|
||||
|---|---|---|
|
||||
| 1 | 1.1 cell image roll | Removes the only crash mode we have seen in production. 22 of 23 cells still have it. One afternoon. |
|
||||
| 2 | 3.1 refresh rotation grace window | Turns the entire "slow auth → mass sign-out" class into a slowdown. One day. |
|
||||
| 3 | 4.1 inventory lock contention | The floor under every 503 and slow phone accept, every day, not just incidents. One week. |
|
||||
| — | 2.2 relay/auth database split | **Deferred 2026-09-04** to ~2026-11-01. Biggest structural fix, but the concrete cause is fixed and alerts now page; see roadmap 2.2 for re-open triggers. |
|
||||
| 4 | 1.2 + 1.3 pruning and reclaim | Defuses the 63 M-row time bomb. Low effort, mostly waiting. |
|
||||
| 5 | 5.1 + 5.2 crash alert, page a human | Cheapest detection uplift; today's incident ran 4 h unpaged. |
|
||||
| 6 | 2.1 private IP | Durable version of a fix that already landed (dynamic NAT ports). Do it on the existing instance. |
|
||||
| 7 | 4.3 + 3.2 desktop hardening | Small, ride the normal desktop release. |
|
||||
| 8 | 4.2, 4.4, 5.4, 1.4, 1.5 | Housekeeping and quality-of-life. |
|
||||
|
||||
## The shared bottleneck: cell rolls
|
||||
|
||||
Every change to what runs on a cell (image, proxy flag, env, relay code) needs a same-cap roll: drain →
|
||||
recreate → verify, one wave at a time, gated by the 15-minute monitor, about an afternoon. Each wave forces
|
||||
the desktops on that cell to re-dial (c7 canary: 807 controls re-dialed in ~10 s) and phones on those
|
||||
desktops reconnect on their normal retry. Users see a few seconds of "reconnecting" per wave.
|
||||
|
||||
So batch. Two rolls, not five:
|
||||
|
||||
- **Roll 1 (now):** current image only (1.1). Do not wait for anything else.
|
||||
- **Roll 2 (week 2–3):** proxy `--private-ip` (2.1) + relay pool `statement_timeout` (2.3) + lock-contention
|
||||
fix (4.1), all in one image/template. Prerequisite: 2.1's peering and private IP exist first.
|
||||
|
||||
## Lanes (independent; different people can own them)
|
||||
|
||||
```
|
||||
Lane A data plane 1.1 roll ──────────────────► Roll 2 (2.1 flag + 2.3 + 4.1) ──► 4.4 recalibrate
|
||||
Lane B auth/DB 1.2 enable pruning ──(10 d)──► 1.3 reclaim 3.1 grace window (any time)
|
||||
Lane C network 2.1 peering + private IP ─────┐ (feeds Roll 2) (2.2 DB split deferred)
|
||||
Lane D desktop 3.2 no same-token retry, 4.3 lease jitter (any release; wire-compatible)
|
||||
Lane E observability 1.5, 5.1, 5.2, 5.4 (Terraform only, any time)
|
||||
Lane F director 4.2 region preference (Cloud Run deploy, any time)
|
||||
Misc 1.4 full apps-root apply (any time; see its check)
|
||||
```
|
||||
|
||||
Hard dependencies: Roll 2 waits on 2.1's network work; 1.3 waits on 1.2 finishing. Everything else is
|
||||
independent. (2.2 deferred; if revived, do it after 2.1 so the new instance is private from day one.)
|
||||
|
||||
## Disruption summary
|
||||
|
||||
| Item | User-visible? | What they see | Mitigation |
|
||||
|---|---|---|---|
|
||||
| 1.1 / Roll 2 | **Yes, transient** | Per wave, desktops on that cell reconnect within seconds; phones follow on retry. | Waves gated by the monitor; run in the US night. Already rehearsed on c7. |
|
||||
| 1.2 pruning | No | Background deletes, 5k rows per batch. | Small first budget; watch `stopReason` and Cloud SQL write throughput. Stop the scheduler if checkpoint alerts fire. |
|
||||
| 1.3 reclaim | **Depends on tool** | `VACUUM FULL` takes an exclusive lock on `refresh_tokens`: sign-in and refresh block for its duration (minutes to tens of minutes on 16 GB). `pg_repack` holds only brief locks. | Use `pg_repack`. If VACUUM FULL, announce a maintenance window. |
|
||||
| 1.4 full apps apply | Should be none, **verify** | Terraform will create a new auth revision (env added). Traffic is pinned to `00031-tox` by name, so the new revision should receive 0 %. | Confirm in the plan that no `traffic` change appears. If it does, stop: the Terraform image variable is not the serving image. |
|
||||
| 1.5, 5.x alerts | No | | |
|
||||
| 2.1 private IP | **Yes, certain** | Google: "Configuring an existing Cloud SQL instance to use private IP causes the instance to restart, resulting in downtime." No in-place path, HA does not avoid it. Expect 1–2 min DB unavailability: sign-in fails, relay renewals retry. **One-way door**: private IP cannot be disabled and the VPC link cannot be removed once set. The proxy flag change rides Roll 2. | Off-peak; only after Roll 1 (old image dies on a 2 min DB blip). Owner decision required before the foundation apply. |
|
||||
| 2.2 DB split (deferred) | **Yes, scheduled** | Relay unavailable for the cutover (drain all cells → copy relay tables → flip `DATABASE_URL` → restart). Minutes if rehearsed. Desktops and phones reconnect automatically after. | Rehearse on staging; do it in the US night; announce. |
|
||||
| 2.3 statement timeout | No beyond Roll 2 | | |
|
||||
| 3.1 grace window | No | Auth deploys are no-traffic candidate → smoke → promote. | Security trade-off: a stolen token replayed inside the window is served once instead of revoking. 60 s is the usual choice. |
|
||||
| 3.2, 4.3 desktop | No | Normal app update. | |
|
||||
| 4.1 lock fix | No beyond Roll 2 | | Verify against real Postgres on 55440 with concurrent probes before shipping. |
|
||||
| 4.2 region preference | **Minor, Asia users** | Phones that start being placed in Asia reconnect once to a nearer cell. | Roll out behind the existing region-preference flag. |
|
||||
| 4.4 | No | | |
|
||||
|
||||
## Checklists
|
||||
|
||||
### 1.1 Cell image roll (Roll 1)
|
||||
- [x] Confirm fleet is quiet: 15-min monitor dry-run passes. #19 green 23:07:53Z (run 33927238469). Canary then failed the evidence provenance check because main moved during the gate; re-gating with a same-commit chain.
|
||||
- [x] Confirm director is on 519f4914 and c7 on 85bf6799 (confirmed 2026-09-04 via instance-template census; 20 serving cells still on `5aedbca5`) (`verify` mode of the same-cap workflow).
|
||||
- [x] Dispatch `cloud-deploy-relay-production-same-cap` waves per the plan in the findings doc; one wave, verify, next. Done 2026-09-05 01:14Z–22:27Z: c8 canary, US batches c9–c10, c13–c16, c19–c26 at protocol 1, then Asia c27 (recovered via `mode=rollback` re-entry after gate freezes on the flat latency bar, fixed by #18877), c28, c29 as single-cell canaries at protocol 0.
|
||||
- [x] After each wave: the transition verifier passed at migration-only and again at general on every cell (assignments carried, heartbeat fresh, hard cap 3 000); no `container die` fleet-wide across the whole roll. The 4408/1006 burst per wave was not measured separately; the verifier's assignment count before and after each restart is the recovery evidence recorded.
|
||||
- [x] Record image census in the findings doc. 2026-09-05 22:27Z: all 19 general cells on `519f4914` except c7 on `85bf6799`; existing-only c1–c6, c11, c12 and migration-only c17, c18 untouched on their older images by design. Selector at gen 148.
|
||||
|
||||
### 1.2 Enable pruning
|
||||
- [x] `auth_token_pruner_image` = digest of `orca-cloud-auth-00031-tox` (`343a0915…`; it contains the entrypoint). orca-cloud #479 merged.
|
||||
- [x] `auth_token_pruner_enabled = true`, `auth_token_pruner_max_rows_per_run = 20000` for the first day (orca-cloud #479).
|
||||
- [x] Targeted plan asserted 9 create / 0 change / 0 destroy. Applied 2026-09-05 02:06Z.
|
||||
- [x] Trigger one run by hand; read the summary event. 02:18Z: `time-budget`, 73 batches, 365k scanned, 1 040 deleted (1 021 revoked, 19 expired), no errors. Scan-bound.
|
||||
- [ ] Raise the budget to the default 200k after a clean day; watch Cloud SQL write MB/s and the checkpoint alert.
|
||||
- [ ] 1.5: log metric + policy on `stopReason != complete`.
|
||||
|
||||
### 1.3 Reclaim
|
||||
- [ ] Wait for steady-state runs deleting ~0 rows.
|
||||
- [ ] `pg_repack -t refresh_tokens` off-peak (needs the extension; check `pg_available_extensions`). Not `VACUUM FULL` without a window.
|
||||
- [ ] Confirm table + index size and `disk/utilization` dropped.
|
||||
|
||||
### 1.4 Full apps-root apply
|
||||
- [ ] Run from CI or a host with the 1Password account (local plan fails on the Cloudflare data source).
|
||||
- [ ] Plan shows exactly the four known drifts and **no traffic change** on `google_cloud_run_v2_service.auth`.
|
||||
- [ ] Apply; confirm `status.traffic` still pins `00031-tox` at 100 %.
|
||||
|
||||
### 2.1 Private IP (PRs open: orca-cloud #477 foundation, stablyai/orca #18720 relay flag)
|
||||
- [ ] **Owner decision**: the foundation apply restarts the instance and is irreversible on Google's side. Merging #477 arms the next foundation apply; hold the merge until the window is chosen.
|
||||
- [ ] Director is out of scope: it uses the Cloud Run built-in connector (managed Google path, not the relay VPC NAT), so it consumed none of the exhausted ports; moving it needs Direct VPC egress + a separate DSN secret. Own PR if ever wanted.
|
||||
- [ ] Step 7 (`ipv4_enabled=false`) is blocked until humans have IAP/bastion access and the director is moved; it breaks both today.
|
||||
- [ ] Allocate a `/24` private services range on the relay VPC; `google_service_networking_connection`.
|
||||
- [ ] Add `ip_configuration.private_network` to `google_sql_database_instance.auth` (foundation root). Plan must show update, not replace.
|
||||
- [ ] Apply off-peak; expect a possible restart. Watch auth 5xx alert and relay `sqlFailures`.
|
||||
- [ ] Cell template: proxy args add `--private-ip` (code merged #18720; flag not set). Director: Direct VPC egress or connector, then the same flag. Both ride Roll 2.
|
||||
- [ ] After Roll 2: NAT `port_usage` for relay gateways drops to ~0; then consider `ipv4_enabled = false` (removes the public IP; breaks the local `cloud-sql-proxy --token` workflow unless it also goes private).
|
||||
|
||||
### 2.2 Database split (deferred to ~2026-11-01; checklist kept for when it is revived)
|
||||
- [ ] New `google_sql_database_instance.relay` (private IP from day one, its own size and flags). Staging first.
|
||||
- [ ] Relay schema applies cleanly to an empty instance (it does at startup).
|
||||
- [ ] Rehearsal on staging: drain → `pg_dump` relay tables → restore → flip `relay_database_url` secret → restart director + cells → phones/desktops reconnect. Time it.
|
||||
- [ ] Production: announce a window; same steps; verify `orca_relay_runtime_metrics` controls recover to pre-cutover count.
|
||||
- [ ] Update `production-cloud-sql-app-consumers` budget test and both alert policies' `database_id`.
|
||||
|
||||
### 2.3 Relay pool statement timeout (merged stablyai/orca #18722; ships Roll 2)
|
||||
- [x] `statement_timeout` on the relay `pg.Pool` (5 s, env-configurable; schema pool untimed; `57014` retryable), below the control-renewal deadline; DDL on an untimed connection (same pattern as auth #476).
|
||||
- [x] Postgres test on 55440: a held lock fails the query fast and the bounded retry takes over.
|
||||
|
||||
### 3.1 Refresh rotation grace window (orca-cloud #478 merged 2026-09-04; deploy pending owner go)
|
||||
- [ ] Fix the deploy-script env strip for `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` (pre-existing; found by #478).
|
||||
- [x] `rotateRefreshToken`: if `rotated_at` within 60 s and not revoked, return the existing successor (idempotent), no revoke, no audit.
|
||||
- [x] Outside the window or a third presentation: unchanged (revoke + audit).
|
||||
- [x] Tests: replay inside window returns same successor; outside revokes; concurrent double-present yields one successor.
|
||||
- [x] Deploy via `deploy-auth-production` (candidate → smoke → promote). Deployed 2026-09-04 23:15Z as `orca-cloud-auth-00035-gos`, cap 20 kept, 0 5xx; `successor_material` column present; sealed successors being written. (candidate → smoke → promote).
|
||||
|
||||
### 3.2 / 4.3 Desktop (merged stablyai/orca #18719; ships next desktop release)
|
||||
- [x] 3.2: on refresh timeout, re-read stored session before retrying; do not re-send a token already rotated locally.
|
||||
- [x] 4.3: ±10 % jitter on control lease renewal; unit test on the distribution; wire-compatible (server accepts early renewals already).
|
||||
|
||||
### 4.1 Lock contention (partial: stablyai/orca #18722 merged; ships Roll 2)
|
||||
- [x] Replace the global `FOR UPDATE` over `relay_cells` with per-cell row locks; counters delta-only. Remaining: `assignOnce` placement lock is still global (optimistic snapshot follow-up). with per-cell row locks or `pg_advisory_xact_lock(cell)`; counters delta-only.
|
||||
- [x] Postgres tests on 55440 with concurrent probes (in #18722). Staging load run still owed; `postgres_retries` per hour drops in staging load run.
|
||||
- [ ] Ships in Roll 2; then 4.4 recalibrates the retries bar from a week of data.
|
||||
|
||||
### 4.2 Region preference
|
||||
- [ ] Director: honor requested region when the preferred region has headroom, else sticky. Behind the existing flag.
|
||||
- [ ] Measure with `orca_relay_runtime_metrics` region counters before/after.
|
||||
|
||||
### 5.x Observability
|
||||
- [x] **Relay-root runtime-metric drift**: resolved by dropping the `region` label to match live state (stablyai/orca #18734). Applied 2026-09-04 23:11Z: 8 never-applied `control_*` renewal metrics + the incident dashboard created, 0 destroyed, 21 live metrics untouched.
|
||||
- [x] 5.1 `container die` log metric per cell (`relay_cell_process_exit`, applied 2026-09-04 via #18717), > 3 / 15 min, relay channel.
|
||||
- [ ] 5.2 Add a paging channel (**needs owner input**: destination) to `auth_alert_notification_channels` for refresh rejections + latency.
|
||||
- [x] 5.4 One dashboard (applied 2026-09-04 23:11Z): `orca_relay_cloud_sql_wal_checkpoint`, NAT drops, `orca_auth_refresh_401`, summed `controls`.
|
||||
@@ -0,0 +1,67 @@
|
||||
# Relay improvement roadmap (written 2026-09-04, after the auth/relay outage)
|
||||
|
||||
Owner-facing list of what is left to make the relay more robust, in priority order. Evidence and history
|
||||
for every item is in [`relay-reconnect-2026-09-findings.md`](./relay-reconnect-2026-09-findings.md)
|
||||
(Findings 1–13). Everything already landed on 2026-09-04 is listed at the end so this file is complete on
|
||||
its own.
|
||||
|
||||
## 1. Finish what 2026-09-04 started (this week)
|
||||
|
||||
| # | Item | Why | How | Size |
|
||||
|---|---|---|---|---|
|
||||
| 1.1 | **Roll all 23 cells onto the current relay image** | Every cell still runs the image that exits the whole process on a Postgres connect timeout (Finding 6). The fixed image runs only on the director and c7. Any future DB stall repeats the 200-crashes-in-48h pattern. | `cloud-deploy-relay-production-same-cap` waves, gated by the 15-min monitor. Roll inputs and canary results are in the findings doc ("Roll inputs", "Canary blast radius"). | one afternoon |
|
||||
| 1.2 | **Enable the refresh_tokens pruning job** (orca-cloud #476, merged, off) | `refresh_tokens` is 63 M rows / 26 GB and grows forever; its size is what turned a slow disk into a sign-out storm (Finding 13). | Build an auth image from main (the 21:04Z deploy already contains the entrypoint: `orca-cloud-auth-00031-tox`, digest `343a0915…`), set `auth_token_pruner_enabled = true` and the image digest in `infra/terraform-apps/environments/production.tfvars`, apply targeted. First run with a small `auth_token_pruner_max_deleted_rows`. Watch the run summary's `stopReason`, not the exit code. ~48 M rows drain in ~10 days at 200k/hour. | 1 hour + 10 days of watching |
|
||||
| 1.3 | **Reclaim the disk after pruning** | Deletes leave dead tuples; the 16 GB table does not shrink on its own. | `pg_repack` (or `VACUUM FULL` in a maintenance window; it takes an exclusive lock) on `refresh_tokens` off-peak, after 1.2 finishes. | 1 evening |
|
||||
| 1.4 | **Full Terraform apply of the orca-cloud apps root** | The production plan carries four drifts from other merged work: `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` env on the auth service (#476), a skill-share log exclusion filter change, skill pressure threshold 16→8, an artifacts bucket lifecycle rule. Locally it also fails on the 1Password Cloudflare data source. | Run from CI or a machine with the 1Password account; review the four drifts as ordinary changes. | 30 min |
|
||||
| 1.5 | **Alert on the pruning job** | A run that only ever times out exits 0 and reads as green. | Log metric on the job's summary event where `stopReason != "complete"`, policy on the relay channel. | 1 hour |
|
||||
|
||||
## 2. Remove the shared fate between auth and relay (2.1 and 2.3 this quarter; 2.2 deferred)
|
||||
|
||||
| # | Item | Why | How | Size |
|
||||
|---|---|---|---|---|
|
||||
| 2.1 | **Private IP for Cloud SQL, `--private-ip` on the cell proxies** (do this on the existing shared instance; do not wait for 2.2) | Cells reach the database's public IP through Cloud NAT. Dynamic port allocation (landed) raised the ceiling from 64 to 4096 ports per VM, but the NAT is still in the path and its logs are still the only place port exhaustion shows up (Finding 11). | Add a private IP to `orca-cloud-auth-db` (foundation root, orca-cloud), peer the relay VPC, switch the proxy flag in the cell template, roll. | 1–2 days |
|
||||
| 2.2 | **Split the relay database from the auth database** — *DEFERRED 2026-09-04 (owner decision): revisit ~2026-11-01 once pruning is done and there is a month of alert history* | One Cloud SQL instance serves `orca_auth`, `orca_relay`, `orca_push`, `orca_skills`. The auth table's growth stalled the relay for a day (Findings 10, 13). Deferral rationale: the concrete cause is fixed (disk 250 GB, WAL 16 GB, index, pruning), 2.3 + 1.1 turn a future stall into retries, and the checkpoint/disk/headroom alerts now page. Re-open if the checkpoint-loop or connection-headroom alert fires, or a large new auth-side table is planned. | New instance for `orca_relay`; migrate with a short relay drain. Relay state is small so the cutover is minutes. | 1–2 weeks incl. rehearsal on staging |
|
||||
| 2.3 | **Statement timeouts on the relay pool** (the auth pool got one in #476) | A relay query stuck behind a checkpoint fsync should fail fast and let the bounded retry take over rather than hold a pool slot for seconds. | `statement_timeout` on the relay `pg.Pool` in `cloud/apps/relay`, tuned under the lease renewal deadline. | half a day |
|
||||
|
||||
## 3. Make the desktop refresh path forgiving (next 2 weeks)
|
||||
|
||||
| # | Item | Why | How | Size |
|
||||
|---|---|---|---|---|
|
||||
| 3.1 | **Refresh-token rotation grace window** | The server revokes the whole family the first time a just-rotated token is presented again. On 2026-09-04 that turned a 30 s server slowdown into 21,605 sign-outs. A short window (e.g. 60 s) where the immediately-previous token is still accepted, returning the same new token, is standard practice. | In `apps/auth/src/tokens/refresh-tokens.ts`: accept `rotated_at` within the window, return the successor instead of revoking. Keep true reuse (outside the window, or a third presentation) as revocation. | 1 day incl. tests |
|
||||
| 3.2 | **Do not retry `/refresh` with the same token on timeout** | Desktop's 30 s `CLOUD_REQUEST_TIMEOUT_MS` expiring is treated like a network error and retried with a token the server may already have rotated. | In `src/main/orca-profiles/profile-cloud-session-refresh.ts`: on timeout, re-read the stored session first, and prefer a longer single attempt for the refresh call specifically. | half a day |
|
||||
| 3.3 | **Un-revoke is impossible; make sign-out recovery obvious instead** | Server-side un-revoke does not help because the desktop deletes its local token on the 401. Landed: desktop notices immediately (#18694) and the phone says "desktop signed out" (#18698). | Nothing more unless we want a re-auth deep link from the phone to the desktop. | — |
|
||||
|
||||
## 4. Chronic relay issues already characterised
|
||||
|
||||
| # | Item | Why | How | Size |
|
||||
|---|---|---|---|---|
|
||||
| 4.1 | **Cell-inventory lock contention** (partial: PR #18722 narrowed the remaining non-placement sites; `assignOnce` placement lock is the follow-up) | `postgres_retries` is a global `FOR UPDATE` over the 23-row `relay_cells` table with a 1 s `lock_timeout`; it is the floor under every 503 and every slow phone accept (Findings 2, 5; memory `relay-cell-inventory-lock-contention`). | Per-cell row locks or an advisory lock keyed by cell; move capacity counters to delta writes. Verify against real Postgres on 55440. | 1 week |
|
||||
| 4.2 | **Region preference is mostly inert** | Phones request an Asia cell on ~19 % of attempts and get one ~6 % of the time; the sticky lane wins silently, so Asia users ride the US path more than intended (memory `relay-region-preference-mostly-inert`). | Let a region preference override stickiness when the preferred region has headroom; measure with `orca_relay_runtime_metrics` region counters. | 2–3 days |
|
||||
| 4.3 | **Desktop lease-rotation waves** | A cell recreate seeds a fleet-wide 1006/4408 reconnect burst ~54 min later, every ~54 min (Finding 3). | Jitter the desktop control lease renewal by ±10 % so the cohort spreads out. | half a day, desktop + wire-compatible |
|
||||
| 4.4 | **Raise `postgres_retries` gate calibration** | The 300 bar was recalibrated (PR #18580) but should track the post-lock-fix baseline once 4.1 lands. | Re-derive from a week of `orca_relay_postgres_transaction_retry` counts. | 1 hour |
|
||||
|
||||
## 5. Observability still missing
|
||||
|
||||
| # | Item | Why | How |
|
||||
|---|---|---|---|
|
||||
| 5.1 | **Cell crash-rate alert** | 201 process exits in 48 h with no page (Finding 6). | Log metric on `container die` for `resource.type="gce_instance"` relay cells, > 3 per 15 min per cell. In `cloud/infra/terraform/relay-observability.tf`. |
|
||||
| 5.2 | **Page a person for auth alerts** | Today's four auth policies (orca-cloud #475) route to the relay Slack channel only. A repeat of 2026-09-04 deserves a page. | Add a PagerDuty/phone notification channel to `auth_alert_notification_channels` for refresh rejections and latency. |
|
||||
| 5.3 | **Pruning job alert** | See 1.5. | |
|
||||
| 5.4 | **Dashboard that puts the four signals side by side** | Diagnosis took hours because checkpoint state, NAT drops, auth 401 rate, and fleet controls live in four consoles. | One Cloud Monitoring dashboard: `orca_relay_cloud_sql_wal_checkpoint`, NAT `dropped_sent_packets_count`, `orca_auth_refresh_401`, summed `controls`. |
|
||||
|
||||
## Landed on 2026-09-04 (for completeness)
|
||||
|
||||
- Auth service cap 2 → 20 (service-level manual scaling removed); Cloud SQL disk 49 → 250 GB PD-SSD;
|
||||
`max_wal_size` 16384; partial index `refresh_tokens_family_unrevoked` built concurrently by hand.
|
||||
- orca-cloud #474: the above in Terraform + deploy workflow; replayed dead token answers 401 without
|
||||
re-revoking or re-auditing. Deployed as `orca-cloud-auth-00031-tox` 21:04Z.
|
||||
- orca-cloud #475: auth alerts (refresh 401 > 100/5 min, 429 > 20/5 min, 5xx > 10/5 min, p99 > 10 s). Applied.
|
||||
- orca-cloud #476: batched `refresh_tokens` pruner (disabled), auth pool `statement_timeout` 10 s, schema
|
||||
DDL on an untimed connection.
|
||||
- stablyai/orca #18693: both relay NATs on dynamic port allocation 64..4096 (applied US 21:01Z, Asia 21:05Z);
|
||||
alerts for Cloud SQL WAL-checkpoint loop, disk > 70 %, NAT `OUT_OF_RESOURCES` drops. Applied.
|
||||
- stablyai/orca #18694: desktop learns of a revoked session immediately, panes re-fetch on mount, pairing
|
||||
notice says "Sign in again to use Orca Relay".
|
||||
- stablyai/orca #18698: phone shows "Desktop signed out — sign in to Orca on your desktop to reconnect" via
|
||||
the WebSocket close reason (only additive slot old phones tolerate).
|
||||
- Director on image 519f4914; c7 on 85bf6799; other 22 cells still on the old image (see 1.1).
|
||||
@@ -0,0 +1,991 @@
|
||||
# Relay reconnect investigation: findings and evidence
|
||||
|
||||
Working notes for the 2026-09-04 mobile relay reconnect incident and the cell roll that follows.
|
||||
Kept current across context compactions. Newest section first. All times UTC. Host ids are log digests,
|
||||
never raw ids. Nothing here is a production mutation record unless the "Mutations" section says so.
|
||||
|
||||
## Status board
|
||||
|
||||
| Item | State | Where |
|
||||
|---|---|---|
|
||||
| PR #18565 relay accept abandonment + lease jitter + desktop rotation spread + phone probe fail-fast | Open, CI fully green again after the doc move (05:45Z), CodeRabbit + Pullfrog cleared, 3 review rounds; not merged (owner has not asked) | https://github.com/stablyai/orca/pull/18565 |
|
||||
| PR #18569 monitor `relayPostgresRetryExhausted` 0 -> 300 | **Merged** 2026-09-04 ~04:20Z as 4101505b6b | https://github.com/stablyai/orca/pull/18569 |
|
||||
| Same-cap `verify` of c7 (read-only) | **Passed** run 33836527159 | confirms identities, selector gen 110, rehome gen 12, protocol 1, digests |
|
||||
| Monitor dry-run #1 | Froze min 5: `relay.postgres_retries` 380 > 300 | run 33836470590 |
|
||||
| Monitor dry-run #2 | Green to min 13, froze 04:49Z: `director.concurrency` 76.7 > 64 (six-cell crash storm, Finding 6) | run 33837160275 |
|
||||
| Monitor dry-run #3 | Froze min 3 at 05:01Z: `relay.postgres_retries` 339 > 300; no crash, concurrency 5–8 | run 33838698725 |
|
||||
| Owner decision 2026-09-04 ~05:10Z | **Option B approved**: "you can raise the bar. or remove it altogether ... whats the most logical move". Kept the bar (removal would leave contention unwatched during the roll) and recalibrated from measured data. | this thread |
|
||||
| PR #18580 monitor `relayPostgresRetries` 300 -> 2000 | Open, awaiting CI; mutation-checked (300 fails the new test) | https://github.com/stablyai/orca/pull/18580 |
|
||||
| PR #18565 CI | Was red on `root directory guard` because this findings file sat at repo root; moved to `cloud/docs/` in 8ebff89106 | |
|
||||
| PR #18580 | **Merged** 2026-09-04 05:23Z as 79d5fb469a (Pullfrog cancelled by the merge; independent Opus review requested instead, per owner) | |
|
||||
| Monitor dry-run #4 | Froze min 12 at 05:37:35Z: `cell.production-gce-c27.health`/`.ready` = 0. Retries green all 12 samples under the new 2000 bar. Cause: c27 (asia-east2) container died 3x 05:37:00–05:38:01Z, Finding 6 crash class. | run 33840364323 |
|
||||
| Monitor dry-run #5 | Froze at sample 1 (05:41Z): c27 health/ready still 0. MIG autoheal `recreateInstance` on c27 fired 05:38:12Z after the 3 crashes; instance RECREATING, process up with 0 controls (was ~395). Second c27 recreate in 7 h (Finding 3 seed pattern). Waiting for c27 to settle before dry-run #6. | run 33841327879 |
|
||||
| Monitor dry-run #6 | **Passed** 06:06:31Z: 16 samples, no freeze (started 05:47:42Z) | run 33841783747 attempt 1 |
|
||||
| c7 `canary-apply` | **Succeeded.** Dispatched 06:07:15Z; drain 06:10Z; MIG recreate 06:16–06:23Z; new image listening 06:23:42Z; verify + trust proof passed; restored to `admission=general` 06:25:21Z; canary authority sealed. c7 is on `85bf6799…`. | run 33843071283 |
|
||||
| PR #18581 doc reconcile (Aug 23 figure: 2,200–3,000 raw log lines vs 1,510 on the gate metric) | **Merged** | https://github.com/stablyai/orca/pull/18581 |
|
||||
| Same-cap `verify` c7 target=519f4914 rollback=85bf6799, gen 112 | **Passed** (read-only) | run 33856355648 |
|
||||
| Monitor dry-run #7 (gen 112) | Froze at sample 1 (09:05:31Z): `director.errors` 4 > 0, the four 2.0 s pg-connect 500s from the 09:00 cascade still inside the 5-min delta window. Dispatched 4 min too early. | run 33856521278 |
|
||||
| Monitor dry-run #8 (gen 112) | Green for 15 of 16 samples (09:09:38–09:24), froze on the final sample 09:25:22Z: `director.errors` 1 > 0. The one 500 was `/v1/admin/evacuation-status` at 09:23:50Z, 2.01 s latency = director pg-connect timeout, called by **the monitor's own collector** (`incident-monitor-sources.ts:492`). First evacuation-status 500 since Sep 1. The gate froze on a request it made itself. | run 33856905229 |
|
||||
| Monitor dry-run #9 (gen 112) | Froze: c13/c23 crashed 50 s after dispatch, then c14/c20/c9 at 09:34. | run 33858650691 |
|
||||
| Monitor dry-run #10 | Dispatched 09:46:13Z; froze at sample 5 (09:56:59Z): `director.errors` 12. All twelve at 09:55:17–21Z, 0.8–2.1 s latency, 10 on `/v1/regions` + 2 on `/v1/assign`; c16 and c8 crashed at 09:55:19 in the same second. A single 4-second Postgres connect stall hit director and cells together. | run 33859947207 |
|
||||
| Monitor dry-run #11 | Froze at sample 2 (10:08:07Z): `director.concurrency` 79.8 > 64, the c8/c20 re-dial. They crashed 10:05:54, 3 s before the waiter's quiet check passed (log ingestion lag). | run 33861578009 |
|
||||
| Monitor dry-run #12 | Dispatched 10:17:38Z after 10 quiet min; froze at sample 2 (10:19:24Z): `cell.production-gce-c16.health` 0. c16 did **not** crash (no container die, MIG NONE/HEALTHY, readiness=true throughout, `/health` 200 in 230 ms at 10:21). At 10:19:07–16 it logged "control activity renewal failed" x4 and a burst of 1006 closes, sqlFailures 1 -> 14, sqlLatencyMsMax 2588: a pg stall on the old image that did not reach the unhandled path. The probe's single fetch (30 s timeout) came back unavailable during that stall and `unavailableIsZero` turned it into health=0. | run 33862504601 |
|
||||
| Monitor dry-run #13 | Green 14 of 16 samples (10:48:38–11:03), froze 11:04:43Z: c9 crashed 11:04:23, c28 11:04:25 (then looped 11:05:04, 11:05:41); c15 probe also read 0 (stall, no crash). Missed by ~90 s. **Dispatched by hand 10:48:15Z** into a 43-min crash lull (last die 10:05:54; last director 500 10:31:49). The re-armed waiter never fired: its MIG-stable check used `grep -vc True`, which exits 1 when nothing matches, so `&&` short-circuited on the *healthy* case. Waiter armed 10:20Z: 10-min quiet + every MIG stable + 60 s recheck, then dispatch, then canary c7 on green. Held at 10:24 and 10:31 by lone director `/v1/assign` 500s (2 s pg-connect stalls, no cell crash). Director 500 events since 08:46: 6 (gaps 2.7/21/31/29/7.6 min). At 10:39 the waiter was re-armed with a 6-min director-500 window (the monitor's own delta is 5 min) instead of 10, since the gate only needs the 15 min *after* dispatch to be clean. Cell crashes have stopped since 10:05 (33+ min, longest gap since 08:40). 12 dry-runs: 1 pass (#6), 11 freezes, none on a real fleet-health regression. | Cascade gaps since 09:00: 31, 2.9, 5.1, 16.1, 4.0 min (median 5); a 15-min clean window is ~28% per attempt at this rate. | |
|
||||
| Monitor dry-run #14 | Dispatched 11:26:53Z by the fixed waiter (first autonomous dispatch); c14, c23, c25, c15, c24, c19 died 11:30:59–11:31:08 (six cells, 13 min after the last cascade). Froze on c8 (and others) health/ready probes. Waiter re-armed 11:06Z (grep bug fixed: `grep -c` under `|| true`), same chain; held through the 11:17 cascade and c14/c28 recreates. 13 dry-runs: 1 pass, 12 freezes. Since 08:40: 10 cascades, 75 container dies, gaps 20/31/3/5/16/4/6.5/58/13 min; only 3 windows of >=17 clean minutes existed in 2.6 h, and dry-runs hit two of them (#6 passed, #13 lost the third by 90 s). |
|
||||
| Monitor dry-run #15 | Waiter armed 11:33Z (6-min director-500 window, 8-min crash window, all MIGs stable), chained canary; still holding at 12:04Z. Since 11:00: 8 cascades, 98 dies, gaps 13/13.6/3.6/14.5/4.4/6.1/3.0 min, **max gap 14.5 min**, so no 15-min clean window has existed in the last hour. 14 dry-runs: 1 pass, 13 freezes. |
|
||||
| Monitor dry-run #15 verdict | Dispatched 12:28:49Z; froze at sample 2 (12:30:41Z): **12 cells** health/ready = 0 at once (c4, c5, c7, c10, c15, c16, c18, c20, c22, c25, c27, c28), including c4/c5 (0 controls all day, `/health` 200 in 190 ms a minute later) and c7 (new image). Six old-image cells also crashed 12:30:02–21. This was a fleet-wide SQL stall, not a cascade: every cell's `sqlLatencyMsMax` hit 4–6 s (c7 4865, director 5140), director pool waiting 1258, 15 cell pg-connect timeouts, director sqlFailures 92. Cloud SQL CPU 0.73, backends 160, new connections normal, memory 0.46, so the *instance* was not saturated; something held the database for ~5 s. Postgres log 12:31:23–28 shows a burst of `could not obtain lock on row in relation "relay_cells"` from NOWAIT (single-row and full-inventory) sweeps, i.e. the row locks were held during recovery. Cloud SQL transactions/min flat (~30k), reads flat,
|
||||
network flat: the database was neither busy nor saturated, it was *waiting*. The stall bracket
|
||||
(12:30:02–12:30:41) is where every cell's SQL max hit 4–6 s at once. Lock retries in that window were
|
||||
ordinary (49/29/13 per min). Best reading: a ~5 s Postgres-side wait event shared by every session
|
||||
(lock on a hot row held across a long transaction, or an instance-level pause), not CPU/IO. Cell
|
||||
`sqlLatencyMsMax` was already 1.5–2.2 s fleet-wide in the four minutes before, i.e. the old cells' 1 s
|
||||
`lock_timeout` plus queueing. | run 33872946111 |
|
||||
| Monitor dry-run #16 | Dispatched 12:38:57Z; froze at sample 1 (12:40:11Z): `cell.production-gce-c27.latency_ms` 2071 > 2000, a fifth distinct freeze signal, the probe's own round-trip absorbing a checkpoint sync. **Loop stopped by me at 12:41Z**: with the disk in the checkpoint loop (Finding 10) no bar can hold for 15 min, so further dry-runs only burn the shared rollout lease. 16 dry-runs: 1 pass, 15 freezes. Re-arm after the disk change lands. |
|
||||
| Cloud SQL checkpoint loop | **Broke on its own 12:39–12:45Z**: disk writes 48 -> 4 MB/s at 12:39 with transactions and network flat and no Cloud SQL operation; 12:40:17 checkpoint synced 0.047 s; 12:45:53 checkpoint was `time`-triggered again (first since 11:55) with sync 0.096 s and write spread over 269 s. Cause of the break unknown (most likely WAL fell back under `max_wal_size` once a burst of full-page writes aged out). It can re-enter the loop on the next large checkpoint; the disk-size fix remains the durable one. |
|
||||
| Monitor dry-run #17 | Dispatched ~12:49Z (all guards clean); froze at sample 1 (12:52:05Z): `director.errors` 4, from the c9/c22 crash loop that began 12:50:34, ~90 s after dispatch. Checkpoints stayed healthy (85 ms), so this is the old image's baseline crash rate, not the disk. 17 dry-runs: 1 pass, 16 freezes. |
|
||||
| Monitor dry-run #18 | **Dispatched by mistake 13:48:56Z into the outage**: my gcloud credentials expired ~13:45Z, every guard query returned empty, and the waiter's `grep -c . || true` read empty as "quiet". Froze at sample 1 (13:49:43Z) on `director.ready=0`, `auth.health=0`, and cell probes; no canary dispatched, no production mutation. All waiter loops killed at 13:51Z. Lesson: a quiet-window check must fail closed when its data source errors. Waiter had been re-armed 12:53Z. |
|
||||
| Gate decision | Owner asked at 09:36Z to choose: A keep looping / B recalibrate `directorErrors` 0 -> small n / C human bypass. Ten dry-runs, four froze on this bar. Recommendation B+A. Note: B alone would not have passed #9 or #10 (cell health probes and a 12-error burst); it fixes the single-500 false freezes (#7, #8) only. | |
|
||||
| Batch roll | **Deferred by plan**: roll once with the lock-fix image instead of twice. | |
|
||||
| PR #18606 lock removal (root cause) | **Merged** 09:2xZ as 7b108abf71 after review, fix, re-verify; CI green | https://github.com/stablyai/orca/pull/18606 |
|
||||
| Image publish for 7b108abf71 | **Done** 08:36:49Z run 33854111305: `sha256:519f4914217f08cabcdcd34825965db8473ec37c6591553a3af0d65dcdeeb183` | |
|
||||
| Director deploy on 519f4914 | **Succeeded** 08:45Z run 33854355791; serving `orca-cloud-relay-00570-siv`, rollback tag on 00569-ret (also 519f4914), 00565-fes (85bf6799) still deployable. Dispatched 08:37:45Z (blue/green; prior revision 00565-fes on 85bf6799 kept as rollback). Note: `predecessor-image-digest` is a required input even with bootstrap=false; pass the serving digest. | `cloud-deploy-relay-production-director.yml` |
|
||||
| c7 on new image, 2 h in | 817 controls, **0 container die** since restore (was ~1 per 15 min on old image); `sqlLatencyMsMax` still 1.0 s = lock wait unchanged, which #18606 targets | |
|
||||
| Terraform alert `relay_postgres_retry_exhausted` at `> 0` | Firing continuously since #18521; recalibration not done (own change) | `cloud/infra/terraform/relay-observability.tf:447,469` |
|
||||
|
||||
## Mutations performed (complete list)
|
||||
|
||||
1. Merged PR #18569 to main (code/docs only).
|
||||
2. Merged PR #18580 and #18581 to main (monitor bar + docs).
|
||||
2b. Merged PR #18606 to main (relay lock change; no serving effect until the image is deployed).
|
||||
2c. Dispatched `cloud-publish-relay-production` for 7b108abf71 (builds and pushes an image; changes nothing serving). Done: 519f4914.
|
||||
2d. Dispatched `cloud-deploy-relay-production-director` on 519f4914 (preserve placement, no prune, rehome gen 12). Succeeded 08:45Z; serving revision 00570-siv. Rollback: `gcloud run services update-traffic orca-cloud-relay --region us-central1 --to-revisions orca-cloud-relay-00565-fes=100` (85bf6799, still Ready). Not needed so far.
|
||||
3. 2026-09-04 06:07:15Z: dispatched `cloud-deploy-relay-production-same-cap` `canary-apply` for production-gce-c7 only (run 33843071283). Completed successfully 06:26Z: c7 isolated, drained (807 controls re-dialed), template + MIG rolled to 85bf6799, verified, restored to general admission. Selector generation advanced 110 -> 112 (isolate + restore).
|
||||
4. Nothing else. Both monitor dispatches were `mode=dry-run` (read-only). The same-cap dispatch was `mode=verify` (read-only, confirmed by step gates `if: inputs.mode != 'verify'` on every mutating step).
|
||||
|
||||
## Finding 6 (2026-09-04 ~05:00Z): the old cell image crashes the whole process on a Postgres connect timeout
|
||||
|
||||
**This is the most important open finding.** The 23 GCE cells run image `sha256:5aedbca5…` = orca-cloud
|
||||
commit e3e92d95d3 (2026-08-14). In that build `beginProof` is called as `void this.beginProof(...)`.
|
||||
When `verifyCellAssignment` inside it throws (pg-pool `timeout exceeded when trying to connect`, 2 s
|
||||
`connectionTimeoutMillis`), the rejection is unhandled and Node exits 1. Docker restarts the container
|
||||
in ~1 s, but every control on that cell (~800 hosts) drops and re-dials `/v1/assign` at once.
|
||||
|
||||
Evidence, cell c7 instance 4545742188814054238, 2026-09-04:
|
||||
|
||||
```
|
||||
04:46:47.951 stderr [orca-relay] control activity renewal failed (x5)
|
||||
04:46:49.527 stderr Error: timeout exceeded when trying to connect
|
||||
at pg-pool/index.js:45:11
|
||||
at async PostgresPoolPressure.connect (postgres-pool-pressure.js:30:20)
|
||||
at async PostgresDatabase.query (database.js:645:24)
|
||||
at async RelayAssignmentStore.verifyCellAssignment (assignment-store.js:2024:22)
|
||||
at async HostSessionRegistry.beginProof (host-session-registry.js:376:15)
|
||||
04:46:49.527 stderr Node.js v24.19.0
|
||||
04:46:49.835 dockerd: container die … exitCode=1 image=…relay@sha256:5aed…
|
||||
04:46:50.258 dockerd: container start
|
||||
04:46:52.761 stdout [orca-relay] listening on https://c7.relay.onorca.dev
|
||||
```
|
||||
|
||||
2026-09-04 05:36:59–05:38:01Z: c27 died 3x in 62 s plus one other instance (5464389947731541178); this froze dry-run #4 on c27's health probe.
|
||||
|
||||
Fleet-wide `container die … exitCode=1` on the relay image, last 48 h: **201 events on 19 instances**
|
||||
(c28 x38, c29 x37, c27 x19). Hourly counts track the lock-contention curve (peak 23/h at 21Z Sep 3).
|
||||
Every one has the same `Node.js v24…` crash banner. On 2026-09-04 04:46:35–04:47:41Z six cells
|
||||
(c7, c8, c19, c21, c22, c25) died within 66 s: ~4,800 hosts re-dialed, `/v1/assign` returned 16,321
|
||||
503s in one minute (baseline ~20), director concurrency hit 85 (Cloud Run cap 80), Cloud SQL
|
||||
`new_connection_count` 119 -> 287/min. Fleet recovered by 04:51Z. That is what froze dry-run #2.
|
||||
|
||||
Fix status: `guardSessionTask` wrapping `beginProof` landed in orca-cloud #436 (2026-08-27) and is in
|
||||
the target image `sha256:85bf6799…` (main 11aace8dec). The roll is the fix. Not caused by anything in
|
||||
this session: the same-cap verify finished ~04:25Z and never reached a mutating step; no compute
|
||||
operations exist for those instances; heap/event-loop were flat before the crash.
|
||||
|
||||
Autoheal amplifier: MIG health check is `/health` every 10 s, timeout 5 s, unhealthy after 3, so a
|
||||
crash loop of ~30 s+ triggers `compute.instances.repair.recreateInstance`. All ~20 recreates in the
|
||||
48 h to 2026-09-04 05:40Z were the three Asia cells (c27 x6, c28 x7, c29 x8; gcloud prints local
|
||||
-07:00 times). c27 recreated 05:38:12Z after 3 crashes in 62 s; its ~395 controls went to 0 and the
|
||||
monitor's `cell.production-gce-c27.health/ready` probe read 0 for the whole recreate (~several min),
|
||||
freezing dry-runs #4 and #5. Each recreate also seeds a Finding 3 rotation cohort. Rolling the Asia
|
||||
cells early in the batch phase should be weighed against the canary-first rule; c7 stays the canary.
|
||||
|
||||
Implication for the gate: the monitor's `director.concurrency` freeze is *correctly* detecting these
|
||||
crash storms. A dry-run only passes in a 15-minute window with no cell crash, roughly 1 in 3 windows
|
||||
at current rates. Retrying in quiet hours is legitimate; the bar is not wrong.
|
||||
|
||||
## Finding 5: `relay.postgres_retries` at 300 is 3x under today's baseline
|
||||
|
||||
Retries per 5 min, cells + director, last 24 h: p50 579, p90 1039, p99 1398, max 1505; **65% of
|
||||
windows over 300**. Quiet hours (03–08Z) p50 235, max 512. When the 300 bar was set (2026-08-26)
|
||||
healthy bursts reached 234. Baseline has roughly tripled in 10 days. Skill notes say do not raise this
|
||||
bar; I have not. Best odds for a clean 15 min are 02–04Z and 17–18Z (9/12 five-minute windows under
|
||||
300 in each).
|
||||
|
||||
## Finding 4: exhausted-retry bar was the wrong single blocker (fixed)
|
||||
|
||||
`relayPostgresRetryExhausted: 0` never cleared after #18521 reached the director (22:12Z Sep 3): 236/236
|
||||
five-minute windows non-zero; post-#18521 p50 42 / p90 147 / max 220; Aug 23 incident peak 467.
|
||||
Recalibrated to 300 in #18569 (merged). Dry-run #1 immediately revealed Finding 5 behind it.
|
||||
|
||||
## Finding 3: the 00:50Z control-close wave was desktop lease rotation, not a rollout
|
||||
|
||||
2026-09-04 00:49–00:51Z: 2,745 control closes on 19 instances; 1157/1632 code 1006 and 973/1030 code
|
||||
4408 `control rebound` had ageMs in the 53-minute bin. Relay grants a flat 55 min lease; desktops
|
||||
rebind 60–120 s early; so every host that (re)connected in the same minute rebinds as one cohort
|
||||
forever. Seed: c27 MIG autoheal recreate 23:23Z (`compute.instances.repair.recreateInstance`) dumped
|
||||
~420 controls. Harmonics at 23:55, 00:04, 00:25, 00:49Z. Each rebind is an `activateControl`
|
||||
transaction that can take the inventory lock. Fix in #18565: relay lease 55 min ± 5 min (symmetric,
|
||||
so mean rebind rate unchanged), desktop early window 1–6 min.
|
||||
|
||||
## Finding 2: fleet-wide lock contention, worse on Sep 3
|
||||
|
||||
| window | 55P03 retries/h (cells) | cell sqlFailures/h |
|
||||
|---|---|---|
|
||||
| Sep 2 18Z – Sep 3 07Z | 660–1470 | 680–1620 |
|
||||
| Sep 3 08Z–16Z | 3600–7100 | 3700–7700 |
|
||||
| Sep 3 23Z | 7468 | 7585 |
|
||||
|
||||
100% of sampled retries are 55P03; director phase is `cell-inventory`. Every cell pins
|
||||
`sqlLatencyMsMax` at 1.0–1.2 s = the pre-#18521 1 s pool `lock_timeout`. Not load (controls flat
|
||||
~26k, Cloud SQL CPU 46–53%). No `cloud-*` workflow explains the 08Z step. The lock is a global
|
||||
`SELECT * FROM relay_cells FOR UPDATE` (23 rows) taken by assignment, control activation, activity
|
||||
acquire, and sweeps, held to COMMIT.
|
||||
|
||||
## Finding 1: root cause of the phone's 24 s hang (the original symptom)
|
||||
|
||||
`acceptClient` runs four serialized Postgres calls; the fourth (`acquireActivity`) contends for the
|
||||
global lock. Under contention the cell finishes after the phone's 12 s bound, then
|
||||
`PendingHostDataReservation.bind` throws `host_data_reservation_already_bound` because the phone's
|
||||
close already released the reservation. Every "first frame handler failed already_bound" line is that
|
||||
post-mortem (31 events 23:06–01:01Z across 12 instances). Fix in #18565: abandon the accept after each
|
||||
DB step once the socket is closed; new event `orca_relay_client_accept_abandoned {stage, elapsedMs}`
|
||||
and metric fields `clientAcceptsAbandonedByStageDelta` / `clientAcceptAbandonedMsMax`. Phone side:
|
||||
direct probe now fails fast on `reconnecting` so relay recovery is not queued behind three doomed
|
||||
LAN redials (~3.5 s saved per foreground). #18518 (merged, not yet on the phone) covers the
|
||||
stage-aware dial bound.
|
||||
|
||||
Host 666077865f2e: stable throughout. 4408 rotation 00:27:45Z; 1006 quit 00:52:24Z on old adhoc;
|
||||
sticky reassignment to c27 00:52:35Z on new build; rotation closes 01:44:55Z and 02:23:15Z with
|
||||
splices intact. No drain/4404/wrong-cell.
|
||||
|
||||
## Finding 7 (2026-09-04 ~05:10Z): retries bar recalibration basis (PR #18580)
|
||||
|
||||
Chose 2000 over removal. The metric is the gate's own source (`orca_relay_postgres_retries`
|
||||
log metric, director + cells summed per five minutes, ALIGN_DELTA 300 s):
|
||||
|
||||
| window | p50 | p90 | p99 | max | > 300 |
|
||||
|---|---|---|---|---|---|
|
||||
| 2026-09-01 | 56 | 105 | 206 | 456 | 0% |
|
||||
| 2026-09-02 | 109 | 186 | 294 | 377 | 1% |
|
||||
| 2026-09-03 | 430 | 924 | 1320 | 1504 | 55% |
|
||||
| 2026-09-04 to 05Z | 285 | 1012 | 1211 | 1211 | 44% |
|
||||
|
||||
15-minute pass rate, last 24 h: bar 300 -> 22%, 800 -> 66%, 1000 -> 86%, 1500 -> 99%, 2000 -> 100%.
|
||||
Aug 23 incident on this metric: 1510 then 646 (single windows), so retries no longer separate an
|
||||
incident from baseline; exhausted (467 vs bar 300; healthy 72 h max 184), director concurrency,
|
||||
and pool bars carry that role. Note: my earlier "p99 1398 / 65% over 300" in Finding 5 came from
|
||||
raw log line counts; the metric-based numbers above are what the gate actually evaluates.
|
||||
Baseline tripled between Sep 2 and Sep 3 with no deploy; still unexplained (Finding 2).
|
||||
|
||||
## Decision needed from the owner (resolved: B)
|
||||
|
||||
The same-cap roll is blocked only by the monitor gate, and the gate is blocked by `relayPostgresRetries: 300`
|
||||
(Finding 5: 65% of windows breach it; even the 04:55Z quiet window hit 339). Three options:
|
||||
|
||||
- A. Keep waiting for a naturally quiet 15 min. Odds per attempt ~1 in 3 in quiet hours, lower by day.
|
||||
Each attempt is free and read-only. Could take hours.
|
||||
- B. Recalibrate `relayPostgresRetries` from measured data, same method as #18569: 24 h p99 is 1398, the
|
||||
Aug 23 incident ran 2200–3000, so ~1500 clears healthy windows with ~1.5–2x incident separation
|
||||
(less margin than the exhausted bar had). Overrides the "do not raise" note in the skill facts.
|
||||
Argument for: the roll being gated is the thing that reduces retries. Argument against: the bar is
|
||||
doing its job of saying contention is high.
|
||||
- C. A human dispatches the roll with a different gate policy. Not something I can or should do.
|
||||
|
||||
My recommendation: B, with the number chosen from the table in Finding 5 and the roll following
|
||||
immediately so the bar can be re-tightened after the fleet is on the 500 ms lock wait.
|
||||
|
||||
## Finding 12 (2026-09-04 13:12Z): **INCIDENT IN PROGRESS. The auth service is at its 2-instance cap and rejecting 90% of desktop token calls with 429; the relay fleet has emptied.**
|
||||
|
||||
Timeline: 13:04–13:06 the old-image cascades and NAT stalls drove ~1,400 desktops to re-dial. Their relay
|
||||
JWTs (5-min TTL) expired mid-storm, so they hit `orca-cloud-auth` `/v1/desktop/auth/refresh` and
|
||||
`/v1/desktop/auth/relay-token` together. The auth service is Cloud Run `maxScale=2`, `concurrency=80`,
|
||||
1 vCPU throttled (`auth_max_instances = 2` in orca-cloud `infra/terraform-apps/environments/production.tfvars`,
|
||||
applied by `deploy-auth-production.yml`). Both instances pinned at concurrency 85 from 13:02; from 13:07
|
||||
Cloud Run's front door returns **429 "no available instance"** (0 s latency, never reaches the container):
|
||||
12,045 at 13:07, 54,292 at 13:08, 46,025 at 13:08, 42,529 at 13:09. Sep 3 total auth 429s: **0**.
|
||||
Without a fresh relay token every desktop's `/v1/assign` gets 401 (1,433 distinct hosts 401'd, 0 got 200
|
||||
since 13:07) and every cell closes its control with `4401 relay authorization expired`. Fleet controls:
|
||||
13,375 (12:55) -> 7,633 (13:08) -> **249 (13:12)**, splices 1. Auth container CPU 0.15–0.5, so the cap is
|
||||
the limit, not the code. Every desktop is now in its refresh-retry loop hammering the same 2 instances:
|
||||
this is a self-sustaining thundering herd and will not clear on its own. At 13:14Z: fleet **30 controls**
|
||||
across 23 cells; successful relay-token issuance 5,000–6,500/min until 13:05, then 1,059 / 734 / 733 /
|
||||
443 / 220 / 214 / 148 / **4** per minute through 13:13; auth 429s 54k -> 25k/min only because desktops
|
||||
are backing off, not because the service recovered. Note `AUTH_MAX_INSTANCES: 2` is also hardcoded in
|
||||
orca-cloud `.github/workflows/deploy-auth-production.yml` (lines 33–34), so a redeploy would re-pin it;
|
||||
change both the workflow env and the tfvars.
|
||||
|
||||
**Immediate mitigation (owner action, not applied):** raise the auth service's max instances. Fastest:
|
||||
`gcloud run services update orca-cloud-auth --region us-central1 --max-instances 20` (or `10`, matching
|
||||
the other apps' `max_instances = 10`), then land the same in `auth_max_instances` so Terraform does not
|
||||
revert it. Auth is stateless behind Cloud SQL (`refresh_tokens` table); backends 210 of 400, so 20
|
||||
instances x a small pool is within budget. Also consider the desktop's refresh backoff: it re-dials on
|
||||
401 immediately with no jitter, so a 429 storm sustains itself.
|
||||
|
||||
**13:51Z status: my gcloud session lost auth at ~13:45Z; all production monitoring from this session is
|
||||
blind until re-authenticated (`gcloud auth login`, interactive). Last confirmed state 13:40Z: fleet 0
|
||||
controls, auth maxScale 2, 7,600 auth 429/min. All autonomous dispatch loops are stopped.**
|
||||
|
||||
**17:19Z–17:21Z MITIGATION APPLIED (owner said "fix it NOW").** State at 17:19Z, four hours in: all 23
|
||||
cells at 0 controls, auth 429 ~2,000/min, auth 2xx ~40/min, and the 2xx that got through took 13–28 s
|
||||
(both instances saturated). Mutation 1: `gcloud run services update orca-cloud-auth --max-instances 20`
|
||||
created revision `orca-cloud-auth-00018-4jc` (same image `auth@sha256:1710ff6c`, same env/concurrency,
|
||||
only maxScale 2 -> 20) but the service pins traffic to `00023-qud` **by revision name**, so the new revision
|
||||
was immediately `Retired` and nothing changed. Mutation 2 (17:21:30Z): `gcloud run services update-traffic
|
||||
--to-revisions orca-cloud-auth-00018-4jc=100`. Lesson: the auth service's traffic block is name-pinned
|
||||
(the deploy workflow does an explicit traffic switch), so a bare `services update` never reaches users.
|
||||
Terraform still says `auth_max_instances = 2`; the next `deploy-auth-production.yml` run will revert this
|
||||
unless the tfvars and the workflow's `AUTH_MAX_INSTANCES` are changed first.
|
||||
|
||||
## Finding 13 (2026-09-04 17:19Z–18:10Z): **the auth outage is a database problem, not (only) a Cloud Run cap; `refresh_tokens` has 63 M rows and reuse-revokes scan whole families**
|
||||
|
||||
Mutations this window (all online, no restarts, all by hand in project onorca-cloud):
|
||||
1. 17:19Z `gcloud run services update orca-cloud-auth --max-instances 20` → new revision `00018-4jc`, but traffic is
|
||||
pinned by revision name so it was `Retired`; 17:21:30Z `update-traffic --to-revisions 00018-4jc=100`.
|
||||
2. Still 2 instances at 17:31Z: the SERVICE has its own `scaling.maxInstanceCount=2` in **manual scaling mode**
|
||||
(`run.googleapis.com/maxScale: '2'` on service metadata, set by Terraform `infra/terraform-apps/auth.tf`), which
|
||||
overrides the revision cap. `--scaling=auto` then `--max 20` at 17:31:45Z. Instances 2→20 by 17:38Z; 429s fell
|
||||
6,000/2 min → 60/2 min at 17:36Z and controls briefly reached 11.
|
||||
3. Then latency, not capacity, became the wall: every refresh took 100+ s inside Postgres (desktop client timeout
|
||||
is 30 s, `CLOUD_REQUEST_TIMEOUT_MS`), so 20 instances × 80 concurrency filled again with requests nobody was
|
||||
waiting for, and 429s returned (~1,500/2 min from 17:40Z).
|
||||
4. 17:27Z Cloud SQL disk 62 GB → 250 GB (IOPS ceiling 1,470 → ~7,500). 18:00Z `max_wal_size` 1.5 GB → 16 GB
|
||||
(the checkpoint loop: `checkpoint starting: wal` every 45–60 s since 13:06Z).
|
||||
5. 18:07Z `CREATE INDEX CONCURRENTLY refresh_tokens_family_unrevoked ON refresh_tokens(family_id) WHERE
|
||||
revoked_at IS NULL` (an earlier attempt with `AND rotated_at IS NULL` was wrong for the revoke predicate; its
|
||||
invalid remnant `refresh_tokens_family_live` was dropped).
|
||||
|
||||
Evidence: `refresh_tokens` = 63.3 M live tuples, 16 GB table + 10 GB indexes; every refresh inserts a row and
|
||||
nothing ever deletes (30-day TTL rows are never pruned). Query Insights 17:33–17:39Z: `UPDATE refresh_tokens SET
|
||||
revoked_at = $1 WHERE family_id = $2 AND revoked_at IS NULL` = 21,000 s of execution per 6 min, ~90–120 k rows
|
||||
updated per minute; io_time 15,000 s read; pg_stat_activity 180+ backends in `IO/DataFileRead` on that statement,
|
||||
200 backends total for orca_auth (20 instances × pool max 10). `session-refresh-reuse-detected` audit events per
|
||||
hour: ~100 all day → 8,805 (13Z), 15,511, 19,486, 24,897, 26,935 (17Z). Mechanism: a desktop's refresh times out
|
||||
client-side at 30 s, the server had already rotated the token, the desktop retries with the same token, the
|
||||
server calls that reuse and revokes the family (Bitmap scan on `refresh_tokens_family` + heap filter over every
|
||||
row the family ever had), then the desktop retries the dead token again, and each retry re-runs the same
|
||||
full-family scan (already-revoked families short-circuit nowhere). Reuse-detected 401 also **signs the user out**
|
||||
on the desktop (`isOrcaCloudAuthFailure` → `clearCloudSessionIfUnchanged`), so every user who hit this during the
|
||||
outage must sign in again.
|
||||
|
||||
Durable fixes (orca-cloud PR in preparation on branch `auth-revoke-only-live-tokens`): `AUTH_MAX_INSTANCES` and
|
||||
`auth_max_instances` → 20; Terraform disk 250 + `max_wal_size=16384`; the partial index in the schema; an
|
||||
`already-revoked` short-circuit in `rotateRefreshToken` that skips the family UPDATE and the audit insert. Still
|
||||
open after that: prune `refresh_tokens` (expired or revoked rows older than N days), a server-side statement
|
||||
timeout shorter than the desktop's 30 s so the client and server agree on failure, and an alert on auth 429s.
|
||||
|
||||
**19:11Z RESOLVED at the database layer.** `refresh_tokens_family_unrevoked` went valid at 19:11:17Z (build
|
||||
18:07–19:11, two full table scans of 2.1 M blocks under load). Within 60 s: refresh latency 100 s → 0.1 s, auth 429
|
||||
→ 0, active orca_auth backends 200 → 2, checkpoints back on the 5-min timer (`checkpoint starting: time` at 18:35,
|
||||
18:41, 19:00, 19:11). Director `/v1/assign` returning 200. Fleet controls 0 → 17 by 19:14Z.
|
||||
|
||||
**Residual: mass sign-out.** 19:11–19:14Z: 3,857 refresh 401s from 3,829 distinct IPs, then near zero. Every one is
|
||||
a desktop whose family was revoked by reuse-detection during the outage; the desktop clears its cloud session on
|
||||
401 (`clearCloudSessionIfUnchanged`) and stops retrying. Those users must sign in again before the relay sees
|
||||
them. Fresh `/session` sign-ins: 1, 5, 3 per minute at 19:10–19:12. Recovery of controls is now paced by users
|
||||
signing in, not by infrastructure. Total `session-refresh-reuse-detected` events 13:00–19:00Z ≈ 100k, against a
|
||||
~100/hour baseline.
|
||||
**Affected-user count (19:22Z, from `refresh_tokens`):** 23,318 live token families revoked in the window,
|
||||
**21,605 distinct users**. Only ~3,800 desktops had seen their 401 by 19:15Z; the rest were closed or asleep
|
||||
and will find themselves signed out on next launch, so sign-ins will trickle for days.
|
||||
|
||||
**Desktop UX finding (owner's own Mac, 19:22Z):** a revoked desktop keeps showing the account card as
|
||||
"Connected" and the pairing pane as "Orca Relay: Unavailable" / `relay_control_not_active` indefinitely; the
|
||||
local trace writes no relay events. Only quit + relaunch surfaced the sign-out prompt, after which sign-in →
|
||||
relay-token → `/v1/assign` 200 (0.15 s) → working pairing, all within 10 s. Follow-ups: the relay coordinator's
|
||||
401 path should flip the account card to reconnect-required immediately, and the pairing error should say "Sign
|
||||
in again to use Relay" when the cause is an auth failure. Announcement wording: "If Relay shows Unavailable, quit
|
||||
and reopen Orca, then sign in when prompted."
|
||||
|
||||
orca-cloud PR #474 (branch `auth-revoke-only-live-tokens`): caps → 20, disk 250 / max_wal_size 16384 in
|
||||
Terraform, partial index in the schema, `already-revoked` short-circuit. Do not deploy auth to any environment
|
||||
with a large `refresh_tokens` before building the index concurrently there.
|
||||
|
||||
**Wave 1 of the roadmap (2026-09-04 21:35Z onward):** five Opus agents in isolated worktrees: 3.1 grace window
|
||||
(orca-cloud), 4.1+2.3 relay locks + pool timeout, 3.2+4.3 desktop refresh/jitter, 5.1+5.4 observability,
|
||||
2.1 private IP (plan only, both repos). First back: stablyai/orca PR #18717 (crash alert + dashboard). Its key
|
||||
finding: cell exits log to `cos_system` with uppercase `jsonPayload.MESSAGE` and `SYSLOG_IDENTIFIER=docker`,
|
||||
so every earlier `jsonPayload.message:"container die"` count in this doc that read 0 was querying the wrong
|
||||
field. Verified: 87 exits 12–13Z on the agent's filter, 0 in the last 6 h. Monitor dry-run 33922255205
|
||||
dispatched 21:41Z as the Roll 1 gate.
|
||||
Dry-run 33922255205 froze at 21:46Z on `signal_missing cloud_sql.backends`. Cause: Cloud Monitoring published
|
||||
no `num_backends` point for the auth instance between 21:40 and 21:46 (every other minute of the last 100 has
|
||||
one; measured directly via the timeSeries API). A Google-side publish gap, not a database or monitor defect;
|
||||
the monitor's freeze-on-missing rule is correct. The 12–13Z monitor failures were a different cause (active
|
||||
probes reading 0 during the crash cascade). Re-dispatched at 21:50Z.
|
||||
Dry-run #2 (33922844671) froze at 21:52:21Z on `auth.health observed 0` — verdict read from the state.json
|
||||
artifact, not the log (the log only prints checkpoints). Auth served `/health` 200 continuously, including the
|
||||
21:52:05 probe. Cause: the probe requires `/health` AND `/ready` on the first attempt; auth has no `/ready`
|
||||
(404 by design), so every auth sample takes the forced 11 s retry, and on the third sample the retry fetch threw
|
||||
at the network layer on the runner (no request reached Cloud Run) and `check()` recorded the exception as
|
||||
health=false. Neither freeze was fleet health. Fix delegated (relay-ops: a thrown fetch is not a reading; auth
|
||||
does not require `/ready`). **Sequencing constraint for Roll 1:** monitor evidence must be < 5 min old at
|
||||
canary dispatch, so the owner's go must precede the dry-run, and a green dry-run must be followed by the
|
||||
canary dispatch immediately.
|
||||
|
||||
stablyai/orca PR #18719 (3.2 + 4.3, desktop): the replay engine was not the refresh function but
|
||||
`RelayAuthCoordinator.scheduleRetry`, since `shouldRetryRelayConnectionError` treats any non-HTTP error
|
||||
(including a refresh `TimeoutError`) as retryable and re-reads the same stored token on backoff. Fix: refresh
|
||||
gets one 60 s attempt; an ambiguous failure (no status line) records the token and blocks re-sending it for
|
||||
30 s (bounded, not permanent); definitive 5xx gets exactly one retry after re-reading the store; a 401 on an
|
||||
ambiguously-attempted token logs `orca_cloud_refresh_possible_replay`. Lease renewal gets ±10 % full jitter
|
||||
(base shrunk so the latest sample stays ≥ 90 s before expiry); server resets the full 55-min TTL on any rebind
|
||||
(`host-session-registry.ts:736-743`) so early renewal is free. Verified the retry-path claim and both server
|
||||
cites against main.
|
||||
|
||||
2.1 private IP: orca-cloud PR #477 (foundation: servicenetworking API, /24 peering range 10.42.128.0, private
|
||||
network on the instance, `prevent_destroy`; real production plan 3 add / 1 in-place change, staging unchanged)
|
||||
and stablyai/orca PR #18720 (relay: `relay_cloud_sql_private_ip` variable, conditional `--private-ip` in the
|
||||
cell startup template; default false renders byte-identical to main). Findings that change the plan: Google
|
||||
states the private-IP change **restarts the instance** with no in-place path, and it is a one-way door (cannot
|
||||
disable private IP or remove the network link). The director uses the Cloud Run built-in connector, not the
|
||||
relay VPC NAT, so it never consumed the exhausted ports and is out of scope. Disabling public IP later breaks
|
||||
the local proxy workflow and the director. #18720 merges (inert); #477 held for owner decision.
|
||||
|
||||
4.1 + 2.3 relay: stablyai/orca PR #18722. Premise correction: #18521 and #18606 had already bounded and
|
||||
narrowed most of the fleet-wide lock before today; what remained were the sticky-refresh retry (all 23 rows →
|
||||
the one pinned row), reservation reconciliation (23 → the 2 involved rows), a dead pool-default fallback, and
|
||||
an absolute counter write (→ delta with capacity guard). Placement (`assignOnce`) deliberately keeps the
|
||||
ordered inventory lock: least-loaded selection is fleet-wide and dynamic target-only locking previously caused
|
||||
cross-cell cycles; converting it to optimistic snapshot + conditional delta is the remaining 55P03 floor and a
|
||||
follow-up. Pool `statement_timeout` was already 5 s but hardcoded; now env-configurable, `57014` added to the
|
||||
retryable set (it was terminal before), schema DDL on an untimed max:1 pool. Independently re-ran the new and
|
||||
adjacent suites here against 55440: 66/66. Harness note: 55440 is not idempotent across full runs (2
|
||||
pre-existing failures on a second run); reset the schema between runs. Rollout: director first, watch
|
||||
`orca_relay_postgres_transaction_exhausted` and `cellInventoryHoldMsP95` before cells.
|
||||
|
||||
#18719 first CI run failed only on `windows-host-job.win32.test.ts` (EPERM on temp-dir cleanup), a Windows
|
||||
PTY test the PR does not touch and which no other recent run failed on; rerun dispatched rather than waved.
|
||||
|
||||
3.1 grace window: orca-cloud PR #478 merged (not yet deployed; deploy is an owner gate because the startup
|
||||
schema apply adds a nullable column to `refresh_tokens` with a brief ACCESS EXCLUSIVE). Semantics: within
|
||||
`ORCA_CLOUD_REFRESH_ROTATION_GRACE_MS` (60 s default, 300 s cap, 0 = off) a re-presented rotated token gets the
|
||||
SAME successor refresh token + a fresh access token, no revoke, no audit, provided the successor is still the
|
||||
live head. Third presentation / outside window / revoked family: unchanged (revoke + audit). Successor plaintext
|
||||
is stored sealed (AES-256-GCM, key = HKDF of the predecessor token; the DB never holds the key). Cost stated
|
||||
plainly: a stolen token replayed inside 60 s is served once instead of tripping detection; DB-read + stolen
|
||||
predecessor recovers the successor offline until pruned. Rotation now runs in one transaction (proved by a
|
||||
forced-INSERT-failure rollback test; the 8-way race alone did not kill the non-transactional mutant). Verified
|
||||
locally 27/27 incl. the Postgres suite against 55440, and CI ran it on PG 16 and 17 (4/4 each, not skipped).
|
||||
Deploy wiring: env is set by BOTH Terraform and the deploy workflow, with a test pinning all three sources to
|
||||
one value. **Pre-existing bug surfaced:** the deploy script strips every env var it does not own, so the
|
||||
Terraform-set `ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS` (from #476) silently reverts to the compiled default on each
|
||||
release. Latent only because both defaults are 30. Follow-up: add it to `authEnvironment` + the workflow env.
|
||||
|
||||
Monitor probe fix: stablyai/orca PR #18723. A thrown fetch (DNS/TCP/TLS/8 s abort) is now "no reading" and is
|
||||
re-asked once after 1 s; only a second throw is `false`. A non-ok HTTP answer is still `false` with no extra
|
||||
retry. `latencyMs` is the slowest answering round trip, never a sleep. `requiresReady` is per endpoint: auth
|
||||
(no `/ready` by design) is judged on `/health` + latency; director and cells unchanged. No threshold or rule
|
||||
touched; `auth.ready` had no consumer. 81/81 relay-ops tests and 9/9 evidence-script tests locally. The monitor
|
||||
runs at `main` head, so once merged the next dry-run uses it.
|
||||
|
||||
Applying #18717 (22:10Z): the cell-exit log metric `orca_relay_cell_process_exit` is created; the alert policy
|
||||
raced descriptor propagation (404) and is being retried. **Not applied, deliberately:** the dashboard. Its
|
||||
targeted plan drags in `google_logging_metric.relay_snapshot[*]`, and that plan is `32 to add, 21 to destroy`:
|
||||
the Terraform source adds a `region` label to every runtime metric (`EXTRACT(jsonPayload.region)`) which the
|
||||
live metrics do not have, and a label change on a log metric is a delete+create. Replacing 21 live metrics
|
||||
resets their history and would blank the 14 existing relay alert policies during the swap. That is
|
||||
pre-existing drift in the relay root (unapplied since the region work), not something #18717 introduced. It
|
||||
needs its own reviewed apply in a quiet window, ideally with the runtime-metric replacement acknowledged as
|
||||
intentional. Dashboard apply waits on that.
|
||||
|
||||
**Wave 1 closed 22:20Z.** Merged: orca-cloud #478 (grace window); stablyai/orca #18717 (crash alert +
|
||||
dashboard TF), #18719 (desktop no-replay + jitter), #18720 (private-IP flag, off), #18722 (relay per-cell
|
||||
locks + pool timeout), #18723 (monitor probe fix). Applied to production: cell-exit log metric + alert policy.
|
||||
Held for owner: orca-cloud #477 private IP (restart, one-way); the dashboard apply (behind the runtime-metric
|
||||
label drift); the auth deploy carrying #478; Roll 1. Every wave-1 code change now sits on main un-deployed:
|
||||
the next relay image build carries #18722 + #18723's monitor runs at main head already; the next auth deploy
|
||||
carries #478.
|
||||
|
||||
**Landing (2026-09-04 20:50Z–21:02Z, owner: "if you are confident the cloud changes are valid, you can land them"):**
|
||||
|
||||
- Merged: orca-cloud #474, #475, #476; stablyai/orca #18693, #18694, #18698. Neither repo has branch
|
||||
protection or environment reviewers; `verify` / `cloud-verify` green on main after each.
|
||||
- Applied to production by targeted saved plans (each plan asserted create-only / exact-attribute before
|
||||
apply, via `terraform show -json`): 4 relay resources (WAL-checkpoint log metric + 3 alert policies), 8 auth
|
||||
resources (3 log metrics, propagation sleep, 4 alert policies), and the us-central1 NAT
|
||||
(`enable_dynamic_port_allocation` false→true, ports 64..4096). Google's docs: switching to dynamic does not
|
||||
break existing connections when max ≥ 1024 and max ≥ old min; only lowering max or reverting to static is
|
||||
disruptive. asia-east2 NAT deliberately left for after a US soak.
|
||||
- Not applied: the untargeted apps-root plan also carries 4 unrelated drifts (`ORCA_CLOUD_REFRESH_TOKEN_TTL_DAYS`
|
||||
env on the auth service from #476, a skill log exclusion filter change, skill pressure threshold 16→8, an
|
||||
artifacts bucket lifecycle rule) and fails on the 1Password Cloudflare data source locally. The foundation
|
||||
root plans clean (disk 250 / max_wal_size already match). Those drifts belong to whoever runs the next full
|
||||
apps apply in CI.
|
||||
- `deploy-auth-production` on main 8034955 (run 33919143723) **succeeded 21:04Z**: serving revision
|
||||
`orca-cloud-auth-00031-tox` at 100%, previous `00018-4jc`, cap 20, smoke passed on both URLs. First 15 min on
|
||||
the new revision: 31×200 / 1×401 on `/refresh`, max latency 56 ms, no 5xx. The new
|
||||
`refresh_token_prune_cursor` table exists, so the new schema applied.
|
||||
- US NAT soak (21:01–21:06Z): 0 drops, 0 proxy dial errors, 0 cell exits, port_usage 11, sqlMax ~1.07 s.
|
||||
Asia NAT then applied 21:05:28Z from the pre-verified saved plan (same three attributes). The deploy script strips env vars it does not own, so the Terraform
|
||||
TTL var will not be on the new revision until the full apps apply lands; the auth code defaults to 30 d.
|
||||
- Terraform locally needs `GOOGLE_OAUTH_ACCESS_TOKEN="$(gcloud auth print-access-token)"`; ADC is stale.
|
||||
|
||||
**Alerting + NAT follow-ups (19:58Z, superseded by the landing block above):**
|
||||
|
||||
- stablyai/orca PR #18693 (`relay-nat-ports-and-sql-alerts`): both relay NATs switch to dynamic port
|
||||
allocation (64–4096 per VM); new relay-channel alerts for the Cloud SQL WAL checkpoint loop (log metric on
|
||||
`checkpoint starting: wal`, > 3 per 5 min), Cloud SQL disk > 70%, and NAT `OUT_OF_RESOURCES` drops. No
|
||||
existing workflow applies these resources; the PR body carries the targeted plan.
|
||||
- orca-cloud PR #475 (`auth-observability-alerts`): log metrics + policies for auth refresh 401 (> 100 per 5
|
||||
min; Sep 3 baseline 20–80 per hour), 429 (> 20 per 5 min; baseline 0), 5xx (> 10 per 5 min), and Cloud Run
|
||||
p99 latency > 10 s. Production routes to the relay Slack channel.
|
||||
- Desktop stale auth-status fix: stablyai/orca PR #18694 (`desktop-cloud-session-revoked-status`). Main pushes
|
||||
an auth-status-changed IPC when a 401 clears the session; panes re-fetch on mount; the pairing notice says
|
||||
"Your Orca account session expired. Sign in again to use Orca Relay" and hides Retry. StrictMode regression
|
||||
test verified red on the old guard. Does not help desktops already revoked today (session cleared before
|
||||
this code); it fixes every future revocation.
|
||||
- orca-cloud PR #476 (`auth-refresh-token-pruning`): batched `refresh_tokens` pruner as a scheduled Cloud Run
|
||||
job (revoked rows kept 30 d, rotated rows 60 d against a 30 d TTL, 5k-row batches, 200 ms pauses, persisted
|
||||
cursor, per-run budget) plus a 10 s `statement_timeout` on the auth request pool with schema DDL on an
|
||||
untimed connection. Merges cleanly onto #474 and does not need its index (walks the primary key;
|
||||
EXPLAIN-asserted no seq scan). CI ran the Postgres integration tests for real on PG 16 and 17. Ships
|
||||
`auth_token_pruner_enabled = false` in both environments: enabling needs an image digest from a build that
|
||||
contains the new entrypoint. Operating rules once enabled: monitor the run summary's `stopReason` and
|
||||
`deletedRows`, not the exit code (a run that only ever times out exits 0); ~48 M rows drain in ~10 days at
|
||||
200k/hour; deleting them leaves dead tuples, so the 16 GB is not reclaimed without a separate VACUUM FULL or
|
||||
pg_repack pass, which is its own change.
|
||||
- Phone-side copy when the desktop is signed out: stablyai/orca PR #18698 (`phone-desktop-signed-out-reason`).
|
||||
Real path traced: the director resolves the phone to the host's last cell (durable assignment row), and the
|
||||
cell's `acceptClient` rejects with 4404. The only additive slot every shipped peer tolerates is the WebSocket
|
||||
close *reason* (relay-hello and resolve schemas are zod strict; a new close code drops old phones off the
|
||||
host-offline cadence). Desktop closes its control with reason `signed-out` only when the cloud session is gone
|
||||
(null context after a 401, or explicit sign-out); quit and relaunch stay reasonless. Cell remembers it per
|
||||
host for the dormant-assignment TTL, forgets on re-auth, and echoes it as the 4404 close reason; phone
|
||||
renders "Desktop signed out — sign in to Orca on your desktop to reconnect" with the same retry cadence.
|
||||
Old×new matrix in the PR body; nothing changes for any old peer. Merges cleanly with #18694.
|
||||
|
||||
## What actually blocks the roll now (12:58Z summary for the owner)
|
||||
|
||||
0. **Cloud NAT ports** (Finding 11, found 12:55Z): every us-central1 cell reaches Cloud SQL's public IP
|
||||
through a NAT with the default 64 ports/VM; port_usage pinned at 64 and 1,514 dropped SYNs to
|
||||
Cloud SQL:3307 in one 4-min window. This is the 2 s connect stall that kills old-image cells and is
|
||||
still active after the disk loop broke. Fix: `min_ports_per_vm = 1024` (or dynamic allocation) on
|
||||
`google_compute_router_nat.relay_gce` in `cloud/infra/terraform/relay-gce-foundation.tf`, targeted
|
||||
apply; durable fix is a private IP on the Cloud SQL instance. Online, no VM restart.
|
||||
1. **Cloud SQL disk** (Finding 10): 49 GB PD-SSD saturated since 11:58Z, checkpoint loop, fleet-wide
|
||||
4–6 s stalls every ~45 s. Fix: bigger disk and/or `max_wal_size`. Owner: `stablyai/orca-cloud`
|
||||
`infra/terraform-foundation/database.tf` `google_sql_database_instance.auth` (no `disk_size`,
|
||||
`disk_autoresize`, or `database_flags` set today, so Terraform is at defaults: 10 GB initial, autoresize
|
||||
grew it to 49 GB). Add `disk_size = 200` (+ `disk_autoresize = true`) and optionally
|
||||
`database_flags { name = "max_wal_size" value = "4096" }`; production tfvars are
|
||||
`infra/terraform-foundation/environments/production.tfvars`; applied by `deploy-production.yml` in
|
||||
that repo. Online, no restart for disk; `max_wal_size` is also a non-restart flag. Note Terraform
|
||||
`disk_size` below the live 49 GB would be a destructive shrink, so 200 is safe and 49 is the floor. **This is now the first thing to do**; nothing else can pass a
|
||||
15-min gate while it persists, and it is also what is killing the old-image cells several times an hour.
|
||||
2. **Old cell image** (Finding 6): dies on every stall. Fixed by rolling 519f4914 (canary inputs ready).
|
||||
3. **Gate policy**: `directorErrors: 0` and per-cell health probes freeze on any single stall. Recalibrate
|
||||
after 1 and 2, or bypass by hand for the canary.
|
||||
|
||||
## Plan agreed with the owner (2026-09-04 ~06:45Z), in execution order
|
||||
|
||||
Owner: "feel free to improve operations to make things more effective ... continue driving everything e2e
|
||||
until this process is complete." Owner has had multi-day experiences with cell rolls and does not want a
|
||||
9-hour sequential roll.
|
||||
|
||||
1. **Lock-removal PR** (root cause). *Status 08:55Z: pushed as branch `relay-single-row-reservation`
|
||||
(2 commits). Opus adversarial review found one real defect: `acquireActivity` moving a client-chosen
|
||||
activity id across cells locked the old cell's row before the new one, cycling with placement's
|
||||
ascending inventory lock (reviewer reproduced it as paired 55P03s on real Postgres; no 40P01 because
|
||||
lock_timeout == deadlock_timeout == 1 s). Fixed with `lockCellRows` (ordered, 500 ms bound); census now
|
||||
fails on any inline `relay_cells FOR UPDATE` outside the named helpers. Three-cell Postgres test moves
|
||||
an activity high->low while the target row is held; 5/5 revert-mutants fail it. 480 SQLite tests +
|
||||
tsc green. Also fixed a pre-existing test leak (`relay_cell_connection_snapshots`) that made
|
||||
`assignment-control-supersession-postgres` fail on reruns. Reviewer re-verified 65569be3de: cycle
|
||||
repro completes in 7 ms (was 1022 ms + paired 55P03); no remaining out-of-order pair in the store;
|
||||
flagged two evasions in the new census guard, closed in the third commit (whole-statement scan,
|
||||
covers query() too, mutation-checked with both evasions). Headroom Postgres test's one failure is
|
||||
pre-existing on main (verified by swapping in main's store).* Make `activateControl` superseded-control cleanup, `acquireActivity`
|
||||
existing-lease branch, and `changeActivity` use the existing single-row
|
||||
`adjustCellReservationAtomically` instead of the 23-row `lockCellInventory`. Keep the global lock only
|
||||
for placement (`resolve`/assignment) and sweeps. Real-Postgres contention test on port 55440.
|
||||
2. **Faster same-cap rollout workflow.** (a) paced drain instead of `graceMs: 0` so a cell's ~800 hosts
|
||||
re-dial over minutes, not one second (director cap is 5 x 80 = 400 in-flight); (b) cells in a batch run
|
||||
in parallel once drains are paced; (c) post-canary batches use a short freshness check instead of a new
|
||||
15-min dry-run, since the in-job safety recheck already runs before each drain; (d) job timeout > 75 min.
|
||||
Target: 22 cells in ~6 batches x ~25 min.
|
||||
3. **Build image** with (1) merged, then one roll of the fleet with (2). Asia cells c27/c28/c29 first.
|
||||
4. Re-tighten the monitor retries bar; recalibrate the Terraform exhausted alert.
|
||||
5. Consider deleting the 55-min control lease rebind entirely (no recorded reason; liveness is the 75 s
|
||||
watchdog + 90 s activity lease). Separate PR after (1) so its effect is measurable.
|
||||
|
||||
## Faster same-cap rollout: design (step 2 of the plan), from reading the real limits
|
||||
|
||||
What actually bounds parallelism today (measured on the c7 canary, run 33843071283):
|
||||
|
||||
| step | c7 duration | bound by |
|
||||
|---|---|---|
|
||||
| prechecks (recheck, backend init, resolve, verify) | 43 s | none |
|
||||
| isolate + drain + transition wait | 7 min | drain is `graceMs: 0`; `verify-relay-capacity-transition --activity restart-safe` polls until leases drain |
|
||||
| Terraform template + MIG recreate + wait-until stable | 8 min | GCE recreate; per cell, independent |
|
||||
| verify new incarnation + trust proof + restore | 1.5 min | none |
|
||||
|
||||
Real constraints: (1) the director is 5 x 80 = 400 in-flight `/v1/assign`; a `graceMs: 0` drain of ~800
|
||||
hosts pins it at cap for ~2 min (observed 79.75/84.75 p99). (2) `production-cloud-sql-rollout` lease and
|
||||
workflow concurrency group serialise the whole run, by design, and the per-cell job shares it via
|
||||
`holder-key`. Nothing else forbids parallel cells.
|
||||
|
||||
Changes, smallest first:
|
||||
1. **Paced drain.** `HostSessionRegistry.drain(graceMs)` already sends `drain {graceMs}` and closes each
|
||||
session after `graceMs`, but the desktop's `handleDrain` re-dials immediately regardless of graceMs
|
||||
(`relay-origin-pool.ts:150-162`), so graceMs only delays the *close*, not the stampede. Fix on the
|
||||
cell: stagger the drain *send* across sessions over a window (e.g. 800 sessions over 120 s = ~7/s),
|
||||
which needs no desktop change and works for every desktop version in the field. New admin body field
|
||||
`spreadMs` (optional, default 0 keeps today's behaviour); canary script passes `spreadMs: 120000`.
|
||||
Requires the cell to be on an image with the change, so it applies to batches after the first
|
||||
post-lock-fix roll, not to this one.
|
||||
2. **Parallel cells in a batch.** In `cloud-deploy-relay-production-same-cap.yml` make `cell_2..cell_4`
|
||||
`needs: [gate]` instead of chaining, gated on the same evidence (drop the `+75 min x wave-index`
|
||||
allowance, it exists only because of chaining). Each job already takes the rollout lease with the
|
||||
run's `holder-key`, so they re-enter it rather than fail. With paced drains, 4 cells x ~800 hosts
|
||||
over 120 s is ~27 dials/s, well under the director cap. Raise `timeout-minutes` to 90.
|
||||
3. **Post-canary batches skip the 15-min dry-run.** The in-job "Recheck aggregate SQL, pool,
|
||||
reconnect, migration, and selector safety" step (`pnpm incident:relay-preflight`) already runs a
|
||||
live one-shot check before each drain. For `batch-apply` with a sealed `canary-run-id` from the
|
||||
same commit, accept a dry-run of any age (the canary's) plus that live recheck; keep the 15-min
|
||||
requirement for `canary-apply`. Change lands in `relay-monitor-evidence.mjs verify-authority` +
|
||||
`relay-production-same-cap-wave.mjs` + their node:test suites.
|
||||
|
||||
**Correction after reading the cell job (07:35Z):** (2) parallel cells is not a flag flip. Each cell job
|
||||
asserts the exact selector generation `expected + 2 x wave-index` and exact memberships derived from
|
||||
predecessors having completed (`ISOLATED_*`/`RESTORED_*` in the job, `applyExactAdmissionSelector`
|
||||
compare-and-swap), and all cells share one Terraform state lock. Making that concurrent means a batch-level
|
||||
isolate/restore in the gate and a rewrite of the 650-line job's expectations. That is the multi-day trap
|
||||
the owner described. Deferred.
|
||||
|
||||
What is cheap and removes most of the wall-clock: (3). The per-batch 15-min dry-run costs 15 min each
|
||||
*and* fails ~50% of the time on old-image crashes, which is where hours go. Implement: `batch-apply` with a
|
||||
verified canary authority accepts a passed dry-run up to 6 h old and may re-use one already consumed
|
||||
(the consumed-marker check exists to stop replaying stale evidence; the canary binding plus the in-job
|
||||
live preflight at drain time replace it). Files: `relay-monitor-evidence.mjs` (`--after-canary`),
|
||||
`incident-live-preflight-cli.ts` (same flag), the same-cap workflow + job, and both test suites.
|
||||
Revised expectation: 22 cells = 6 sequential batches x ~70 min = ~7 h wall-clock but *unattended-safe*
|
||||
and with one dry-run total, versus today's 6 dry-runs at ~50% each. (1) paced drain rides the lock-fix
|
||||
image.
|
||||
|
||||
## Recommended next steps (superseded by the plan above; kept for history)
|
||||
|
||||
1. Resolve the gate decision above, then: monitor dry-run -> c7 `canary-apply` only -> verify -> stop.
|
||||
Each rolled cell leaves the Finding 6 crash class.
|
||||
2. Merge #18565; publish; a later same-cap roll carries it to cells.
|
||||
3. Remove the global inventory lock from per-connection paths (`acquireActivity` existing-lease
|
||||
branch, `activateControl` superseded-control cleanup, `changeActivity`) by using the existing
|
||||
`adjustCellReservationAtomically` single-row update. Own PR, after the roll.
|
||||
4. Recalibrate the Terraform alert `relay_postgres_retry_exhausted` to 300/300 s (observability root).
|
||||
5. Whether to raise `relayPostgresRetries` is a human call; the data is in Finding 5.
|
||||
|
||||
## Canary blast radius (read before dispatching c7)
|
||||
|
||||
- What `canary-apply` does to c7, in order: isolate (selector -> migration-only, no new
|
||||
assignments), `/v1/admin/drain graceMs:0` (every control on c7 re-dials the director and is
|
||||
reassigned), Terraform template + MIG update to the target image, wait stable, verify new
|
||||
incarnation + exact digest + protocol, prove per-host trust, restore c7 to general admission.
|
||||
On any failure c7 is left isolated (migration-only) with rehome disabled; nothing else is touched.
|
||||
- c7 at 05:20Z: 788 controls, 5 splices, 800 connections. So ~790 desktops re-dial once. The fleet
|
||||
already absorbs this exact event 201 times / 48 h uncontrolled (Finding 6); the controlled version
|
||||
isolates first, so no new assignment lands on c7 mid-roll. Expect a director concurrency blip, not
|
||||
a freeze-class one (six cells at once gave 85; one cell should stay well under 64).
|
||||
- Precedent: the identical workflow (pre-move, in orca-cloud) ran 9 successful `apply` canaries and
|
||||
batches on 2026-08-27 (last: c20 -> 5aedbca5). Its failures that day all stopped at the read-only
|
||||
"Recheck aggregate SQL..." or "Require durable rehome disabled" step, before `MUTATION_STARTED`.
|
||||
The moved copy in this repo has one run: the read-only `verify` of c7 (passed, including WIF auth).
|
||||
- c7 side note: MIG autoheal recreated the c7 instance four times on 2026-09-01 08:02-08:42 PDT
|
||||
at ~13 min spacing. Same crash class as Finding 6 (health check failing during restart loops).
|
||||
|
||||
### Canary observed effect (c7 drain, 2026-09-04 06:10Z)
|
||||
|
||||
- c7 807 controls -> 0 between 06:08:52Z and 06:10:52Z. Director `/v1/assign`: 200s 32 (06:09) -> 2628 (06:10)
|
||||
-> 340 (06:11); 5xx 1969 (06:10) -> 31 (06:11). Director max-concurrency p99 7.9 -> 79.75 (06:10) -> 84.75
|
||||
(06:11), i.e. at the Cloud Run cap of 80 for ~2 min. My pre-dispatch estimate ("well under 64") was wrong.
|
||||
- Confounder: c10 (us-central1, instance 2803000337345335589) crashed 06:09:56Z on the old-image class
|
||||
(Node.js banner + container die), so ~1,600 hosts re-dialed in the same minute, not ~800. Coincidental;
|
||||
the fleet has one of these every ~15 min.
|
||||
- Recovery: 06:13 903 / 06:14 1471 assign 200s from 640 distinct desktop IPs; 503s 78 -> 183 -> 29/min.
|
||||
No cell crash 06:12–06:16Z. Drain step passed ~06:16Z; template/MIG apply started.
|
||||
- 06:16:03–06:17:08Z, during c7's template apply (not its drain): c27 (x4) and c29 (x3) crash-looped on the
|
||||
old-image pg-pool connect timeout in `beginProof`, both MIGs autoheal-recreated (c27's second recreate in
|
||||
40 min). Fleet 23 -> 21 reporting cells, controls 13286 -> 12462, assign 503s 1000/min at 06:17, director
|
||||
concurrency p99 74.8. Cloud SQL CPU 0.70 max, backends 174 max (bar 250). Same multi-cell pattern occurred
|
||||
at 01:31Z (4 cells) and 04:47Z (5 cells) with nothing rolling; the c7 drain's SQL load 6 min earlier may
|
||||
have nudged the pool timeouts but the class is pre-existing. c7 MIG RECREATING onto new template
|
||||
`…20260904061618…` = the expected image swap.
|
||||
- 06:20Z: 849 assign 503s. Closes 06:19:30–06:21: 162x1006 age<5min (hosts bouncing off the recreating
|
||||
c27/c29), 73x4408 + 53x1006 in the 50-min age bin (Finding 3 rotation cohort). Not roll-caused.
|
||||
c7 MIG `recreating=1` on the new template since 06:16:18Z; c27 and c29 MIGs also RECREATING (autoheal).
|
||||
- 06:23:16Z c7 instance restarted in place (MIG RECREATE keeps name/id relay-c7-bwjc / 4545742188814054238),
|
||||
pulled `relay@sha256:85bf6799…` 06:23:37Z, listening + readiness true 06:23:42Z. Apply step passed 06:24Z;
|
||||
verify step running. Isolate -> ready on new image took ~14 min end to end.
|
||||
- Post-restore c7 on new image (06:25:42–06:26:42Z): controls 143 -> 273 -> 377 refilling, sqlQueries
|
||||
~1,500/30 s, `sqlLatencyMsMax` 518 -> 1003 -> 1155 ms, still 55P03 `cell-inventory` retries. So the new
|
||||
image alone does not remove lock waits; the request-path 500 ms cap from #18521 applies to the director's
|
||||
paths, and cell-side `acquireActivity`/`activateControl` still ride the global lock (step 3 in next steps).
|
||||
Watch: does c7's sqlLatencyMsMax settle below the old 1.0–1.2 s pin once refill finishes, and does c7 stop
|
||||
appearing in `container die` (the real win: guardSessionTask).
|
||||
- 08:25Z (2 h after restore): c7 817 controls, 0 crashes since 06:25Z. Fleet crashes last 2 h: c27 x6,
|
||||
c28 x5, all old-image Asia cells. The new image stops the crash class as predicted; it does not move
|
||||
lock latency (c7 sqlLatencyMsMax 1005 ms), which is #18606's job.
|
||||
- Implication for the batch phase: every drain will push director concurrency past the monitor's 64 bar
|
||||
for ~1-2 min. The batch job rechecks safety *before* it drains (read-only step), so that is fine per wave,
|
||||
but never run a monitor dry-run concurrently with a wave, and prefer batches of 2 over 4 until the fleet
|
||||
is on the new image and the crash class is gone.
|
||||
|
||||
## Post-merge dispatch plan for #18606 (image -> director -> cells)
|
||||
|
||||
1. `gh workflow run cloud-publish-relay-production.yml --ref main -f mode=publish` (after the squash lands
|
||||
on main). Resolve the digest by tag, never by parsing the log (it mixes relay and fence-broker digests):
|
||||
`gcloud artifacts docker images describe us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay:sha-<merge-sha> --format='value(image_summary.digest)'`.
|
||||
2. Director: `gh workflow run cloud-deploy-relay-production-director.yml --ref main -f image-digest=<new>
|
||||
-f regional-placement-mode=preserve -f prune-incompatible-revisions=false -f expected-rehome-generation=12
|
||||
-f bootstrap-runtime-identity=false -f predecessor-image-digest=<currently serving digest>`
|
||||
(no monitor evidence needed; requires rehome disabled at gen 12, which it is). Last run 33826514754 used
|
||||
the same shape. Watch director `orca_relay_postgres_transaction_retry` per minute before/after.
|
||||
3. Cells: same-cap `verify` c7 with target=<new>, rollback=85bf6799; fresh dry-run; `canary-apply` c7;
|
||||
then batches (3 per batch, Asia c27/c29/c28 first). Each batch: new dry-run unless the batch-reuse
|
||||
change (design section above) has shipped.
|
||||
|
||||
## Finding 8 (2026-09-04 08:40Z): ten-cell crash cascade during the director deploy, not caused by it
|
||||
|
||||
Timeline: candidate revision 00570-siv created 08:38:39Z, first log 08:39:20Z; traffic still 100% on
|
||||
00565-fes through 08:43 (assign logs by revision). Cell crashes: c28 (5031087219978409220) looped 08:37:55–
|
||||
08:40:07 (9x), then at 08:40:20–08:40:45Z **ten** instances died within 25 s (c10 2803…, 5110…, 532…, 5464…,
|
||||
7536…, 7726…, 8671…, 8928…, 8966…). All old-image `beginProof` pg-pool timeouts. Fleet controls 13,423 ->
|
||||
6,157 by 08:43; assign 503s 3,912 (08:42) and 4,624 (08:43) per minute, director concurrency 85 (cap 80),
|
||||
Cloud Run autoscaled 5 -> 10 instances, Cloud SQL CPU 0.55 -> 0.99. Deploy finished cleanly at 08:45Z with
|
||||
the new director taking the tail of the storm; by 08:46 503s were ~30/15 s, controls 7,913 and rising,
|
||||
director lock retries 29/min (vs 105–157/min pre-deploy) and exhausted 2/min (vs 65/min at 08:36).
|
||||
Same class as 01:31Z (4 cells) and 04:47Z (5 cells) today; this was the biggest. c7, on the new image
|
||||
since 06:25Z, did not crash. What triggered the pool timeouts fleet-wide at 08:40 is not established; Cloud
|
||||
SQL CPU was 0.78–0.88 in the minutes before, the highest of the day, so the cells' 2 s connect timeout is
|
||||
the plausible tipping point under a busy database. Every cell still on 5aedbca5 remains exposed to this.
|
||||
|
||||
## Finding 9 (2026-09-04 08:56Z): #18606 on the director cut lock retries ~10x
|
||||
|
||||
`orca_relay_postgres_retries` per 5 min, director only: 08:21–08:41 windows 419–689 (old image, incl. the
|
||||
crash storm); 08:46/08:51/08:56 (new image 519f4914, refilling ~7k hosts): **61 / 69 / 54**. Exhausted:
|
||||
104–178 -> **11 / 14 / 12**. Inventory hold p95 ~200 ms, max 255 ms, ~366 holds/min. Cells (still old
|
||||
image) 17–44 -> 0–3, because the director no longer holds the 23-row lock on their behalf. This is the
|
||||
first direct measurement of the root-cause fix under real load. Cloud SQL CPU peaked 0.99 during the
|
||||
cascade and is decaying (0.86 at 08:55); the monitor freezes above 0.80, so no dry-run until it clears.
|
||||
|
||||
Fourth cascade 09:00:12–09:00:18Z: c23, c8, c16, c26, c22 (five cells, 11 container-die events in 6 s,
|
||||
all `5aedbca5`, exitCode 1, Node banner, pg-pool `client closed the connection` burst right before). Cloud
|
||||
SQL CPU 0.84 -> 0.78 in the preceding minutes, director concurrency 18–22 (idle), so this one fired
|
||||
*without* a database or director spike. Fleet had just recovered to 13,015. Cadence today: 01:31 (4),
|
||||
04:47 (5), 08:40 (10), 09:00 (5), 09:31 (c13, c23), 09:34 (c23 again, c14, c20, c9; c14/c20 crash-looping),
|
||||
09:39 (c21, c24), 09:55 (c16, c8), 09:59 (c20), 10:05 (c8, c20), 10:19 (c16 stalled, no crash), then a 58-min
|
||||
lull, 11:04 (c9; c28 died 13x in 4 min, autoheal recreate 11:09Z, its 3rd recreate today), 11:17 (c10, c28
|
||||
again, c22, c23, c14 x9 looping; 23 dies in ~90 s; fleet 13.3k -> 10.8k), 11:31 (c14, c23, c25, c15, c24, c19), 11:34 (c20, c26, c29 x4, c14, c27 x3, c25; fleet 13.1k -> 10.3k).
|
||||
Three cascades in 17 min. 11:38–11:45 c27 crash-looped 17x and c28 4x (Asia cells), c29 recreating.
|
||||
11:59 (c21, c9, c10, c23), 12:02 (c19; 4,109 assign 503s that minute, mostly hosts bouncing off the
|
||||
recreating cells, code 1006 age<5min x217), 12:09–12:12 (c19, c27 x6, c28 x5, c13, c22, c15, c26, c14;
|
||||
8 cells, c27/c28 recreating again). Cloud SQL CPU 0.62–0.85 through it. 12:20 (six more cells). Cascade
|
||||
cadence since 11:00 is now ~every 8 min; the waiter has held correctly the whole time and there has been
|
||||
no dispatchable window. Loop continues unattended; findings stop logging each cascade from here unless the
|
||||
class changes. Every cell that has died today is on 5aedbca5; c7 (85bf6799,
|
||||
5.5 h) has not. Cell dies per hour today:
|
||||
01Z 5, 02Z 7, 03Z 4, 04Z 7, 05Z 4, 06Z 9, 07Z 11, 08Z 30, 09Z 26, 10Z 2, 11Z 68+ (to 11:42).
|
||||
Director concurrency pinned at 85 for 09:32–09:33; 503s 4,141 and 4,396 per minute. 09:39: c21, c24
|
||||
(2,870 503s). Crashes per instance 08:10–09:40Z: c28 x14, c27 x5, c23 x5, c22 x4, c14 x4, c13/c20 x3,
|
||||
then c26/c9/c24/c16/c8 x2. Mean gap between cascades since 08:40: ~12 min. Every 15-min gate attempt
|
||||
now has well under even odds; the c7-style canary that ends this needs a gate it can pass. The old image is now cascading roughly hourly regardless of load; the
|
||||
only cell on a fixed image (c7) has 0 crashes in 2.5 h across all four.
|
||||
|
||||
Director 500s: 4 in the 09:00 window, all 2.0 s latency on `/v1/assign` or `/v1/resolve` = pg-pool connect
|
||||
timeout surfacing as a 500. Pre-existing (Sep 3: 03h/08h/16h one each, same 2.0 s shape; 06:09Z today on
|
||||
the old image during the c7 drain). The monitor's `directorErrors: 0` bar freezes on any of these, so a
|
||||
dry-run needs a 15-min window with none; at ~1 per cascade that is a real but modest constraint.
|
||||
|
||||
**Gate observation (09:26Z):** `directorErrors: 0` counts every non-503 5xx on the director, including
|
||||
the monitor's own admin calls. The director on 519f4914 still sees an occasional 2.0 s pg-pool connect
|
||||
timeout (~1 per 20 min under today's Cloud SQL load), which surfaces as a 500 on whichever request drew
|
||||
it. Two consecutive dry-runs (#7, #8) froze on exactly this: one, isolated, 2 s 500. That bar was set for
|
||||
"unexpected director 5xx"; a single connect timeout that the client retries is not an incident. Candidate
|
||||
recalibration (own PR, not done): `directorErrors` 0 -> 2 per 5 min, or exclude the monitor's own
|
||||
user-agent. Not changing it unasked; noting that at ~3 per hour the 15-min gate passes ~1 in 2 attempts.
|
||||
|
||||
**Did the director deploy make cells crash more? (checked 09:45Z)** Cell `container die` per 30 min:
|
||||
06:00 9, 07:00 2, 07:30 9, **08:30 30** (director candidate 08:38, traffic 08:43–08:45; the 10-cell burst
|
||||
was 08:40:20, before the move), 09:00 11, 09:30 12. Per hour today 05:4 06:9 07:11 08:30 09:23 vs Sep 3
|
||||
same hours 2/7/8. So today is 2–3x worse than yesterday and was rising before the deploy; after the deploy
|
||||
it is ~11–12 per 30 min, in line with 06:00–07:30. Cloud SQL backends (~230 max) and new connections
|
||||
(~5k/30 min) are flat across the deploy. Latest crash (c21 09:39:11) is `Connection terminated due to
|
||||
connection timeout` with cause `Connection terminated unexpectedly` in `verifyCellAssignment` <-
|
||||
`beginProof`, the same unhandled path. Conclusion: no evidence the deploy worsened it; the old image's
|
||||
crash rate simply climbed all day. Director lock retries stayed ~10x lower after the deploy.
|
||||
|
||||
**Checkpoint-phase check (10:00Z, negative result):** Postgres checkpoints complete every 5 min at ~:07.
|
||||
Cell crashes bucketed by phase within that 5-min cycle show a mild :00–:29 s cluster today (22 of 103)
|
||||
that is absent on Sep 3 (7 of 114), so checkpoints are not the trigger. Disk write bytes in cascade
|
||||
minutes are at or below the median except 09:00. Cloud SQL memory 0.47, transaction rate flat. The
|
||||
09:55 stall (11 director + 4 cell pg-connect timeouts in the same 4 s) came with `could not obtain lock
|
||||
on row in relation "relay_cells"` from a NOWAIT sweep at 09:55:36, i.e. someone was holding the full
|
||||
inventory at that moment. On the new director that can only be placement or a sweep; on the old cells it
|
||||
is still every rebind. What stalls *connections* (not locks) for 2 s fleet-wide remains unexplained;
|
||||
Cloud SQL is `db-custom-4-15360` REGIONAL PD_SSD 49 GB at 0.5–0.75 CPU when it happens.
|
||||
|
||||
**Stall census (10:01Z):** 33 pg-connect-timeout stall events today (clusters of timeouts < 20 s apart).
|
||||
Before 08:35 they were 1–9 timeouts each and 10–60 min apart; from 08:35 the big ones are 16, 22, 21,
|
||||
17 timeouts and 5–30 min apart. No second-of-minute phase (start seconds spread across all buckets), so
|
||||
not a fixed timer. Cloud SQL backends by state at 09:55: active peaked 42 at 09:52, idle-in-transaction
|
||||
≤ 10, nothing near the 400 ceiling; memory 0.47; disk normal. Each stall is a few seconds where *new*
|
||||
connections to Cloud SQL (via the auth proxy socket) time out at the 2 s `connectionTimeoutMillis`,
|
||||
hitting every process that happens to need a fresh pool connection in that window. Old-image cells die
|
||||
on it (unhandled), new-image director logs a 2 s 500 and continues. Root cause of the stall itself is
|
||||
outside the relay code (Cloud SQL proxy or instance); not chased further here.
|
||||
|
||||
## Finding 10 (2026-09-04 12:40Z): Cloud SQL disk write saturation since 11:58Z is driving the stalls
|
||||
|
||||
`orca-cloud-auth-db` is `db-custom-4-15360` on a **49 GB PD-SSD** (81% used). PD-SSD performance scales
|
||||
with size: 49 GB gives roughly 1,470 write IOPS and ~23 MB/s write throughput. Measured:
|
||||
|
||||
| | before 11:58Z | 11:59Z onward |
|
||||
|---|---|---|
|
||||
| disk write MB/s | 4–6 | **30–50** (over the ~23 MB/s cap) |
|
||||
| disk write IOPS | 500–800 | 800–1,475 (at the ~1,470 cap in 11:59, 12:15, 12:24, 12:34) |
|
||||
| checkpoint `sync=` | 0.07–0.2 s (Sep 3 max 0.65 s, 290 checkpoints) | 2–20 s; 27 of 39 checkpoints in 12Z were >= 2 s |
|
||||
| checkpoints per hour | 12 (timed, every 5 min) | 39 (WAL-triggered, every ~45 s; `write=` fell from 270 s to 30 s) |
|
||||
| Cloud SQL CPU / memory | 0.5–0.8 / 0.47 | same (not the bottleneck) |
|
||||
|
||||
Every 4 s+ fleet-wide SQL stall since 11:04 (11:04, 11:17, 11:31, 11:34, 12:09, 12:10, 12:18, 12:20,
|
||||
12:30) sits inside a slow checkpoint `sync` window; the 12:30:49 checkpoint synced 5.88 s (longest file
|
||||
5.47 s), matching the 12:30:02–41 stall. During fsync the WAL writer stalls and every session waits, which
|
||||
is why the stall hit all 23 cells and the director at once regardless of the relay lock changes. The
|
||||
old-image cells then die on the pool timeout; the new image survives. What raised write volume ~8x at
|
||||
11:58Z is not established (autovacuum ran on every relay table 11:55–11:57 and checkpoints are being
|
||||
forced by WAL volume, so a write amplifier inside Postgres is the leading candidate; relay transaction
|
||||
rate and Cloud SQL network bytes were flat). This is the first cause found today that is *upstream* of
|
||||
the relay code and it explains the afternoon acceleration (11Z 68 dies, 12Z 47 by 12:34).
|
||||
|
||||
Corrections after digging (12:45Z): relay query volume, renewals, reconnects, and assignments per 5 min
|
||||
were **flat** across 11:58 (sqlQ ~330k, renewals ~115k), so the relay did not start writing more. WAL
|
||||
recycling per checkpoint went 7 -> 10–11 files (16 MB each) at 45 s intervals, i.e. WAL output rose from
|
||||
~0.4 MB/s to ~4 MB/s while data-file writes rose to 30–50 MB/s; checkpoints switched from `time` to `wal`
|
||||
triggered at 11:58:24. No Postgres slow-statement or "checkpoints too frequently" lines. This is
|
||||
write amplification inside Postgres (full-page writes after each of the now-frequent checkpoints on
|
||||
hot pages, plus autovacuum on every relay table each minute) on a disk too small for its IOPS ceiling,
|
||||
not new relay load. Instance label `managed_by=terraform`, created 2026-07-09; the instance resource is
|
||||
**not** in `cloud/infra/terraform` (only the database, user, and secret are, via
|
||||
`local.relay_database_instance_name`), so it lives in the other Terraform root (orca-cloud, per
|
||||
[[orca-cloud-terraform-split-findings]]). `storageAutoResize=true` with limit 0, so Cloud SQL will grow
|
||||
the disk only when it fills, not when IOPS saturate; disk is 81% full.
|
||||
|
||||
Onset precisely: the 11:55:37 `time` checkpoint wrote 67,258 buffers (10.5% of shared_buffers, the
|
||||
day's largest) over 163 s and completed 11:58:24. Every checkpoint since has been `wal`-triggered at
|
||||
~45 s spacing (`max_wal_size` reached), each writing 13–20k buffers with 9–11 WAL files recycled. This is a
|
||||
self-sustaining loop: a checkpoint completes -> every subsequent write to a hot page emits a full-page
|
||||
image into WAL -> WAL fills `max_wal_size` in ~45 s -> next checkpoint -> repeat. The relay's hot rows
|
||||
(`relay_cells`, `relay_assignments`, activity leases, cell runtime) are updated tens of thousands of
|
||||
times a minute, so full-page-write amplification is large. Before 11:58 the 5-min timed checkpoints kept
|
||||
WAL well under the limit; a one-off larger checkpoint tipped it over and the disk's write ceiling keeps
|
||||
it there. Query Insights: io_time +30% in the 12:00 bucket, lock_time flat.
|
||||
|
||||
**Owning workflow / mitigation (not applied):** raise the Cloud SQL data disk (PD-SSD IOPS and MB/s scale
|
||||
linearly with GB; 49 -> 200 GB roughly quadruples the ceiling, online, no restart) in the Terraform root
|
||||
that owns `google_sql_database_instance` for `orca-cloud-auth-db`, applied through that root's workflow.
|
||||
A second, flag-level lever is raising `max_wal_size` (default 1 GB) so timed checkpoints resume; that is
|
||||
also a Cloud SQL instance setting in the owning Terraform root. Per the standing rule, not applied from
|
||||
this session. Until then the fleet-wide 4–6 s stalls recur on
|
||||
every slow checkpoint sync, the old-image cells die on each one, and no 15-min gate window will exist.
|
||||
|
||||
## Finding 11 (2026-09-04 12:55Z): **Cloud NAT port exhaustion** on the us-central1 cells is the second stall class
|
||||
|
||||
`google_compute_router_nat.relay_gce` (us-central1, `AUTO_ONLY` IPs, no `min_ports_per_vm`, no dynamic
|
||||
port allocation, i.e. the default **64 ports per VM**). `router.googleapis.com/nat/port_usage` per VM
|
||||
hit **64 = the cap** in exactly the minutes the cells' Cloud SQL proxies logged `dial tcp
|
||||
35.188.82.89:3307: i/o timeout` (12:20–12:22, 12:41–12:43, 12:51–12:53), and
|
||||
`nat/dropped_sent_packets_count` went 0 -> 56/552/590, 82/272/133, 395/1565/1842 in those same minutes.
|
||||
Hourly: port_usage max was 25–50 all of Sep 3 and until 10Z today, 64 in 11Z and 12Z; dropped packets 0
|
||||
until 11Z (219), then 5,491 in 12Z. Open NAT connections rose 400–600 -> 815–874. Every cell's Cloud SQL
|
||||
traffic egresses through this NAT to the instance's public IP (the instance has no private IP:
|
||||
`ipv4Enabled=true`, `privateNetwork` unset). When a VM's 64 ports fill, new TCP SYNs to 3307 are dropped,
|
||||
the proxy's dial times out, and the relay pool's 2 s `connectionTimeoutMillis` fires: that is the exact
|
||||
2 s stall the old image dies on and the new director surfaces as a 500. The dial timeouts hit c7 and c8
|
||||
hardest because they carry the most controls and open the most DB connections.
|
||||
|
||||
What raised port demand today: each old-image crash re-opens a full pool through fresh NAT ports, the
|
||||
autoheal recreates do the same, and the 55P03 retry storms keep more connections mid-transaction, so
|
||||
crashes and NAT exhaustion feed each other. This is why the afternoon accelerated even after the disk
|
||||
loop broke at 12:39.
|
||||
|
||||
**Owning change (not applied):** `cloud/infra/terraform/relay-gce-foundation.tf`
|
||||
`google_compute_router_nat.relay_gce` (this repo): set `min_ports_per_vm = 1024` (or enable
|
||||
`enable_dynamic_port_allocation = true` with `max_ports_per_vm = 4096`) and, if needed, add manual NAT IPs
|
||||
(each IP supplies 64,512 ports across VMs). Online change, no VM restart. The durable fix is giving the
|
||||
Cloud SQL instance a **private IP** and pointing the proxy at `--private-ip`, which takes DB traffic off
|
||||
NAT entirely; that is a Cloud SQL instance change in the orca-cloud foundation root plus a startup-script
|
||||
flag here. Per the standing rule, not applied from this session.
|
||||
|
||||
Direct proof: `resource.type="nat_gateway" AND jsonPayload.allocation_status="DROPPED"` shows **1,514
|
||||
dropped allocations to 35.188.82.89:3307** in 12:50–12:54 alone, every one of them the Cloud SQL public
|
||||
IP. The NAT has zero manual IPs (AUTO_ONLY) and no port settings in Terraform, so it is at Google's
|
||||
default 64 ports/VM. No workflow in this repo applies `relay-gce-foundation.tf` broadly (the roll
|
||||
workflows apply cell templates with `-target`), so the NAT change needs a targeted apply of
|
||||
`google_compute_router_nat.relay_gce`, which is an owner-run Terraform step.
|
||||
|
||||
Original write-up of the symptom before the NAT correlation follows.
|
||||
|
||||
The 12:50:30–12:50:50 stall (every cell 3.7–3.9 s SQL max, six old-image cells died) happened with
|
||||
checkpoints healthy (85 ms) and disk at 6 MB/s, so it is not Finding 10. The cells' Cloud SQL Auth Proxy
|
||||
logged `failed to connect to instance: dial error: dial tcp 35.188.82.89:3307: i/o timeout`. Count of
|
||||
those per hour today: 08Z 1, 11Z 15, **12Z 416**; all of Sep 3: 4. Cloud SQL `up`/backends/connections
|
||||
did not blip. So new TCP connections to the instance's public IP on 3307 are timing out from the cells'
|
||||
proxies in bursts, which is exactly the "2 s connect timeout" the old image dies on. Query Insights for
|
||||
12:49–12:54 attributes 1,380 s of lock wait to the placement CTE (`WITH assignment_state AS
|
||||
MATERIALIZED …`) and 469 s to the single-row reservation UPDATE: the lock queue is the *consequence* of
|
||||
connections stalling mid-transaction, not the cause. Not chased further; candidates are the proxy's
|
||||
connection churn under the crash loops (each recreated cell opens a fresh pool) and the instance's
|
||||
public-IP path. Relay code cannot fix this; it is Cloud SQL / network. Dial timeouts by minute today: 12:20 24, 12:21
|
||||
66, 12:41 22, 12:42 6, 12:51 160, 12:52 137, i.e. bursts of 20–160 s each, and they hit c7 (new image,
|
||||
89 today) and c8 (93) hardest, so it is not the old image's connection churn either. Cloud SQL `up`=1
|
||||
throughout. The proxy dials the instance's public IP `35.188.82.89:3307`; a burst of i/o timeouts to a
|
||||
healthy instance points at the path (public-IP egress / NAT / proxy connection limits), not at Postgres.
|
||||
That is the same 2 s that the old image dies on and that the new director surfaces as a 500.
|
||||
|
||||
## Roll inputs (verified by the read-only `verify` run)
|
||||
|
||||
**Image census from instance templates, 2026-09-04 21:45Z (authoritative, read from `gcloud compute
|
||||
instance-templates`):** 20 serving cells on `5aedbca5` (c8, c9, c10, c13–c16, c19–c29) — the image that exits
|
||||
the process on a Postgres connect timeout (Finding 6); c7 on `85bf6799`; c4, c5, c17, c18 (draining /
|
||||
migration-only) on `0e83408b` / `36a56b10`; c1, c2, c3, c6, c11, c12 (existing-only) on Jul/Aug images. Target
|
||||
for Roll 1 is `519f4914` (director already on it). Monitor dry-run dispatched 21:45Z as the roll gate; waves
|
||||
require owner go.
|
||||
|
||||
|
||||
- target-image-digest `sha256:519f4914217f08cabcdcd34825965db8473ec37c6591553a3af0d65dcdeeb183` (lock fix; supersedes 85bf6799 as target)
|
||||
- previous target `sha256:85bf67993869a769642995d0863f4c2b6b569c3850c2d8390ec2ca5f2b179e28` (c7 is on this; use as c7's rollback)
|
||||
- rollback-image-digest `sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563`
|
||||
- target/rollback rehome protocol 1 / 1; expected-rehome-generation 12; selector generation **112** (110 before the c7 canary)
|
||||
- existing-only c1,c11,c12,c2,c3,c4,c5,c6; migration-only c17,c18; general c10,c13–c16,c19–c29,c7,c8,c9
|
||||
- confirmation for canary: `ROLL_RELAY_SAME_CAP <target-digest> production-gce-c7`
|
||||
- monitor evidence is single-use and must be < 5 min old at dispatch (plus 75 min per predecessor wave)
|
||||
- monitor dry-run dispatch (read-only, runs at `main` head so a merged bar change applies immediately):
|
||||
`gh workflow run cloud-monitor-relay-production.yml --ref main -f mode=dry-run -f expected-selector-generation=110
|
||||
-f expected-existing-only-cells=<existing-only list> -f expected-migration-only-cells=production-gce-c17,production-gce-c18
|
||||
-f expected-general-cells=<general list> -f migration-policy=strict -f recovery-source-cell-id=none -f capacity-cell-id=none`
|
||||
|
||||
## Queries that worked (copy-paste)
|
||||
|
||||
- Cell metrics: `resource.type="gce_instance" AND jsonPayload.event="orca_relay_runtime_metrics"`
|
||||
- Container crashes: `resource.type="gce_instance" AND jsonPayload.MESSAGE:"container die" AND jsonPayload.MESSAGE:"relay@sha256"`
|
||||
- Crash banner: `resource.type="gce_instance" AND jsonPayload.message:"Node.js v24"`
|
||||
- Retries: `jsonPayload.event="orca_relay_postgres_transaction_retry"` (no resource filter to get both)
|
||||
- Director lines are `textPayload`; cell lines are `jsonPayload.message`
|
||||
- Cloud Run concurrency: Monitoring API `run.googleapis.com/container/max_request_concurrencies`
|
||||
- Dry-run final state: download artifact `relay-monitor-dry-run-<run>-<attempt>`, read `*.state.json` (the log's `schemaVersion` lines are only checkpoints, not the final verdict)
|
||||
|
||||
## 2026-09-04 22:50Z onward: owner go received; driving the gates
|
||||
|
||||
Owner: "sure, feel free to drive these." Sequence chosen: Roll 1 first (highest uplift), auth deploy with
|
||||
#478 second, pruner enable third, label drift resolved by matching Terraform to live state, #477 still held.
|
||||
|
||||
| Step | Result |
|
||||
| --- | --- |
|
||||
| Monitor dry-run #19 (gen 112, strict) | **Passed** 23:07:53Z, run 33927238469 attempt 1. First green since the probe fix (#18723). 16 samples, no freeze. Dispatched 22:51:33Z after confirming: 0 `container die` in 3 h, director 5xx in the last 4 h were all 503s (excluded by the `director.errors` filter). |
|
||||
| c8 `canary-apply` onto 519f4914 (rollback 5aedbca5) | **Failed at 23:09:07Z before any mutation**: `relay monitor evidence provenance does not match` in `verify-authority`. Run 33928330631. Gate job passed, `cell_1 / rollout` failed on the manifest check, `seal_canary` skipped, lease released. Cause: the manifest binds `commitSha`; the dry-run ran at main `264c9ed8d2`, the canary dispatched at `--ref main` resolved to `4fab8e2f15` because unrelated PRs merged to main during the 15-minute gate. Verified no side effects: c8 MIG still on template `…c8-20260827…` (5aedbca5), stable, 25 controls; no `/v1/admin/drain` or isolate calls in the director log. |
|
||||
| Constraint learned | Both workflows must run at the **same main commit**. The production environment's deployment branch policy allows only `main`, and the job gates on `github.ref == 'refs/heads/main'`, so a pinned tag/branch is not an option. Any merge to stablyai/orca main during the 15-minute dry-run invalidates the evidence. Mitigation for the retry: dispatch the canary within seconds of the green, and do not merge anything to stablyai/orca main myself during the window. A durable fix (accept evidence whose commit is an ancestor with identical workflow/script content) is a follow-up, not a same-day change to a safety check. |
|
||||
| Label drift (5.x) | Resolved by dropping the `region` label from Terraform to match the 21 live metrics (stablyai/orca #18734, merged). Targeted plan asserted `27 no-op, 9 create, 0 destroy`; applied 23:11Z: 8 `orca_relay_control_*` renewal metrics that had never been applied, plus `google_monitoring_dashboard.relay_incident`. `orca_relay_controls` createTime unchanged (2026-07-13), label extractors unchanged. |
|
||||
| Pruner enable (1.2) | orca-cloud #479 merged: `auth_token_pruner_enabled = true`, image digest of `00031-tox`, `max_rows_per_run = 20000`. Targeted plan asserted 9 create / 0 change / 0 destroy (job, scheduler at `41 * * * *` UTC, two service accounts, five IAM grants). **Not yet applied**: waiting until the roll canary has landed so the first hourly run does not overlap a drain. |
|
||||
| Auth deploy with #478 (3.1) | Dispatched 23:13Z from orca-cloud main `f0fa4b5` (run 33928663526). Candidate startup adds nullable `successor_material` under a brief ACCESS EXCLUSIVE lock. |
|
||||
| Auth deploy result | **Succeeded** 23:15:37Z: `orca-cloud-auth-00035-gos` serving 100 %, cap 20 preserved, 0 5xx. `refresh_tokens.successor_material` present (nullable text); 298 sealed successors written in the first 15 min against 924 rotations; `session-refresh-reuse-detected` at baseline (5 / 15 min). Grace window is live. |
|
||||
| Monitor dry-run #20 | Froze 23:35:38Z on `runtime_power_unknown cell.production-gce-c11.powered`. Two window restarts earlier (23:24, 23:25) on `signal_stale auth.errors` (Cloud Monitoring publish lag 181–255 s vs 180 s bar). Cause: one transient rejection of the per-cell MIG GET in `readResourceInventory` yields `targetSize: null` → `runtimeKnown=false` → hard freeze. c11 is a parked existing-only cell (MIG size 0, stable) and was fine. Not fleet health. Fix delegated: stablyai/orca #18740 (retry the MIG read once, mirroring #18723). Run 33928912676. |
|
||||
| Monitor dry-run #21 | **Green** 23:54Z at main `8064d1f991`, but main had moved to `0a821e5bc8` during the window; the chain re-gated instead of dispatching (the canary would have failed provenance again). Run 33930229711. |
|
||||
| Monitor dry-run #22 | **Green** 00:10Z at `0a821e5bc8`; main moved to `2e80972450`. Re-gated. Run 33931177390. |
|
||||
| Monitor dry-run #23 | Froze 00:18:31Z on `cell.production-gce-c29.latency_ms` 2635 > 2000, the probe's own round-trip from a US runner to asia-east2; c29 controls 17→19 and `sqlLatencyMsMax` flat ~1050 through the minute, no crash, no checkpoint stall. c29 probe max was 0 in the three previous gates, so a one-off. Run 33932092775. |
|
||||
| Blocking constraint | Main receives unrelated merges every 5–10 min (23:08, 23:15, 23:17, 23:40, 23:42, …). A 15-min gate bound to an exact commit cannot be consumed under that traffic. Delegated a durable fix: `verify-authority` accepts evidence whose commit is an ancestor of the canary commit **and** has no diff on the monitor/deployer trusted paths; fails closed on shallow clones or unknown commits. Chain re-armed on dry-run #24 (run 33932679796) meanwhile. |
|
||||
| Monitor dry-run #24 | Froze 00:28:00Z on `director.instances` 4 < 5. Cloud Run active-instance count read 4 for exactly one minute (00:27), 5 in every other minute for 3 h; min/max scale is pinned at 5; no new revision. A routine single-instance recycle. Not fleet health. Bar `directorInstancesMin: 5` with `latest-sum` cannot tolerate that; recalibrate to 4 or use a 3-min window minimum (follow-up, not same-day). Run 33932679796. Chain dispatched #25 (run 33933193511) at `86cd327749`. |
|
||||
| Monitor dry-run #25 | **Green** 00:46Z at `86cd327749`; main moved to `8096cb2803`. Fourth green gate lost to unrelated main traffic (#19, #21, #22, #25). Run 33933193511. Chain's re-gate #26 (run 33934079533) cancelled by me. |
|
||||
| Fixes merged 00:55Z | stablyai/orca #18740 (MIG inventory read retried once before `runtime_power_unknown`; 2 tests) and #18754 (`verify-authority` and the batch canary authority accept evidence sealed at an **ancestor** commit when every trusted monitor/deployer path is byte-identical; fails closed on shallow clones and unknown commits; deploy/rehome jobs now check out with `fetch-depth: 0`; 5 new tests, 18/18 pass). Reviewed both diffs; trusted-path set verified to exist on main. |
|
||||
| Monitor dry-run #27 | Dispatched 00:56Z at `74ad08ec66` (first gate whose evidence the new rule can consume). Run 33934541092. Chain re-armed with the same ancestor + identical-trusted-code rule so an unrelated merge no longer forces a re-gate. |
|
||||
| Monitor dry-run #27 | **Green** 01:11:35Z at `74ad08ec66`; main had moved to `38bde20121` with identical trusted code, so the new rule (#18754) let the chain dispatch. Run 33934541092. |
|
||||
| c8 `canary-apply` #2 (run 33935407461) | Provenance check **passed** (first consumption of ancestor evidence). Isolate → gen 113, drain, template+MIG applied 01:14–01:22, new c8 came up on `519f4914` and `relay_capacity_transition_verified` (migration-only, image exact, heartbeat fresh) at 01:23:50. Then the step's next call, `curl --fail-with-body` to c8 `/v1/admin/runtime-status`, got a **503 with a 27-byte body** at 01:23:51 and the step exited 22. Director `cell-status` at 01:23:50.8 returned 200; c8's own logs show nothing at that second; c8 health/ready both 200 seconds later; backend HEALTHY (the health check had just flipped TIMEOUT→HEALTHY at 01:22:16 and UNKNOWN→HEALTHY at 01:23:47 as the new instance warmed). Read: a single 503 at the load-balancer/warm-up edge on a curl with no retry, on a cell that was already verified healthy one line earlier. Failsafe ran: c8 kept **migration-only**, rehome control disabled, selector gen 113. c8 is serving (40 controls at 01:39, sqlLatencyMsMax ~30 ms) on the target image, just not admitted for general traffic. Nothing to roll back. |
|
||||
| Recovery | The job has an explicit resume path: `mode=rollback` with `rollback-image-digest` = the image the cell already runs skips isolate/apply, verifies, and restores general admission (`ROLLBACK_RESUME=true`). Dispatched gate #28 (run 33936966508) at gen 113 with c8 in migration-only; on green the chain dispatches that resume for c8 with rollback digest `519f4914` and target `5aedbca5` (the validator only requires them to differ). |
|
||||
| Follow-up | The verify step's bare `curl --fail-with-body` needs the same "no reading is not a verdict" retry the monitor got (#18723/#18740); a 503 immediately after `verify-relay-capacity-transition` passed is not evidence of a bad cell. |
|
||||
| Monitor dry-run #28 | **Green** 01:58:59Z at gen 113 with c8 in migration-only. Run 33936966508. |
|
||||
| c8 recovery (run 33937756402, `mode=rollback`, rollback digest = 519f4914) | **Succeeded** 02:02Z. `ROLLBACK_RESUME=true` path: isolate/apply skipped, converged-Terraform check passed, verify passed (`relay_capacity_transition_verified` general, image `519f4914`, heartbeat fresh), activate → **gen 114**, c8 general. No restart, no drain. c8 at 43 controls, sqlLatencyMsMax 36 ms. **c8 is the second cell on 519f4914** (with c7 on 85bf6799). Because the recovery ran as `rollback`, `seal_canary` was skipped, so no canary authority exists for a `batch-apply`; the next cell runs as another `canary-apply`. |
|
||||
| Merged 02:05Z | stablyai/orca #18769: bounded retries on every admin-endpoint curl/fetch in the same-cap job and the rehome/canary/verify scripts (`--retry 3 --retry-delay 2 --retry-connrefused`, per-attempt bodies to a file; script helper 2 attempts on network error or 500/502/503/504 only; 4xx never retried; 650/650 tests). Trusted-path change, so the next gate runs at a commit containing it. |
|
||||
| Pruner enabled (1.2) | Terraform applied 02:06Z (8 creates, then the deploy-identity job IAM grant after a propagation 404, 9/9). Job `orca-cloud-auth-token-pruner`, image `343a0915…`, scheduler `41 * * * *` UTC, budget 20 000 rows/run. First run by hand (exec `sf5ct`): cold start 3m20s, then `stopReason: time-budget` at 480 s: 73 batches, 365 000 scanned, **1 040 deleted** (1 021 revoked, 19 expired, 0 rotated), ~6.4 s/batch of 5 000, `completedFullPass: false`. No errors, no lock-wait or checkpoint alert. Scan-bound, not budget-bound: at this pace a full pass over the table takes many hourly runs, and the row budget is never the limiter. Leave the budget alone; watch hourly runs for `stopReason` and a rising `deletedRows` as the cursor reaches the rotated backlog. |
|
||||
| Monitor dry-run #29 | **Green** 02:20:58Z at gen 114, main `e2b70a5eba` (contains #18740, #18754, #18769). Run 33938052374. |
|
||||
| c9 `canary-apply` (run 33938818286) | **Succeeded end to end** 02:21–02:34Z: isolate → gen 115, drain, template+MIG to `519f4914`, verify passed on the first try (retry-hardened step), trust proof, activate → **gen 116**, general. `seal_canary` **succeeded**: batch authority now exists. c9 at 38 controls, sqlLatencyMsMax 33 ms. No `container die` in 30 min. Three cells on new images (c7 `85bf6799`, c8 and c9 `519f4914`); 17 serving cells still on `5aedbca5`. |
|
||||
| Monitor dry-run #30 | Dispatched 02:36Z at gen 116 (run 33939533990). On green the chain dispatches **batch 1**: `batch-apply` c10,c13,c14,c15 bound to canary run 33938818286 (sealed at gen 116, same commit `e2b70a5eba`). Preflight: all four on `5aedbca5`, MIGs stable, no crash in 20 min. Sequential cells inside the job (wave-index 0..3), each with its own isolate/drain/apply/verify/restore, so ~12 min per cell, ~50 min total. |
|
||||
| Monitor dry-run #30 verdict | **Green** 02:52:15Z at gen 116, `e2b70a5eba`. |
|
||||
| Batch 1 (run 33940290163) | Dispatched 02:52:27Z: `batch-apply` c10,c13,c14,c15, canary authority run 33938818286, same commit. |
|
||||
| Batch 1 attempt 1 (run 33940290163) | **Failed at 02:54:39Z in the live preflight, before any mutation**: `relay live preflight failed: cloud-monitoring/signal_stale`. The step's `--retry-freshness` (5 attempts, 15 s apart, freshness-only codes) is passed only for `WAVE_INDEX != 0`; the first cell takes a single sample, so one Cloud Monitoring publish lag > 180 s at that instant fails the batch. Every candidate series was current again by the time I checked. c10 untouched (template `…c10-20260827…`, 47 controls), no selector write, gen still 116, failsafe no-op. Gate #31 dispatched 02:57Z (run 33940508865); chain re-dispatches the same batch (canary authority 33938818286 still valid: same gen 116, same commit). Fix delegated: wave 0 gets the same freshness retry. |
|
||||
| Monitor dry-run #31 | **Green** 03:13:26Z at gen 116; main at `cb7f7dd11a` with identical trusted code. Run 33940508865. |
|
||||
| Batch 1 attempt 2 (run 33941253533) | Dispatched 03:13:38Z: c10,c13,c14,c15, canary authority 33938818286. Runs at `cb7f7dd11a` (batch authority is accepted across the ancestor since trusted paths are unchanged). |
|
||||
| Merged 03:14Z | stablyai/orca #18778: `--retry-freshness` on every same-cap wave including the first, and the retry loop now stops before the next wait would push evidence past the wave's age bound (it was checked only at entry before). Twin carve-out in the capacity job filed as a follow-up. |
|
||||
| Batch 1 cell 1 (c10) | **Succeeded** 03:14–03:27Z (preflight, drain, apply, verify, restore). c13 started 03:27Z. |
|
||||
| Batch 1 cell 2 (c13) | **Succeeded** 03:27–03:38Z. c14 started 03:38Z. |
|
||||
| Batch 1 cell 3 (c14) | **Succeeded** 03:38–03:50Z. c15 started 03:50Z. |
|
||||
| Batch 1 complete (run 33941253533) | **All four succeeded** 03:13–04:00Z: c10, c13, c14, c15 on `519f4914`, selector **gen 124**. Fleet at 936 controls, 23 cells. Two `container die` at 03:35:41/44 were **c13's new container** exiting during boot (`applyPostgresSchema` → `Connection terminated due to connection timeout`, exit 1, 2 s runtime each) because the `cloud-sql-proxy` sidecar had not finished starting; the third start at 03:35:45 succeeded and c13 has been serving since (57 controls). A boot-order race in the container spec, not a serving-cell crash. Follow-up: schema pool should wait for the proxy socket, or the container should depend on the proxy's readiness. **8 cells on new images** (c7 85bf6799; c8, c9, c10, c13, c14, c15 519f4914), 12 on `5aedbca5`: c16, c19–c26 (US), c27–c29 (Asia). |
|
||||
| Monitor dry-run #32 | **Green** 04:19:50Z at gen 124, main `436ef827dd` (contains #18778). Run 33943539025. |
|
||||
| c16 `canary-apply` (run 33944255902) | Dispatched 04:20:02Z. On success it seals the authority for batch 2 (c19,c20,c21,c22). |
|
||||
| c16 canary (run 33944255902) | **Succeeded** 04:20–04:32Z, activate → gen 126, batch authority sealed. 9 cells on new images. |
|
||||
| Monitor dry-run #33 | Failed 04:58:56Z on `continuity_deadline_exceeded` (1 500 004 ms > 1 500 000 ms). One `signal_stale cloud_sql.lock_waits` at 04:46 (189 s vs 180 s bar, Cloud Monitoring publish lag) restarted the 15-min window at sample 12; the restart could not complete inside the 25-min continuity cap. No health failure at any sample; no `container die` since c16's own boot race at 04:30. Run 33944873727. Chain re-gates. Note for recalibration: `cloudDataMaxAgeMs: 180000` vs observed Cloud Monitoring publish lag of 181–255 s has now cost three gates (#20 twice, #33). |
|
||||
| Freshness recalibration | stablyai/orca #18798 (open, merge after batch 2 dispatch): `cloudDataMaxAgeMs` 180 s → 330 s, derived from Google's documented visibility delays (Cloud Run 60+120 s, Cloud SQL 60+165 s) and the 5-min window-sum query (a label series that stops emitting reads as up to 300 s old while its sum is complete, which is the 255 s `auth.errors` case) plus ~30 s collect latency. Director-admin and the lock-wait carry keep their own 180 s pins. A freshness-only failure may miss 2 consecutive samples without restarting the window; the sample still counts and is still threshold-checked; a 3rd miss, collector failure, runner gap, or any breach restarts/freezes as before. 92/92 tests. |
|
||||
| Monitor dry-run #34 | **Green** 05:17:31Z at gen 126, `436ef827dd`. Run 33946093029. |
|
||||
| Batch 2 (run 33946819345) | Dispatched 05:17:43Z: c19,c20,c21,c22, canary authority 33944255902 (c16). |
|
||||
| Merged 05:19Z | stablyai/orca #18798 (freshness bar 330 s + two-sample tolerance). Next gate runs at a commit containing it. |
|
||||
| Batch 2 cell 1 (c19) | **Succeeded** 05:19–05:32Z. c20 started. |
|
||||
| Batch 2 cell 2 (c20) | **Succeeded** 05:32–05:43Z. c21 started. |
|
||||
| Batch 2 cell 3 (c21) | **Succeeded** 05:43–05:59Z. c22 started. |
|
||||
| Batch 2 complete (run 33946819345) | **All four succeeded** 05:17–06:12Z: c19, c20, c21, c22 on `519f4914`, selector **gen 134**. Fleet at 1 090 controls, 23 cells, refresh 401s at baseline (1–4 per 3 min). One `container die` at 06:08:45 was **c22's new container** exiting during boot (exit 1, 2 s runtime; started 06:08:43, restarted 06:08:46 and serving since), the same proxy-sidecar boot race seen on c13 and c16. No serving-cell crash. **Census: 15 of 23 serving cells on new images** (c7 `85bf6799`; c8–c10, c13–c16, c19–c22 `519f4914`), 7 on `5aedbca5`: c23–c26 (US), c27–c29 (Asia). Next: gate at gen 134 → canary c23 → batch c24,c25,c26; then canary c27 → batch c28,c29. |
|
||||
| Monitor dry-run #35 | **Green** 06:32:01Z at gen 134, `b33d1972bc` (contains #18798, first gate at the 330 s freshness bar). Run 33949334606. |
|
||||
| c23 `canary-apply` (run 33950075843) | Dispatched 06:32:13Z at main `b0c67eaf88` (ancestor gate SHA, identical trusted code). On success it seals the authority for batch 3 (c24,c25,c26). |
|
||||
| c23 canary (run 33950075843) | **Succeeded** 06:32–06:46Z, activate → gen 136, batch authority sealed. No `container die` during boot. 16 of 23 serving cells on new images; 6 on `5aedbca5` (c24–c26 US, c27–c29 Asia). |
|
||||
| Monitor dry-run #36 | Dispatched 06:46Z at gen 136, run 33950746574 (`58553bfe1c`). On green the chain dispatches batch 3 (c24,c25,c26) under canary authority 33950075843. |
|
||||
| Monitor dry-run #36 result | **Green** 07:02:49Z at gen 136, `58553bfe1c`. |
|
||||
| Batch 3 (run 33951468008) | Dispatched 07:03Z: c24,c25,c26, canary authority 33950075843 (c23). |
|
||||
| Batch 3 cell 1 (c24) | **Succeeded** 07:04–07:18Z. c25 started. |
|
||||
| Batch 3 cell 2 (c25) | **Succeeded** 07:18–07:31Z. c26 started. |
|
||||
| Batch 3 complete (run 33951468008) | **All three succeeded** 07:03–07:44Z: c24, c25, c26 on `519f4914`, selector **gen 142**. Fleet at ~1 230 controls, 23 cells, refresh 401s at baseline. **Zero `container die`** during the batch (no boot race on c24–c26). **All 20 US serving cells now on new images** (c7 `85bf6799`; c8–c10, c13–c16, c19–c26 `519f4914`). Remaining on `5aedbca5`: c27, c28, c29 (asia-east2, probe hard cap 3000 ms). |
|
||||
| Monitor dry-run #37 | Dispatched 07:48Z at gen 142, run 33953555224 (`4c5077d57a`). On green the chain dispatches the c27 canary (first Asia cell). |
|
||||
| Monitor dry-run #37 result | **Green** 08:04:24Z at gen 142, `4c5077d57a`. |
|
||||
| c27 `canary-apply` (run 33954264945) | Dispatched 08:04Z, first Asia cell (asia-east2-a). On success it seals the authority for batch 4 (c28,c29). |
|
||||
| c27 canary (run 33954264945) | **Failed closed before any mutation** 08:07:25Z at "Verify exact current generation, digest, cap, and rollback point": `runtime predecessor mismatch fields=regionalRehomeProtocol`. **Operator input error, not a cell fault**: the chain script hardcoded `target-rehome-protocol=1 / rollback-rehome-protocol=1` for every cell, but `relay_region_rehome_source_cell_ids` lists only the 16 US cells (c7–c10, c13–c16, c19–c26), so the Asia startup template omits `ORCA_RELAY_REHOME_*` and c27–c29 report protocol 0 by design. `MUTATION_STARTED` never set, failsafe no-op, selector stays gen 142, c27 still serving on `5aedbca5`, no `container die`. Gate #37 evidence consumed. Fix: chain script now takes `PROTO`; Asia round dispatches with protocol 0 (the per-host trust proof step is protocol-gated and skips, as designed for non-source cells). Follow-up: the job already reads `relay_region_rehome_source_cell_ids`; it could derive the expected protocol from membership instead of trusting the operator input. |
|
||||
| Monitor dry-run #38 | Dispatched 08:12Z at gen 142, run 33954621425 (`e95d247be1`). On green the chain dispatches the c27 canary with protocol 0. |
|
||||
| Monitor dry-run #38 result | **Green** 08:28:36Z at gen 142, `e95d247be1`. |
|
||||
| c27 `canary-apply` #2 (run 33955359385) | Dispatched 08:28Z with `target/rollback-rehome-protocol=0`. |
|
||||
| c27 canary #2 (run 33955359385) | **Failed closed, no mutation** 08:31:19Z. Predecessor check passed with protocol 0; the isolate step then died at argument parsing: `production capacity target is not approved`. The same-cap job shells out to `prepare-relay-production-capacity-canary.mjs` for isolate/drain/activate, whose `PRODUCTION_CAPACITY_CELL_IDS` allowlist is the 16 US capacity cells (c7–c26), while the same-cap wave validator (`SAME_CAP_CELLS`) approves all 19 serving cells including c27–c29. The Asia cells have never been through this job (their Aug 14 rollout used the asia-topology workflow). Both the isolate step and the failsafe threw before any HTTP call, so `MUTATION_STARTED=true` was written but nothing was isolated: selector stays gen 142, c27 general and serving on `5aedbca5`, no `container die`. Gate #38 evidence consumed. Fix: stablyai/orca #18811 (`--approved-cells same-cap` on all four invocations, default unchanged for the US capacity job, census test over every `SAME_CAP_CELLS` member × isolate/drain/activate + the job's cell-shape bash block; 525/525 script tests). Sweep of the other job scripts found no further Asia blocker; gate #39 (run 33955668701) dispatched at gen 142 to prove the selector is unchanged before the next attempt. |
|
||||
| Monitor dry-run #39 | **Green** 08:51:28Z at gen 142: independent proof the selector was untouched by both failed c27 attempts. Not used for dispatch (its commit predates #18811). |
|
||||
| Merged 08:51Z | stablyai/orca #18811 → main `12e05203a4`. |
|
||||
| Monitor dry-run #40 | Dispatched 08:51Z at gen 142 on main `12e05203a4` (contains #18811), run 33956408337. On green the chain dispatches the c27 canary, protocol 0, third attempt. |
|
||||
| Monitor dry-run #40 result | **Green** 09:08:03Z at gen 142, `12e05203a4`. |
|
||||
| c27 `canary-apply` #3 (run 33957151726) | Dispatched 09:08Z, protocol 0, on main containing #18811. |
|
||||
| c27 canary #3 (run 33957151726) | **Failed after isolate; failsafe held** 09:17:21Z. Live check 09:26Z: c27 at 0 controls (drained), template still `…20260814235757`, c28/c29 absorbed the hosts (37 each), fleet 1 404 controls / 23 cells, refresh 401s baseline, no `container die` in 60 m. Predecessor check and allowlist passed; isolate → **gen 143** (c27 migration-only), drain sent (graceMs 0, hosts reconnected via director to c28/c29/US). Terraform plan built correctly (template replace + MIG update to `519f4914`), then `validate-relay-capacity-plan.mjs --mode same-cap-cell` rejected it: `cell plan does not contain the reviewed image and capacity`. Its same-cap rule demands exactly one `ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT` and one `ORCA_RELAY_REHOME_AUDIENCE` printf in the startup script; Asia templates omit both because c27–c29 are not rehome sources (same root as attempt 1, third US-only assumption in the job). **No apply ran**: c27 template unchanged, still `5aedbca5`, isolated and draining (drain is one-way in-process; only a restart clears it). Failsafe re-asserted migration-only at gen 143 and rehome disabled. Recovery plan: fix validator (protocol-0 path: require the rehome lines *absent*), merge, gate at gen 143, then `mode=rollback` with rollback-image=`519f4914` (the failed-canary re-entry path; accepts draining + migration-only) to restart c27 onto the target image and restore it; then single-cell canaries for c28 and c29 (batch needs ≥2 cells). |
|
||||
| Plan-validator fix | stablyai/orca #18818 (merged 09:41Z → main `9f2a9a248e`): `validate-relay-capacity-plan.mjs --regional-rehome-protocol 0|1` in same-cap-cell mode; protocol 0 requires the rehome lines *absent*, protocol 1 unchanged; both plan-validation calls in the job pass `DESIRED_REHOME_PROTOCOL`; census test now validates a correct plan for every `SAME_CAP_CELLS` member at its tfvars-derived protocol. 529/529. Residual: the operator-supplied protocol is still unbound for Asia cells (no `SOURCE_CELLS` cross-check outside us-central1), so a wrong value fails late at plan validation rather than early; deriving it from membership is the checklist follow-up. |
|
||||
| Monitor dry-run #41 | Dispatched 09:42Z at gen 143 (c27 expected migration-only) on main `9f2a9a248e` (contains #18811 + #18818), run 33958728141. On green: c27 recovery via `mode=rollback`, rollback-image `519f4914`, protocol 0, confirmation `ROLL_BACK_RELAY_SAME_CAP`. |
|
||||
| Monitor dry-run #41 result | **Green** 09:58:51Z at gen 143, `9f2a9a248e`. |
|
||||
| c27 recovery #1 (run 33959789773, `mode=rollback`) | **Failed closed, no mutation** 10:09:21Z at `Verify monitor evidence provenance`: `relay monitor dry-run authority is incomplete or stale`. The dry-run authority is valid for 5 min after `completedAt` at wave 0 (`EVIDENCE_MAX_AGE_MS`); the gate completed 09:58:51Z but the operator poller (20 s `gh run view` loop) only observed completion at 10:07:09Z during a local network outage, so the dispatch landed at 10:07:11Z, 8 m 20 s after completion. Failed before the rollout lease, isolate, or any Terraform step; c27 unchanged (migration-only, drained, `5aedbca5`, gen 143). Every prior canary dispatched ≤15 s after gate green, so this is a dispatch-latency miss, not a job defect; the freshness bound behaved as designed. |
|
||||
| Monitor dry-run #42 | Dispatched 18:39Z at gen 143 on main `af82126058` (trusted paths byte-identical to `9f2a9a248e`), run 33984753269. Recovery script re-armed behind it (same `mode=rollback` onto `519f4914`, protocol 0). |
|
||||
| Monitor dry-run #42 result | **Green** 18:55:47Z at gen 143, `af82126058`. |
|
||||
| c27 recovery #2 (run 33985902062, `mode=rollback`) | **Failed closed, no mutation** 19:05:02Z, same `authority is incomplete or stale`. Dispatch landed 19:02:39Z, 6 m 52 s after the gate completed. Root cause of both misses is the operator laptop sleeping during the 15 min gate wait (`pmset -g log`: asleep 18:52:28Z → 19:02:17Z; the morning miss coincided with a sleep/dark-wake cycle too), so the 20 s poller never ran inside the 5 min window. Not a job or evidence defect: the freshness bound did its job. Operator fix: poller now runs under `caffeinate -i`. |
|
||||
| Monitor dry-run #43 | Dispatched 19:06Z at gen 143 on main `af82126058`, run 33986121849. Recovery armed behind it under `caffeinate`. |
|
||||
| Monitor dry-run #43 result | **Green** 19:22:53Z at gen 143, `af82126058`. |
|
||||
| c27 recovery #3 (run 33986948522, `mode=rollback`) | **Failed closed, no mutation** 19:25:48Z. Dispatched 13 s after gate green (authority accepted this time), then the live preflight recheck failed: `relay live preflight failed: active-probe/threshold_max`. That is the 2 000 ms `endpointLatencyMs` bar on one endpoint's slowest /health or /ready round trip from the runner (8 s fetch timeout, one retry). The error names no endpoint and the job log prints none; gate #43 had zero failures across 16 samples, so this was a transient probe slow-down in the ~3 min between gate and preflight. Live probe 19:32Z from the operator: director and auth ~130–190 ms, US cells ≤540 ms, Asia cells 690–1 315 ms (c28/c29 /health ~1.3 s, the closest to the bar; c27 ~0.9 s). Existing-only cells c1–c3, c6, c11, c12 return 503 on both paths as expected (unpowered). Failed before the rollout lease, isolate, or any Terraform step; c27 unchanged. Follow-up (checklist): preflight should print the failing signal and observed value. |
|
||||
| Monitor dry-run #44 | Dispatched 19:33Z at gen 143 on main `062db77118`, run 33987646501. Recovery re-armed behind it. |
|
||||
| Monitor dry-run #44 result | **Frozen red** 19:50:01Z after 13 samples: `active-probe/threshold_max cell.production-gce-c27.latency_ms observed=2568 threshold=2000`. No other failure, no continuity event, no `container die` fleet-wide in 60 m. `/health` is a static JSON reply (`app.ts`), so the slow round trip was `/ready` (the probe reports the max of the two) or the path to the cell. Cloud SQL logs for 19:49:38Z–19:51:58Z show six `could not obtain lock on row in relation "relay_cells"` errors and a time-triggered checkpoint completing at 19:50:36Z (write phase 270 s, the spread target, not a stall). c27 is drained with 0 controls, so its `/ready` dependency check was the only thing it was doing. Recovery script stopped as designed (no auto re-gate). Operator probe 19:53Z: c27 and c28 both bimodal, ~0.27 s or ~0.89 s per `/health` from the US, identical shape, nothing c27-specific. Attributing the one 2.6 s sample to the same shared-DB contention that produced the lock errors is the best available reading; the retry at gate #45 tests whether it recurs. |
|
||||
| Monitor dry-run #45 | Dispatched 19:53Z at gen 143 on main `062db77118`, run 33988383401. Recovery re-armed behind it. |
|
||||
| Monitor dry-run #45 result | **Frozen red** 19:54:47Z after 3 samples, same signal: `cell.production-gce-c27.latency_ms observed=2668 threshold=2000`. Two gates in a row now attribute a >2 s round trip to c27 while every other cell passes. |
|
||||
| c27 `/ready` tail analysis | `/health` is static; `/ready` (`relay-readiness.ts`) fetches the auth JWKS (2 s timeout) then runs `SELECT 1`, cached 10 s. Operator probes 19:57Z–20:00Z, 15 each from the US: c27 and c28 have the **same** tail (0.27 s / 0.88 s modes, then 1.3 s, then 2.17–2.27 s at the top); US cells c8/c20 sit at 0.08–0.18 s. Auth JWKS latency over the last hour: 400 requests, max 20 ms, none over 1 s. So the tail is cell→Cloud SQL (US) round trips plus the runner→Asia hop, not auth and not c27-specific; c27 is drained (0 controls) so nothing local competes. Cloud SQL `could not obtain lock on row in relation "relay_cells"` runs at 17–78 per 10 min all day (NOWAIT inventory locks, expected under placement bursts) with no spike in the failing minutes. The bar (`endpointLatencyMs` 2 000 ms, one shot per minute, max of two paths) leaves Asia cells ~10% of samples from tripping; the gate got unlucky twice on c27 and lucky on c28/c29. Not a health finding. |
|
||||
| Monitor dry-run #46 | Dispatched 20:01Z at gen 143 on main `062db77118`, run 33988810139. Recovery re-armed behind it. If this also freezes on an Asia probe, the next move is a per-region latency bar (or p50 over the window) in `incident-monitor.ts`, reviewed and merged before further Asia gates rather than retrying blindly. |
|
||||
| Monitor dry-run #46 result | **Frozen red** 20:07:44Z after 7 samples, third time on `cell.production-gce-c27.latency_ms` (observed 2 685). Operator 40-sample `/ready` probe per Asia cell at 20:10Z: c27 p50 0.88 s / p90 2.15 s / max 2.26 s / 6 over 2 s; c28 p50 0.88 / p90 1.25 / max 2.25 / 1 over; c29 p50 0.88 / p90 0.89 / max 1.27 / 0 over. All 200. `/ready` (`relay-readiness.ts`) fetches the auth JWKS in us-central1 then `SELECT 1` on Cloud SQL in us-central1, so an Asia cell's readiness is two trans-Pacific hops plus the runner→Asia hop; the fleet-wide 2 000 ms bar was calibrated on US cells (0.08–0.5 s). c27 being drained and idle has no local load, so this is path latency, not health. **Stopped retrying gates.** Fix in flight: per-region `cell.<id>.latency_ms` bar (us-central1 stays 2 000, asia-east2 4 000; hard faults still caught by the health/ready equal-1 checks and the 8 s probe timeout) plus attributable preflight failure messages, via review + CI before the next Asia gate. |
|
||||
| Merged 20:33Z | stablyai/orca #18877 → main `a3c1d32995`: per-region `cellEndpointLatencyMs` (us-central1 2 000, asia-east2 4 000; director/auth rules and the `endpointLatencyMs` key unchanged), region carried from tfvars onto every cell expectation, preflight failures now print `source/code signal observed= threshold=`. relay-ops 95/95, cloud suite 633 + 529 + 148 green. |
|
||||
| Monitor dry-run #47 | Dispatched 20:34Z at gen 143 on main `a3c1d32995` (first gate with the per-region bar), run 33989896150. Recovery re-armed behind it. |
|
||||
| Monitor dry-run #47 result | **Green** 20:38:09Z at gen 143 on `a3c1d32995`: first gate under the per-region bar, 16/16 samples, no Asia latency failure. |
|
||||
| c27 recovery #4 (run 33990715317, `mode=rollback`) | **Success** 20:51Z. Dispatched 13 s after gate green. Isolate re-asserted migration-only at gen 143 (already isolated, no change), Terraform applied the same-cap template `…20260905204141` and the MIG replaced the instance, new incarnation on `519f4914`, protocol 0, transition verifier passed at migration-only (1 180 assignments carried, hard cap 3 000, heartbeat fresh), then activate → **gen 144**, c27 general, verifier passed again. No `container die` fleet-wide 19:55Z–20:52Z. c27 now runs the target image; c28/c29 remain on `5aedbca5` (template `…20260814235757`). |
|
||||
| Monitor dry-run #48 | Dispatched 20:53Z at gen 144 (c27 back in general, MIG = c17,c18) on main `61ebffa86e` (trusted paths identical to `a3c1d32995`), run 33991385880. On green the chain dispatches the c28 `canary-apply`, protocol 0. |
|
||||
| Monitor dry-run #48 result | **Green** 21:08Z at gen 144, 16/16 samples, no Asia latency failure. Main had moved to `5cec2c2dfc`; the chain verified the trusted paths were identical to the gate commit and dispatched 12 s after green. |
|
||||
| c28 canary (run 33992169289, `canary-apply`) | **Success** 21:27Z. Isolate → migration-only at **gen 145**, drain already clear, verifier passed on the old image (1 220 assignments carried, hard cap 3 000, heartbeat fresh), Terraform applied same-cap template `…20260905211352`, new incarnation on `519f4914` at protocol 0, verifier passed again at migration-only, activate → **gen 146**, c28 general, verifier passed (1 219 assignments). Seal step recorded the canary. No `container die` fleet-wide 21:08Z–21:30Z. Only c29 remains on `5aedbca5`. |
|
||||
| Monitor dry-run #49 | Dispatched 21:33Z at gen 146 (c28 back in general, MIG = c17,c18) on main `dce5ebd83d` (trusted paths identical to `a3c1d32995`), run 33993075948. On green the chain dispatches the c29 `canary-apply`, protocol 0, the last Roll 1 cell. |
|
||||
| Monitor dry-run #49 result | **Frozen red** 21:52:24Z, `active-probe/continuity_deadline_exceeded observed=1500005 threshold=1500000`. One continuity event at 21:41:27Z, `cloud-monitoring/collector_failed` (a Cloud Monitoring read failed, not tolerated), which reset the continuous window at sample 14; the restarted window reached 10 samples before the 25-minute lineage cap (`INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS`) expired. No health failure in any of the 25 samples, no Asia latency failure, no `container die`. Monitor-side transient, not a fleet finding. The chain re-gated automatically after its 2-minute back-off. |
|
||||
| Monitor dry-run #50 | Dispatched 21:54Z at gen 146 on main `51eed5a1bc`, run 33994385666. **Green** 22:10Z, 16/16 samples. Main had moved to `d7767fb196`; trusted paths identical to `a3c1d32995`. Chain dispatched the c29 `canary-apply` (run 33995164002, protocol 0) 12 s after green. |
|
||||
| c29 canary (run 33995164002, `canary-apply`) | **Success** 22:27Z. Isolate → migration-only at **gen 147**, verifier passed on the old image (1 199 assignments), Terraform applied same-cap template `…20260905221622`, new incarnation on `519f4914` at protocol 0, verifier passed at migration-only, activate → **gen 148**, c29 general, verifier passed (1 199 assignments carried). No `container die` fleet-wide 22:11Z–22:30Z. |
|
||||
| **Roll 1 complete** | Image census 22:30Z from MIG templates: c8–c10, c13–c16, c19–c29 on `519f4914` (18 cells); c7 on `85bf6799` (the earlier rehearsal image, carries the same fix); existing-only c1–c6, c11, c12 and migration-only c17, c18 untouched by design. No serving cell remains on `5aedbca5`. Selector gen 148, membership unchanged from the start of the roll. Zero relay container exits fleet-wide across the roll (01:14Z–22:30Z). Gates used: #19–#50; freezes were all monitor-side (provenance, freshness, flat Asia latency bar, one Cloud Monitoring collector failure), none a fleet health finding. Roll 2 (fresh image with #18722 + #18720) is the next data-plane step and waits on the owner's private-IP window decision. |
|
||||
@@ -0,0 +1,154 @@
|
||||
# Relay Roll 2 and close-out plan (2026-09-05)
|
||||
|
||||
Owner-approved scope 2026-09-05: finish the relay reliability work with one more cell image roll,
|
||||
deferring the Cloud SQL private-IP move (2.1, orca-cloud #477) to a separate owner decision. Roll 1
|
||||
is complete (see `relay-reconnect-2026-09-findings.md`, "Roll 1 complete"); every serving cell runs
|
||||
`519f4914` except c7 on `85bf6799`.
|
||||
|
||||
Estimate: about two working days of effort over one week of calendar time. The cell roll itself is
|
||||
6 to 7 hours of mostly unattended wall clock, run in the US night.
|
||||
|
||||
## Phase 0. Land the code (half a day, no production change)
|
||||
|
||||
### 0a. Split PR #18565
|
||||
|
||||
The branch mixes three relay/mobile/desktop fixes with the operator record. Split so the record
|
||||
lands regardless of how the code review goes.
|
||||
|
||||
- **Docs PR** (new branch off main): `relay-reconnect-2026-09-findings.md`,
|
||||
`relay-improvement-checklist-2026-09.md`, `relay-improvement-roadmap-2026-09.md`, this file.
|
||||
Docs only, merge on CI green.
|
||||
- **Code PR** (rebase #18565 onto main, resolve two conflicts):
|
||||
- `cloud/apps/relay/src/host-session-registry.ts`: conflict with #18698 (signed-out signal).
|
||||
Keep both; the accept-abandonment and lease changes are orthogonal to the signed-out path.
|
||||
- `src/main/runtime/relay/relay-origin-pool.ts`: **drop this branch's version**. #18719 already
|
||||
merged the desktop early-window jitter (1 to 6 min). Also drop
|
||||
`relay-session-broker.test.ts` additions that only exercise the dropped change.
|
||||
- Keep: relay accept abandonment (`orca_relay_client_accept_abandoned` event), relay-side lease
|
||||
jitter, mobile direct-probe fail-fast, and their tests.
|
||||
|
||||
### 0b. Lengthen the control lease (same code PR)
|
||||
|
||||
In `cloud/apps/relay/src/host-session-registry.ts`:
|
||||
|
||||
```
|
||||
CONTROL_LEASE_MS = 6 * 60 * 60 * 1000 // was 55 min
|
||||
CONTROL_LEASE_JITTER_MS = 30 * 60 * 1000 // was 5 min
|
||||
```
|
||||
|
||||
Why 6 h: the lease bounds how long a host stays on a cell after a missed drain and is the only
|
||||
passive rebalancing; 6 h keeps both and cuts control-activation traffic on the inventory lock by
|
||||
about 6x. Nothing else depends on it: the relay JWT (5 min) is refreshed by the desktop on its own
|
||||
schedule and liveness is the 75 s silence watchdog. Wire-safe: the relay sends `leaseExpiresAt` in
|
||||
the hello ack and old desktops schedule from that value.
|
||||
|
||||
Update the comment above the constants and the three assertions in
|
||||
`host-session-client-accept.test.ts` that pin the lease arithmetic. Check that nothing in
|
||||
`cloud/apps/relay-ops` or the monitor thresholds assumes a 55 min rotation period (grep
|
||||
`55`, `CONTROL_LEASE`, `rotation`).
|
||||
|
||||
### 0c. Review and merge
|
||||
|
||||
Review rounds per the standing process (Opus review, then Codex pass). Merge order: docs PR first
|
||||
(no dependency), then the code PR. Record the merge SHA of the code PR; that is the Roll 2 image
|
||||
source.
|
||||
|
||||
## Phase 1. Build and stage the image (half a day)
|
||||
|
||||
Roll 2 image = code PR merge SHA. It carries, relative to `519f4914`:
|
||||
|
||||
| Change | PR | Effect |
|
||||
|---|---|---|
|
||||
| Per-cell inventory locks, delta counters | #18722 | Removes the global `relay_cells FOR UPDATE` behind the phone accept hang |
|
||||
| Relay pool `statement_timeout` 5 s | #18722 | A relay query can no longer hang a cell |
|
||||
| Accept abandonment | #18565 | Cell stops finishing accepts for phones that already closed |
|
||||
| Control lease 6 h ± 30 min | #18565 | Fewer, spread-out rebinds |
|
||||
| `--private-ip` proxy flag support | #18720 | Code only; flag stays unset until 2.1 |
|
||||
|
||||
Steps, in order (from the findings doc's post-merge dispatch plan):
|
||||
|
||||
1. `gh workflow run cloud-publish-relay-production.yml --ref main -f mode=publish`. Resolve the
|
||||
digest by tag, not from the log:
|
||||
`gcloud artifacts docker images describe us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay:sha-<merge-sha> --format='value(image_summary.digest)'`.
|
||||
2. Staging: `cloud-deploy-relay-staging.yml` with the new digest; paired phone plus desktop smoke
|
||||
(connect, background, reconnect). Confirm `orca_relay_client_accept_abandoned` appears only when
|
||||
a client closes early, and that `sqlLatencyMsMax` no longer pins at the lock timeout.
|
||||
3. Director: `cloud-deploy-relay-production-director.yml -f image-digest=<new>
|
||||
-f regional-placement-mode=preserve -f prune-incompatible-revisions=false
|
||||
-f expected-rehome-generation=12 -f bootstrap-runtime-identity=false
|
||||
-f predecessor-image-digest=<serving digest>`. Blue/green; prior revision stays as rollback.
|
||||
Watch director `orca_relay_postgres_transaction_retry` per minute before and after. The director
|
||||
goes first so the per-cell locks are live before any cell restart burst.
|
||||
4. Same-cap `verify` mode against c7 with target=<new>, rollback=`519f4914`. Read-only.
|
||||
|
||||
Go/no-go for Phase 2: director serving the new image for at least 30 min, retries per minute at or
|
||||
below the pre-deploy baseline, no `container die`, no auth 5xx.
|
||||
|
||||
## Phase 2. Roll the cells (one US night, mostly unattended)
|
||||
|
||||
Same machinery as Roll 1: `cloud-monitor-relay-production.yml` dry-run gate, then
|
||||
`cloud-deploy-relay-production-same-cap.yml`. Cells roll one at a time by design (exact selector
|
||||
assertions, single Terraform state, and one cell's ~1.2k-host reconnect burst per restart). Do not
|
||||
add parallelism for this roll.
|
||||
|
||||
Inputs: target=<new digest>, rollback=`519f4914` (c7: rollback=`85bf6799`). Selector membership is
|
||||
unchanged from the end of Roll 1 (gen 148; existing-only c1–c6, c11, c12; migration-only c17, c18).
|
||||
|
||||
Order:
|
||||
|
||||
1. **c7 canary** (`canary-apply`, protocol 1). c7 is the rehearsal cell and the only one not on
|
||||
`519f4914`.
|
||||
2. **c8 canary**, then **batch c9, c10, c13, c14**.
|
||||
3. **c15 canary**, then **batch c16, c19, c20, c21**.
|
||||
4. **c22 canary**, then **batch c23, c24, c25, c26**.
|
||||
5. **Asia c27, c28, c29** as three single canaries at protocol 0 (`PROTO=0`). Batch mode cannot
|
||||
take Asia cells yet and needs at least two cells.
|
||||
|
||||
Each batch needs a same-commit canary authority; each wave needs a fresh 15 min gate. Use the
|
||||
chain script pattern from Roll 1 (wait gate green, check trusted-path ancestry, dispatch within 5 min,
|
||||
log `CANARY <run> <status>`) under `caffeinate -i`. Budget: 11 to 13 min per cell plus 15 min per gate,
|
||||
about 6 to 7 h total.
|
||||
|
||||
Per wave checks (same as Roll 1): transition verifier passes at migration-only and again at general
|
||||
with assignments carried; no `container die` fleet-wide; selector generation advances by exactly 2
|
||||
per cell. After the Asia cells: image census from MIG templates; every general cell on the new digest.
|
||||
|
||||
Failure handling: a failed canary re-enters through `mode=rollback` with rollback-digest = desired
|
||||
image (Roll 1 c27 pattern). A gate freeze on an Asia latency probe despite the 4 000 ms bar is a
|
||||
stop-and-investigate, not a retry. Monitor-side freezes (freshness, continuity deadline) re-gate
|
||||
after a 2 min back-off; the chain does this on its own.
|
||||
|
||||
Record every gate and wave in the findings doc as in Roll 1.
|
||||
|
||||
## Phase 3. After the roll (spread over the following week)
|
||||
|
||||
- **4.4 Recalibrate the retries bar.** After one week of `orca_relay_postgres_transaction_retry`
|
||||
on the new image, re-derive the `postgres_retries` monitor threshold from the new baseline
|
||||
(PR against `cloud/apps/relay-ops/src/incident-monitor.ts` thresholds). About 2 h.
|
||||
- **1.2 Pruner budget.** Raise `auth_token_pruner_max_rows_per_run` to the default 200k after a
|
||||
clean day; watch Cloud SQL write MB/s and the checkpoint alert. Then **1.5** log metric plus
|
||||
policy on `stopReason != complete`.
|
||||
- **1.3 Reclaim.** Once pruner runs delete ~0 rows: `pg_repack -t refresh_tokens` off-peak (check
|
||||
`pg_available_extensions` first; not `VACUUM FULL`). Confirm table, index, and `disk/utilization`
|
||||
dropped.
|
||||
- **Monitor residuals** already in the checklist: `probeEndpointHealth` retry decision still uses the
|
||||
flat 2 000 ms bar; operator protocol unbound for Asia; `probe-relay-rehome-trust` regex.
|
||||
- Update the checklist status header; tick 2.3, 4.1, 4.3 relay-side as deployed.
|
||||
|
||||
## Deferred, owner decision required
|
||||
|
||||
- **2.1 Private IP** (orca-cloud #477). One-way door with a Cloud SQL restart. When chosen: apply the
|
||||
foundation off-peak, then a template-only change that sets the `--private-ip` proxy flag. That is
|
||||
another cell roll unless bundled with a future image.
|
||||
- **5.2 Paging channel** for auth alerts: needs a destination.
|
||||
- **Parallel cell rolls** (2 or 3 at a time): about 1.5 days (relax exact-selector assertions to
|
||||
"exact except in-flight", single coordinator Terraform apply, parallel job shape, tests). Only
|
||||
worth building if more image rolls are planned after Roll 2, and only once the per-cell locks are
|
||||
live so a multi-cell reconnect burst is safe.
|
||||
- **2.2 Database split**: deferred to ~2026-11-01.
|
||||
|
||||
## Not in this plan
|
||||
|
||||
Desktop and mobile changes already merged (#18719 desktop early-window jitter and no same-token
|
||||
refresh retry; #18565 mobile fail-fast once merged) ship with the next desktop and mobile releases
|
||||
on their own schedules. No relay action needed.
|
||||
@@ -10,6 +10,102 @@
|
||||
}
|
||||
},
|
||||
"gates": [
|
||||
{
|
||||
"id": "terminal-output.prestarted-shell-snapshot-adoption",
|
||||
"title": "Prestarted shell adoption paints covered output once",
|
||||
"maturity": "experimental",
|
||||
"protection": "partial",
|
||||
"owner": "terminal-runtime",
|
||||
"layer": "renderer-transport-and-live-electron",
|
||||
"surfaces": [
|
||||
"backend-created first terminal",
|
||||
"daemon snapshot adoption",
|
||||
"deferred live output"
|
||||
],
|
||||
"platforms": ["macos", "linux", "windows"],
|
||||
"providers": ["local", "daemon", "wsl", "ssh", "remote-runtime"],
|
||||
"coveredPlatforms": ["macos", "linux", "windows"],
|
||||
"coveredProviders": ["local", "daemon", "wsl"],
|
||||
"coverageNotes": "macOS daemon-backed Electron journey verifies same PID and terminal identity plus rendered output. Focused renderer contracts pass on Linux, Windows and WSL. Neighboring SSH model and replay contracts pass locally; no new live SSH or paired-runtime journey.",
|
||||
"motivatingLinks": [
|
||||
"https://github.com/user-attachments/assets/e8c6d1dc-6150-4c3d-b55a-3d12efefdd04",
|
||||
"https://github.com/user-attachments/assets/b0328f88-34ac-4d51-8119-9efe17072435"
|
||||
],
|
||||
"invariant": "Adopting a prestarted terminal preserves its existing process and paints snapshot-covered startup output once while retaining subsequent live output. Missing sequence proof or blank snapshots must not authorize dropping output.",
|
||||
"oracle": "Pass snapshot sequence and proven zero keyboard flags through real IPC transport projection. Deliver snapshot-covered and newer output before reattach resolves; drain replay parse callbacks and require one startup marker and the newer output. Repeat with no sequence and blank snapshot to retain unproven bytes. In Electron select a prestarted workspace, type a generated marker and compare PID and stable terminal identities before and after.",
|
||||
"commands": [
|
||||
"pnpm test src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts",
|
||||
"pnpm test src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts src/renderer/src/components/terminal-pane/pty-connection-hidden-snapshot-live-overlap.test.ts src/renderer/src/components/terminal-pane/pty-connection-replay-payload-handling.test.ts src/renderer/src/components/terminal-pane/pty-connection/reattach-payload-ssh-reconnect-model-paint.test.ts",
|
||||
"pnpm test src/renderer/src/components/terminal-pane/pty-connection src/renderer/src/components/terminal-pane/pty-transport"
|
||||
],
|
||||
"testFiles": [
|
||||
"src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts",
|
||||
"src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts"
|
||||
],
|
||||
"assertionRefs": [
|
||||
{
|
||||
"file": "src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts",
|
||||
"assertions": [
|
||||
"zero and nonzero snapshot sequence and proven zero keyboard flags survive IPC projection"
|
||||
]
|
||||
},
|
||||
{
|
||||
"file": "src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts",
|
||||
"assertions": [
|
||||
"startup output covered by the snapshot is painted once",
|
||||
"new output remains visible",
|
||||
"legacy unsequenced and blank snapshots retain bytes"
|
||||
]
|
||||
}
|
||||
],
|
||||
"evidenceRuns": [
|
||||
{
|
||||
"date": "2026-09-04",
|
||||
"runner": "local",
|
||||
"platform": "macos",
|
||||
"command": "pnpm test src/renderer/src/components/terminal-pane/pty-transport-connect-spawn.test.ts src/renderer/src/components/terminal-pane/pty-connection-deferred-reattach-live-output.test.ts src/renderer/src/components/terminal-pane/pty-connection-hidden-snapshot-live-overlap.test.ts src/renderer/src/components/terminal-pane/pty-connection-replay-payload-handling.test.ts src/renderer/src/components/terminal-pane/pty-connection/reattach-payload-ssh-reconnect-model-paint.test.ts",
|
||||
"result": "passed",
|
||||
"durationSeconds": 2.34,
|
||||
"summary": "5 suites / 48 tests pass. Focused 2-suite runs independently pass 27 tests on Linux, Windows and WSL."
|
||||
},
|
||||
{
|
||||
"date": "2026-09-04",
|
||||
"runner": "local",
|
||||
"platform": "macos",
|
||||
"command": "pnpm test src/renderer/src/components/terminal-pane/pty-connection src/renderer/src/components/terminal-pane/pty-transport",
|
||||
"result": "passed",
|
||||
"durationSeconds": 5.67,
|
||||
"summary": "Broader connection/transport gate: 77 files and 813 tests passed, including neighboring restore, reconnect, input and replay behavior. Log: artifacts/worktree-create/orca-draft-replay-broader-gate.log."
|
||||
}
|
||||
],
|
||||
"runtimeBudget": {
|
||||
"p95Seconds": 15,
|
||||
"scope": "focused renderer transport and deferred-adoption contracts"
|
||||
},
|
||||
"flakeHistory": {
|
||||
"status": "unknown",
|
||||
"evidence": "Focused local and remote runs pass; no CI soak history."
|
||||
},
|
||||
"redGreenEvidence": {
|
||||
"status": "partial",
|
||||
"evidence": "Metadata tests fail before forwarding. Corrected parse-draining regression observes two startup markers when the baseline installation is removed, and one after restoration. Initial missing-live-output failure was a harness parse-drain omission and is not red proof. Before/fixed Electron screenshots show duplicate/single startup output."
|
||||
},
|
||||
"performanceBudget": {
|
||||
"required": true,
|
||||
"evidence": "Reuses existing snapshot baseline reconciliation with no new scan, timer or subprocess. Corrected daemon-backed rendered trial reaches replay at 116.6 ms and generated keyboard output at 177 ms after selecting the prestarted workspace. This measures selection/adoption, not ordinary composer creation."
|
||||
},
|
||||
"promotionCriteria": [
|
||||
"Meet manifest CI and soak policy.",
|
||||
"Retain intentional-break and rendered identity/output proof.",
|
||||
"Exercise live SSH and paired-runtime snapshot adoption before claiming full provider coverage."
|
||||
],
|
||||
"knownGaps": [
|
||||
"Composer draft creation and cancellation are not implemented by this gate.",
|
||||
"No new live SSH, Windows or WSL UI run; remote evidence is focused contract tests.",
|
||||
"Mixed-version snapshots without sequence proof intentionally retain legacy behavior."
|
||||
],
|
||||
"demotionRule": "Keep experimental or demote if adoption duplicates covered output, drops newer or unproven output, changes terminal ownership, or flakes without explanation."
|
||||
},
|
||||
{
|
||||
"id": "cmd-j-tabs.host-qualified-candidate-ownership",
|
||||
"title": "Cmd-J tab candidates retain execution-host ownership",
|
||||
|
||||
@@ -414,10 +414,11 @@ describe('PR Checks skip wiring', () => {
|
||||
})
|
||||
|
||||
it('skips e2e detection on docs-only PRs without dropping the draft gate', () => {
|
||||
expect(prWorkflow.jobs['e2e-paths'].needs).toEqual(['code_paths'])
|
||||
expect(prWorkflow.jobs['e2e-paths'].if).toBe(
|
||||
"github.event.pull_request.draft != true && needs.code_paths.outputs.should_run == 'true'"
|
||||
const filter = prWorkflow.jobs.code_paths.steps.find((step) => step.id === 'e2e_filter')
|
||||
expect(filter.if).toBe(
|
||||
"github.event.pull_request.draft != true && steps.filter.outputs.should_run == 'true'"
|
||||
)
|
||||
expect(prWorkflow.jobs['e2e-paths']).toBeUndefined()
|
||||
})
|
||||
|
||||
it('lets verify pass skipped jobs the classifier turned off', () => {
|
||||
|
||||
@@ -39,7 +39,7 @@ const nativeImeSpec = readFileSync(
|
||||
'utf8'
|
||||
)
|
||||
|
||||
const filterStep = prWorkflow.jobs['e2e-paths'].steps.find(
|
||||
const filterStep = prWorkflow.jobs.code_paths.steps.find(
|
||||
(step) => step.name === 'Filter changed E2E specs'
|
||||
)
|
||||
const rollbackStep = prWorkflow.jobs.static_analysis.steps.find(
|
||||
@@ -106,16 +106,16 @@ describe('PR E2E gate contract', () => {
|
||||
// Why: without this the job could lose its filter and run on every PR — the
|
||||
// cost the path filter exists to avoid — while the gate assertions above
|
||||
// stay green.
|
||||
expect(prWorkflow.jobs.e2e.needs).toBe('e2e-paths')
|
||||
expect(prWorkflow.jobs.e2e.if).toBe("needs.e2e-paths.outputs.should_run == 'true'")
|
||||
expect(prWorkflow.jobs['e2e-paths'].outputs.should_run).toBe(
|
||||
'${{ steps.filter.outputs.should_run }}'
|
||||
expect(prWorkflow.jobs.e2e.needs).toBe('code_paths')
|
||||
expect(prWorkflow.jobs.e2e.if).toBe("needs.code_paths.outputs.e2e_should_run == 'true'")
|
||||
expect(prWorkflow.jobs.code_paths.outputs.e2e_should_run).toBe(
|
||||
'${{ steps.e2e_filter.outputs.should_run }}'
|
||||
)
|
||||
expect(prWorkflow.jobs['e2e-paths'].outputs.test_files).toBe(
|
||||
'${{ steps.filter.outputs.test_files }}'
|
||||
expect(prWorkflow.jobs.code_paths.outputs.test_files).toBe(
|
||||
'${{ steps.e2e_filter.outputs.test_files }}'
|
||||
)
|
||||
expect(prWorkflow.jobs.e2e.with.ref).toBe('${{ github.event.pull_request.head.sha }}')
|
||||
expect(prWorkflow.jobs.e2e.with.test_files).toBe('${{ needs.e2e-paths.outputs.test_files }}')
|
||||
expect(prWorkflow.jobs.e2e.with.test_files).toBe('${{ needs.code_paths.outputs.test_files }}')
|
||||
})
|
||||
|
||||
it('enforces every job verify depends on', () => {
|
||||
@@ -360,11 +360,11 @@ describe('PR E2E gate contract', () => {
|
||||
expect(sshLaneCondition).toContain("inputs.ssh_source_changed == 'true' ||")
|
||||
|
||||
expect(e2eWorkflow.on.workflow_call.inputs.ssh_source_changed.type).toBe('string')
|
||||
expect(prWorkflow.jobs['e2e-paths'].outputs.ssh_source_changed).toBe(
|
||||
'${{ steps.filter.outputs.ssh_source_changed }}'
|
||||
expect(prWorkflow.jobs.code_paths.outputs.ssh_source_changed).toBe(
|
||||
'${{ steps.e2e_filter.outputs.ssh_source_changed }}'
|
||||
)
|
||||
expect(prWorkflow.jobs.e2e.with.ssh_source_changed).toBe(
|
||||
'${{ needs.e2e-paths.outputs.ssh_source_changed }}'
|
||||
'${{ needs.code_paths.outputs.ssh_source_changed }}'
|
||||
)
|
||||
expect(filterStep.run).toContain('pr-e2e-source-routing.mjs --ssh-source')
|
||||
expect(filterStep.run).toContain('ssh_source_changed=$SSH_SOURCE_CHANGED')
|
||||
@@ -565,12 +565,12 @@ describe('PR E2E gate contract', () => {
|
||||
expect(prWorkflow.jobs.terminal_ime_native.uses).toBe(
|
||||
'./.github/workflows/terminal-ime-e2e.yml'
|
||||
)
|
||||
expect(prWorkflow.jobs.terminal_ime_native.needs).toBe('e2e-paths')
|
||||
expect(prWorkflow.jobs.terminal_ime_native.needs).toBe('code_paths')
|
||||
expect(prWorkflow.jobs.terminal_ime_native.if).toBe(
|
||||
"needs.e2e-paths.outputs.native_ime_source_changed == 'true'"
|
||||
"needs.code_paths.outputs.native_ime_source_changed == 'true'"
|
||||
)
|
||||
expect(prWorkflow.jobs['e2e-paths'].outputs.native_ime_source_changed).toBe(
|
||||
'${{ steps.filter.outputs.native_ime_source_changed }}'
|
||||
expect(prWorkflow.jobs.code_paths.outputs.native_ime_source_changed).toBe(
|
||||
'${{ steps.e2e_filter.outputs.native_ime_source_changed }}'
|
||||
)
|
||||
expect(filterStep.run).toContain('pr-e2e-source-routing.mjs --native-ime-source')
|
||||
expect(filterStep.run).toContain('native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED')
|
||||
|
||||
@@ -616,7 +616,7 @@ if (!isHelpOrVersion && process.env.ORCA_DEV_INSTANCE_LABEL) {
|
||||
// Why: automation launches this app while someone is working; announce that the
|
||||
// window will come up without taking the foreground so the mode is visible in logs.
|
||||
if (!isHelpOrVersion && process.env.ORCA_BACKGROUND_LAUNCH === '1') {
|
||||
console.error('[orca-dev] Background launch: window shows without stealing focus')
|
||||
console.error('[orca-dev] Background launch: window stays off screen; automate through CDP')
|
||||
}
|
||||
let forwardedExtras = []
|
||||
if (!userPassedPort && !isHelpOrVersion) {
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
# CI efficiency and runner capacity
|
||||
|
||||
Audit date: September 5, 2026. No paid capacity or provider configuration changed.
|
||||
|
||||
## Measurements and changes
|
||||
|
||||
Three recent successful PR runs used 54.6–64.9 aggregate runner minutes:
|
||||
[33998366568](https://github.com/stablyai/orca/actions/runs/33998366568),
|
||||
[33998220287](https://github.com/stablyai/orca/actions/runs/33998220287), and
|
||||
[33998181502](https://github.com/stablyai/orca/actions/runs/33998181502).
|
||||
These are sums of active job durations, excluding skipped jobs; they are not
|
||||
billing minutes or queue time. This small sample is not a historical average.
|
||||
|
||||
- Consolidate E2E routing into the existing code-path detector. The removed
|
||||
detector occupied 20–22 seconds and required another runner allocation and
|
||||
full-history checkout per nondraft code PR. The same routing commands remain,
|
||||
including SSH and native IME selection; actual E2E results remain advisory.
|
||||
A routing-script error now fails the required code-path detector.
|
||||
- Use gzip for PR-only Debian/RPM artifacts. The two sampled Linux packaging
|
||||
jobs took 8m10s and 8m19s overall; one spent 3m47s in electron-builder. Its
|
||||
default Debian/RPM compression is xz. PR artifacts are inspected on the same
|
||||
runner, so their download size offers no benefit. Keep all AppImage, Debian,
|
||||
RPM, payload, launcher, and shutdown checks. Release compression is unchanged.
|
||||
Compression savings need a hosted run; do not equate the full packaging step
|
||||
with removable compression time.
|
||||
- Cancel superseded Mobile Checks and Skill update round-trip PR runs. The
|
||||
skill matrix has 13 jobs. Preserve non-cancelling main/merge-group skill runs,
|
||||
with separate concurrency groups per event.
|
||||
- Reuse the existing script-free root dependency action in Mobile Checks,
|
||||
including the pnpm cache keyed by both root and mobile lockfiles. The root
|
||||
install remains necessary because mobile types import root dependencies.
|
||||
|
||||
The repository already has eight unit shards, path-scoped platform checks,
|
||||
native caches, one shared E2E build, PR cancellation, incremental TypeScript
|
||||
caching, and changed-spec E2E routing. Increasing shards would increase setup
|
||||
work and simultaneous runner demand. Do not adjust the count without comparing
|
||||
critical-path time and aggregate job time on the same commit.
|
||||
|
||||
## Runner recommendations
|
||||
|
||||
The repository is **public**, verified using the GitHub API. Standard
|
||||
GitHub-hosted Linux, Windows, and macOS runners have free compute minutes for
|
||||
public repositories. Queue pressure and third-party provider allowances still
|
||||
matter; artifact storage and larger runners have separate billing rules.
|
||||
See [GitHub Actions billing](https://docs.github.com/en/billing/concepts/product-billing/github-actions).
|
||||
|
||||
1. Keep standard GitHub-hosted runners as the default. Ask GitHub Support for a
|
||||
higher concurrent-job limit before paying for more capacity. The documented
|
||||
standard limits depend on the account plan (Free: 20 total/5 macOS; Team:
|
||||
60/5; Enterprise: 500/50), and increases are subject to approval. The actual
|
||||
account entitlement was not verified. See [limits](https://docs.github.com/en/actions/reference/limits).
|
||||
2. Reserve existing Blacksmith allowance for macOS if that is the priority.
|
||||
Blacksmith documents 3,000 free x64 2-vCPU-equivalent minutes per organization;
|
||||
a 6-vCPU Mac minute consumes 20 equivalents, or 150 actual Mac minutes if
|
||||
it uses the entire free pool. Cloud workflows also use Blacksmith Linux.
|
||||
Moving Linux to hosted GitHub saves shared allowance, but does not necessarily
|
||||
free Mac hardware capacity. Account-specific contracts and usage were not
|
||||
inspected. See [Blacksmith runners](https://docs.blacksmith.sh/blacksmith-runners/overview).
|
||||
3. Treat Ubicloud as an optional small Linux overflow trial. Its documented
|
||||
$2.50 monthly credit buys 1,250 premium 2-vCPU minutes at $0.002/minute, or
|
||||
2,000 standard 2-vCPU minutes at $0.00125/minute. New accounts default to
|
||||
premium and require a credit card. No enforceable hard spending cap was
|
||||
verified, so changing runner labels cannot guarantee the no-spend constraint.
|
||||
One PR's roughly 55–65 runner minutes also makes clear how small this pool
|
||||
is relative to repository activity (hardware speeds differ).
|
||||
See [pricing](https://ubicloud.com/docs/about/pricing) and
|
||||
[setup](https://ubicloud.com/docs/github-actions-integration/quickstart).
|
||||
|
||||
## Machines that also run coding agents
|
||||
|
||||
Do not register the credentialed host directly as a public-PR runner. A PR can
|
||||
execute arbitrary build/test code, and a persistent host lets it access local
|
||||
credentials or affect subsequent jobs. Docker alone is not adequate isolation
|
||||
when it exposes the host home, Docker socket, SSH agent, or office network.
|
||||
|
||||
A possible no-new-hardware experiment is a disposable VM per job, preferably on
|
||||
a dedicated spare machine, with a just-in-time single-job runner, no shared
|
||||
home/keychain/SSH agent or host mounts, restricted network access, and CPU/RAM
|
||||
limits that leave room for coding agents. Destroy the VM after every job;
|
||||
ephemeral runner registration by itself does not clean the machine. Start with
|
||||
trusted branch/manual workloads and keep public fork PRs on hosted runners.
|
||||
Provisioning and ongoing patching are real operational costs even when the
|
||||
machine is already owned. See GitHub's
|
||||
[self-hosted runner security guidance](https://docs.github.com/en/actions/security-for-github-actions/security-guides/security-hardening-for-github-actions).
|
||||
|
||||
## Release waits
|
||||
|
||||
The latest successful sampled Windows release used 13m59s of a 21m56s job in
|
||||
signing wait/download steps. The same release held an Ubuntu job for 11m38s
|
||||
polling the isolated Mac build. These are stronger occupancy opportunities than
|
||||
small checkout savings, especially when approval takes hours.
|
||||
|
||||
[Windows signing without occupying a runner](windows-signing-runner-time.md)
|
||||
describes a staged, same-run design, required protected environments, and
|
||||
rehearsal criteria. No callback integration or protected Windows signing
|
||||
environments currently exist. An environment-gated design adds a GitHub
|
||||
approval after each SignPath approval and changes the current automatic inner
|
||||
signing timeout fallback; those are explicit release-policy decisions, so this
|
||||
PR leaves production signing behavior unchanged.
|
||||
@@ -0,0 +1,137 @@
|
||||
# Windows signing without occupying a runner during approval
|
||||
|
||||
Status: implementation proposal; production signing behavior is unchanged.
|
||||
|
||||
## Measured cost
|
||||
|
||||
In [release run 33821033674](https://github.com/stablyai/orca/actions/runs/33821033674)
|
||||
(September 4, 2026), the Windows job took 21m56s. The inner-binary download step
|
||||
took 13m19s and the installer download step took 40s: 13m59s, or 64% of the job,
|
||||
was spent in the signing download/wait steps. These durations include the
|
||||
download itself, so they are an upper bound on removable idle time, not a
|
||||
prediction of net savings after transferring state between jobs.
|
||||
|
||||
`release-cut.yml` submits both requests with `wait-for-completion: false`, but
|
||||
then invokes `Get-SignedArtifact` on the same Windows runner with one-hour and
|
||||
four-hour completion timeouts. The six-hour job timeout accommodates both
|
||||
waits. Changing the submission flag again, polling less often, or running the
|
||||
wait inside a container does not release the runner slot.
|
||||
|
||||
This is runner occupancy, not a billing estimate. Standard GitHub-hosted
|
||||
runners in a public repository may be free; removing the waits still releases
|
||||
concurrency for other work. Check actual billing before assigning dollar savings.
|
||||
|
||||
The same release also occupied an Ubuntu runner for 11m38s while
|
||||
`run-release-mac-build-workflow.mjs` waited on the isolated macOS workflow.
|
||||
That is a separate orchestration optimization. Windows development-channel
|
||||
builds deliberately ship unsigned and have no SignPath wait to remove.
|
||||
|
||||
## Proposed execution graph
|
||||
|
||||
Keep all Windows stages in the original `release-cut.yml` run to preserve the
|
||||
current SignPath GitHub artifact provenance boundary:
|
||||
|
||||
1. `build-windows` builds and uploads the unpacked app, original installer,
|
||||
updater metadata, and inner-signing manifest. It submits the inner request,
|
||||
sends the existing notification, exposes the request ID, and finishes.
|
||||
2. `package-windows` depends on that job and uses a protected environment named
|
||||
`windows-inner-signing`. Its runner is allocated only after GitHub approval.
|
||||
It restores the exact build, downloads the signed binaries with a short,
|
||||
bounded completion wait, applies the existing signature restoration and
|
||||
signed `elevate.exe` cache replacement, builds the NSIS installer, uploads it,
|
||||
submits the second signing request, notifies approvers, and finishes.
|
||||
3. `finalize-windows` depends on packaging and uses a second protected environment
|
||||
named `windows-installer-signing`. After approval it downloads the signed
|
||||
installer, regenerates its blockmap and `latest.yml`, runs existing outer and
|
||||
inner signature checks, uploads evidence, and uploads the assets to the draft.
|
||||
4. `publish-release` depends on finalization as well as the existing Linux, macOS,
|
||||
and blocking release gates. It remains the only job that publishes the draft.
|
||||
|
||||
The approver signs in SignPath, waits for that request to finish, and then
|
||||
approves the corresponding pending GitHub job. Each notification should link
|
||||
to both places and explain the order. GitHub approval is an extra action;
|
||||
approving in SignPath alone does not release an environment gate.
|
||||
|
||||
## Required configuration
|
||||
|
||||
The repository environments were inspected through the GitHub API on
|
||||
September 5, 2026. Neither Windows environment exists. `adhoc-mac-build` has no
|
||||
protection rules; it cannot be reused as an approval gate. No SignPath callback
|
||||
handler was found in the repository's workflows, scripts, application, or cloud
|
||||
code.
|
||||
|
||||
Before enabling the graph:
|
||||
|
||||
1. Create both environments in repository Settings → Environments.
|
||||
2. Add the release approvers as required reviewers for each environment. Decide
|
||||
whether a release initiator may approve their own job, and configure that
|
||||
consistently with the existing SignPath policy.
|
||||
3. Restrict deployment branches to the trusted refs used to dispatch release
|
||||
workflows, and check that the release workflow's ref passes the restriction.
|
||||
The workflow ref and the checked-out release tag are different concepts.
|
||||
4. Read back both environments through the API and verify that
|
||||
`required_reviewers` rules exist before changing the release graph. Merely
|
||||
referring to a new environment name in YAML can create an unprotected
|
||||
environment and silently leave the wait on the runner.
|
||||
5. Add a preflight assertion for those rules so accidental removal fails before
|
||||
any signing request is submitted. Verify the API access required for this
|
||||
assertion using the release workflow's token; do not assume an administrator's
|
||||
local `gh` access proves workflow-token access.
|
||||
|
||||
An automatic alternative requires a SignPath completion callback and an
|
||||
authenticated integration that releases the corresponding deployment gate.
|
||||
Confirm the Foundation plan supports the necessary callback before choosing
|
||||
that architecture. Do not introduce a long-running GitHub polling job as the
|
||||
callback substitute: it would continue occupying a slot.
|
||||
|
||||
## State and failure contracts
|
||||
|
||||
- Use artifacts from this exact run and attempt, with a manifest containing the
|
||||
tag, tag commit SHA, workflow SHA, request IDs, artifact IDs, and SHA-256 hashes.
|
||||
Artifact names alone are insufficient. Preserve the original unsigned
|
||||
installer for the existing inner-signing fallback.
|
||||
- Restore `dist/win-unpacked`, the staging list, the installer, and updater
|
||||
metadata as one checkpoint. Use an archive to preserve the tree. Do not ship
|
||||
a fresh rebuild of the app after approving a different binary tree.
|
||||
- Each new Windows runner needs the pinned Node/pnpm toolchain, build
|
||||
dependencies, SignPath module, and electron-builder tool cache. The second
|
||||
runner must populate the NSIS cache before replacing `elevate.exe`; the old
|
||||
code assumes the first installer build already populated that cache.
|
||||
- Retain checkout-from-tag behavior and the existing support for release tags
|
||||
that predate the composite action. Explicitly restore new orchestration code
|
||||
from the workflow SHA when necessary.
|
||||
- Preserve the rule that rerunning a workflow never submits a new signing
|
||||
request. A resume must consume the recorded request and artifacts. Test failed
|
||||
stage reruns, whole-workflow reruns, and missing/expired checkpoints separately.
|
||||
- Keep installer signature checks blocking. Keep inner verification evidence
|
||||
and its current warning-only policy unless changed in a separate decision.
|
||||
- Resolve the current one-hour inner-signing fallback deliberately: an
|
||||
environment approval can remain pending longer than one hour and rejection
|
||||
skips dependent jobs. It cannot reproduce the existing automatic timeout
|
||||
fallback by itself. A first migration should explicitly document the new
|
||||
manual release/cancellation behavior; silently treating rejected approval as
|
||||
permission to ship is not acceptable.
|
||||
- Keep the release-wide concurrency lock while the graph waits, preventing
|
||||
another release from overtaking this draft. This saves worker occupancy, but
|
||||
does not shorten the serialized release queue's human approval time.
|
||||
|
||||
## Validation before production
|
||||
|
||||
First adapt `windows-signing-rehearsal.yml` to exercise the same staged code
|
||||
using the auto-approved test-signing policy. Then run a manual rehearsal with
|
||||
the protected environments and confirm that pending approval has no allocated
|
||||
Windows runner. Verify signed bytes through the existing extraction-based
|
||||
installer checks, not only the outer installer signature.
|
||||
|
||||
Cover approval before SignPath completion, rejected approval, missing signed
|
||||
files, changed checkpoint hashes, lost checkpoints, expired artifacts, failed
|
||||
packaging, and stage reruns without duplicate submissions. Confirm no release
|
||||
becomes public until all platform and signature gates pass. Compare transferred
|
||||
artifact/setup time with the original 13m59s wait sample to measure net savings.
|
||||
|
||||
A separate `workflow_dispatch` continuation can avoid environment provisioning,
|
||||
but changes this design substantially: the original release run finishes,
|
||||
workflow-level concurrency no longer protects the pending draft, and SignPath
|
||||
must accept artifacts assembled from a prior run. That option needs a durable
|
||||
release state machine and provenance validation before production use; it is
|
||||
not a drop-in replacement for the two download steps.
|
||||
@@ -253,6 +253,9 @@ export const electronViteConfig: UserConfig = {
|
||||
'agent-hooks/managed-agent-hook-controls': resolve(
|
||||
'src/main/agent-hooks/managed-agent-hook-controls.ts'
|
||||
),
|
||||
'codex/managed-home-shell-preflight': resolve(
|
||||
'src/main/codex/managed-home-shell-preflight.ts'
|
||||
),
|
||||
// Why: account import mutates the user's macOS Keychain from the CLI.
|
||||
'claude-accounts/keychain': resolve('src/main/claude-accounts/keychain.ts')
|
||||
},
|
||||
|
||||
@@ -93,6 +93,20 @@ When a worker's terminal accepted input but the submit is unconfirmed, use
|
||||
`terminal send --wait-submit <seconds>`: it observes the accepted prompt for that
|
||||
long and, on timeout, returns the input-accepted receipt without resending.
|
||||
|
||||
## Refused starts
|
||||
|
||||
`dispatch` and `worker-start` refuse the following preflight cases with a stable
|
||||
`error.code`; read it before choosing a recovery, and treat `error.data.nextSteps`
|
||||
as the exact recovery text. Older hosts may omit `data`, so treat every field as
|
||||
optional.
|
||||
|
||||
| Code | Meaning | Recovery |
|
||||
| -------------------- | --------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------ |
|
||||
| `task_not_found` | No Task with that id, or not in the bound Run (`data.taskId`, `data.runId`) | Check `task-list --json`; create the Task with `task-create` if it does not exist |
|
||||
| `task_not_startable` | Task cannot start now: not `ready`, or invalid `--retry-of` (`data.status`, `data.unmetDependencies`, `data.retryOf`) | Wait for running dependencies with `check --wait`; retry or unblock failed ones; inspect `dispatch-show` if already dispatched |
|
||||
| `inject_rejected` | `--inject` refused because no recognized agent runs in the target (`data.terminal`, `data.reason`) | Start a recognized agent there or pick another terminal; or dispatch without `--inject` and use `terminal send` |
|
||||
| `runtime_error` | Any other failure, including a target terminal that already owns an active Dispatch | Read the message, inspect state, and do not retry unchanged |
|
||||
|
||||
## Retry, stop, and abandon
|
||||
|
||||
Retry only a positively proven failed or stopped attempt. Placement is never
|
||||
|
||||
File diff suppressed because one or more lines are too long
@@ -56,7 +56,10 @@ vi.mock('../runtime-client', () => {
|
||||
|
||||
vi.mock('../../main/agent-hooks/managed-agent-hook-controls', () => ({
|
||||
applyAgentStatusHooksEnabled: applyAgentStatusHooksEnabledMock,
|
||||
getManagedAgentHookStatuses: getManagedAgentHookStatusesMock,
|
||||
getManagedAgentHookStatuses: getManagedAgentHookStatusesMock
|
||||
}))
|
||||
|
||||
vi.mock('../../main/codex/managed-home-shell-preflight', () => ({
|
||||
prepareManagedCodexHomeBeforeShellLaunch: prepareManagedCodexHomeBeforeShellLaunchMock
|
||||
}))
|
||||
|
||||
|
||||
@@ -15,11 +15,7 @@ import { getDefaultPersistedState } from '../../shared/constants'
|
||||
import { normalizeDisabledTuiAgents } from '../../shared/tui-agent-selection'
|
||||
import type { GlobalSettings } from '../../shared/global-settings-types'
|
||||
import type { PersistedState } from '../../shared/persisted-state-types'
|
||||
import {
|
||||
applyAgentStatusHooksEnabled,
|
||||
getManagedAgentHookStatuses,
|
||||
prepareManagedCodexHomeBeforeShellLaunch
|
||||
} from '../../main/agent-hooks/managed-agent-hook-controls'
|
||||
import { prepareManagedCodexHomeBeforeShellLaunch } from '../../main/codex/managed-home-shell-preflight'
|
||||
|
||||
type AgentHookCommandResult = {
|
||||
enabled: boolean
|
||||
@@ -194,6 +190,8 @@ async function setAgentHooksEnabled(
|
||||
client: RuntimeClient,
|
||||
enabled: boolean
|
||||
): Promise<AgentHookCommandResult> {
|
||||
const { applyAgentStatusHooksEnabled, getManagedAgentHookStatuses } =
|
||||
await import('../../main/agent-hooks/managed-agent-hook-controls.js')
|
||||
const updatedRuntime = await updateRunningRuntime(client, enabled)
|
||||
const offlineUpdate = updatedRuntime ? null : updateEnabledOnDisk(enabled)
|
||||
const settingsPath = offlineUpdate?.settingsPath ?? getDataPath()
|
||||
@@ -234,6 +232,8 @@ export const AGENT_HOOK_HANDLERS: Record<string, CommandHandler> = {
|
||||
})
|
||||
},
|
||||
'agent hooks status': async ({ json }) => {
|
||||
const { getManagedAgentHookStatuses } =
|
||||
await import('../../main/agent-hooks/managed-agent-hook-controls.js')
|
||||
const result: AgentHookCommandResult = {
|
||||
enabled: readHookSettingsFromDisk().agentStatusHooksEnabled,
|
||||
settingsPath: getDataPath(),
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import {
|
||||
injectRejectedRefusal,
|
||||
taskNotFoundRefusal,
|
||||
taskNotStartableRefusal,
|
||||
type DispatchRefusalReceipt
|
||||
} from '../shared/orchestration-dispatch-refusal-contract'
|
||||
import { formatCliError, reportCliError } from './format'
|
||||
import { RuntimeRpcFailureError, type RuntimeRpcFailure } from './runtime/types'
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks()
|
||||
})
|
||||
|
||||
// Why: these are the exact envelopes the RPC dispatcher test proved the runtime emits. This
|
||||
// checkout's formatter never enumerates codes (verified below with a code no build has defined),
|
||||
// which is what lets a client that predates a new code still print its message and nextSteps.
|
||||
describe('orchestration dispatch refusals through the CLI error boundary', () => {
|
||||
it.each([
|
||||
{
|
||||
receipt: taskNotFoundRefusal('Task not found: task_missing', { taskId: 'task_missing' }),
|
||||
recovery: /task-create|task-list/
|
||||
},
|
||||
{
|
||||
receipt: taskNotStartableRefusal(
|
||||
'Task task_child is pending; only ready tasks can be dispatched',
|
||||
{ taskId: 'task_child', status: 'pending', unmetDependencies: ['task_parent'] }
|
||||
),
|
||||
recovery: /task_parent/
|
||||
},
|
||||
{
|
||||
receipt: injectRejectedRefusal('term_worker', 'no_agent_detected'),
|
||||
recovery: /without --inject/
|
||||
}
|
||||
])('prints $receipt.code with its recovery in human and JSON output', ({ receipt, recovery }) => {
|
||||
const failure = envelope(receipt)
|
||||
const error = new RuntimeRpcFailureError(failure)
|
||||
expect(error.code).toBe(receipt.code)
|
||||
|
||||
const human = formatCliError(error, { commandPath: ['orchestration', 'dispatch'] })
|
||||
expect(human).toContain(receipt.message)
|
||||
expect(human).toMatch(recovery)
|
||||
|
||||
const log = vi.spyOn(console, 'log').mockImplementation(() => {})
|
||||
reportCliError(error, true, { commandPath: ['orchestration', 'dispatch'] })
|
||||
const printed = JSON.parse(log.mock.calls[0]?.[0] as string) as RuntimeRpcFailure
|
||||
expect(printed.ok).toBe(false)
|
||||
expect(printed.error).toEqual(receipt)
|
||||
})
|
||||
})
|
||||
|
||||
// Why: a code this build has never defined stands in for a future host's new code; if the
|
||||
// formatter ever starts gating on known codes, this is the assertion that catches it.
|
||||
it('prints an unknown code with its message and nextSteps unchanged', () => {
|
||||
const failure: RuntimeRpcFailure = {
|
||||
id: 'rpc_1',
|
||||
ok: false,
|
||||
error: {
|
||||
code: 'code_from_a_newer_host',
|
||||
message: 'Refused for a reason this CLI has never heard of.',
|
||||
data: { nextSteps: ['Do the thing the newer host suggested.'] }
|
||||
},
|
||||
_meta: { runtimeId: 'runtime_1' }
|
||||
}
|
||||
const error = new RuntimeRpcFailureError(failure)
|
||||
|
||||
expect(formatCliError(error, { commandPath: ['orchestration', 'dispatch'] })).toBe(
|
||||
'Refused for a reason this CLI has never heard of.\nNext step: Do the thing the newer host suggested.'
|
||||
)
|
||||
const log = vi.spyOn(console, 'log').mockImplementation(() => {})
|
||||
reportCliError(error, true, { commandPath: ['orchestration', 'dispatch'] })
|
||||
expect(JSON.parse(log.mock.calls[0]?.[0] as string)).toEqual(failure)
|
||||
})
|
||||
|
||||
function envelope(receipt: DispatchRefusalReceipt): RuntimeRpcFailure {
|
||||
return { id: 'rpc_1', ok: false, error: receipt, _meta: { runtimeId: 'runtime_1' } }
|
||||
}
|
||||
@@ -336,7 +336,7 @@ function createResponseBody(slug: string): object {
|
||||
renderedContentType: 'text/html',
|
||||
createdAt: '2026-08-06T00:00:00.000Z',
|
||||
updatedAt: '2026-08-06T00:00:00.000Z',
|
||||
expiresAt: '2026-09-06T00:00:00.000Z',
|
||||
expiresAt: new Date(Date.now() + 30 * 24 * 60 * 60 * 1000).toISOString(),
|
||||
byteSize: 17,
|
||||
deletedAt: null
|
||||
},
|
||||
|
||||
@@ -33,7 +33,7 @@ function createResponse(slug: string): Response {
|
||||
renderedContentType: 'text/html',
|
||||
createdAt: '2026-08-06T00:00:00.000Z',
|
||||
updatedAt: '2026-08-06T00:00:00.000Z',
|
||||
expiresAt: '2026-09-06T00:00:00.000Z',
|
||||
expiresAt: new Date(Date.now() + 30 * 24 * 60 * 60 * 1000).toISOString(),
|
||||
byteSize: 12,
|
||||
deletedAt: null
|
||||
},
|
||||
|
||||
@@ -43,7 +43,10 @@ const cloudB: OrcaProfileCloudSummary = {
|
||||
linkedAt: 2
|
||||
}
|
||||
|
||||
function createResponse(slug = 'artifact-a', expiresAt = '2026-09-06T00:00:00.000Z'): Response {
|
||||
function createResponse(
|
||||
slug = 'artifact-a',
|
||||
expiresAt = new Date(Date.now() + 30 * 24 * 60 * 60 * 1000).toISOString()
|
||||
): Response {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
artifact: {
|
||||
|
||||
@@ -566,7 +566,7 @@ describe('Claude stream-json connection', () => {
|
||||
)
|
||||
})
|
||||
|
||||
it('reports a self-exit with its status and stderr, and leaves its tree unverifiable', async () => {
|
||||
it('reports a self-exit with its status, stderr, and observed tree verdict', async () => {
|
||||
const scenario = scriptScenario([{ stderr: 'claude: not signed in\n' }, { exit: 1 }])
|
||||
let exit: Error | null = null
|
||||
const connection = await open(launchFor(scenario), {
|
||||
@@ -579,10 +579,11 @@ describe('Claude stream-json connection', () => {
|
||||
// The status and stderr are the only diagnostic a refused start leaves behind.
|
||||
expect((exit as unknown as Error).message).toMatch(/exited \(code 1\): claude: not signed in/)
|
||||
expect(connection.closed).toBe(true)
|
||||
// The root's exit is first-hand, but it left before a descendant snapshot
|
||||
// could be armed, so close() has no tree proof to offer and says so.
|
||||
await expect(connection.close()).resolves.toBe(false)
|
||||
expect(connection.exitVerdict).toEqual({ root: 'exited', tree: 'unverifiable' })
|
||||
// Stderr-triggered capture can win or lose the race with this real child's exit.
|
||||
const closed = await connection.close()
|
||||
expect(connection.exitVerdict.root).toBe('exited')
|
||||
expect(['exited', 'unverifiable']).toContain(connection.exitVerdict.tree)
|
||||
expect(closed).toBe(connection.exitVerdict.tree === 'exited')
|
||||
})
|
||||
|
||||
it.runIf(process.platform !== 'win32')(
|
||||
|
||||
@@ -158,4 +158,42 @@ describe('stopClaudeBackgroundTasks', () => {
|
||||
await stopClaudeBackgroundTasks(session, undefined, () => current)
|
||||
expect(stopTask).toHaveBeenCalledTimes(1)
|
||||
})
|
||||
|
||||
it('stops only the requested live task id', async () => {
|
||||
const backgroundTasks = new ClaudeBackgroundTaskTracker()
|
||||
backgroundTasks.observe({
|
||||
type: 'system',
|
||||
subtype: 'background_tasks_changed',
|
||||
tasks: [
|
||||
{ task_id: 'task-one', task_type: 'local_agent' },
|
||||
{ task_id: 'task-two', task_type: 'local_bash' }
|
||||
]
|
||||
})
|
||||
const stopTask = vi.fn(async (_taskId: string) => {})
|
||||
const session = { backgroundTasks, connection: { stopTask } } as unknown as ClaudeSession
|
||||
|
||||
await expect(
|
||||
stopClaudeBackgroundTasks(session, 5_000, () => true, 'task-two')
|
||||
).resolves.toEqual({ cancelled: true })
|
||||
expect(stopTask).toHaveBeenCalledWith('task-two', { timeoutMs: 5_000 })
|
||||
expect(stopTask).toHaveBeenCalledTimes(1)
|
||||
})
|
||||
|
||||
it('refuses a stale or unknown task id without a provider call', async () => {
|
||||
const backgroundTasks = new ClaudeBackgroundTaskTracker()
|
||||
backgroundTasks.observe({
|
||||
type: 'system',
|
||||
subtype: 'task_started',
|
||||
task_id: 'task-live',
|
||||
task_type: 'local_agent',
|
||||
is_backgrounded: true
|
||||
})
|
||||
const stopTask = vi.fn(async (_taskId: string) => {})
|
||||
const session = { backgroundTasks, connection: { stopTask } } as unknown as ClaudeSession
|
||||
|
||||
await expect(
|
||||
stopClaudeBackgroundTasks(session, undefined, () => true, 'task-stale')
|
||||
).resolves.toEqual({ cancelled: false })
|
||||
expect(stopTask).not.toHaveBeenCalled()
|
||||
})
|
||||
})
|
||||
|
||||
@@ -46,9 +46,12 @@ export async function cancelClaudeTurn(
|
||||
export async function stopClaudeBackgroundTasks(
|
||||
session: ClaudeSession,
|
||||
timeoutMs: number | undefined,
|
||||
isCurrent: ClaudeTurnCancellationGuard = () => true
|
||||
isCurrent: ClaudeTurnCancellationGuard = () => true,
|
||||
taskId?: string
|
||||
): Promise<{ cancelled: boolean }> {
|
||||
const taskIds = session.backgroundTasks.stoppableTaskIds
|
||||
const stoppableTaskIds = session.backgroundTasks.stoppableTaskIds
|
||||
const taskIds =
|
||||
taskId === undefined ? stoppableTaskIds : stoppableTaskIds.includes(taskId) ? [taskId] : []
|
||||
let cancelled = false
|
||||
for (const taskId of taskIds) {
|
||||
if (!isCurrent()) {
|
||||
|
||||
@@ -30,6 +30,7 @@ import {
|
||||
settleClaudeExitedSession
|
||||
} from './claude-structured-session-close'
|
||||
import { readClaudeTranscriptLeafWithReproof } from './claude-transcript-branch-proof'
|
||||
import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire'
|
||||
|
||||
export type { ClaudeStructuredLaunch } from './claude-structured-launch-resolution'
|
||||
export type {
|
||||
@@ -40,6 +41,11 @@ export type {
|
||||
|
||||
const DISPATCH_ACK_TIMEOUT_MS = 10_000
|
||||
|
||||
function backgroundTaskState(session: ClaudeSession): AgentSessionBackgroundTaskState | null {
|
||||
const state = session.backgroundTasks.state
|
||||
return state ? { ...state, supportsTaskStop: true } : null
|
||||
}
|
||||
|
||||
export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAdapter {
|
||||
private readonly sessions = new Map<string, ClaudeSession>()
|
||||
private readonly acquisitions = new ClaudeAcquisitionRegistry()
|
||||
@@ -192,7 +198,10 @@ export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAda
|
||||
session?.translator?.handle(event)
|
||||
this.deps.onEvent?.(event)
|
||||
if (backgroundTasksChanged) {
|
||||
this.deps.onBackgroundTasksChanged?.(event.sessionId, session?.backgroundTasks.state ?? null)
|
||||
this.deps.onBackgroundTasksChanged?.(
|
||||
event.sessionId,
|
||||
session ? backgroundTaskState(session) : null
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -233,18 +242,25 @@ export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAda
|
||||
stopBackgroundTasks: StructuredAgentSessionAdapter['stopBackgroundTasks'] = (input) => {
|
||||
const session = this.session(input.sessionId)
|
||||
const acquisitionGeneration = session.acquisitionGeneration
|
||||
return stopClaudeBackgroundTasks(session, this.deps.requestTimeoutMs, () =>
|
||||
Boolean(
|
||||
this.sessions.get(input.sessionId) === session &&
|
||||
session.fence === input.fence &&
|
||||
session.acquisitionGeneration === acquisitionGeneration &&
|
||||
session.backgroundTasks.state
|
||||
)
|
||||
return stopClaudeBackgroundTasks(
|
||||
session,
|
||||
this.deps.requestTimeoutMs,
|
||||
() =>
|
||||
Boolean(
|
||||
this.sessions.get(input.sessionId) === session &&
|
||||
session.fence === input.fence &&
|
||||
session.acquisitionGeneration === acquisitionGeneration &&
|
||||
session.backgroundTasks.state
|
||||
),
|
||||
input.taskId
|
||||
)
|
||||
}
|
||||
backgroundTaskState: NonNullable<StructuredAgentSessionAdapter['backgroundTaskState']> = (
|
||||
sessionId
|
||||
) => this.sessions.get(sessionId)?.backgroundTasks.state
|
||||
) => {
|
||||
const session = this.sessions.get(sessionId)
|
||||
return session ? backgroundTaskState(session) : undefined
|
||||
}
|
||||
answerPrompt: StructuredAgentSessionAdapter['answerPrompt'] = (input) =>
|
||||
answerClaudePrompt(this.session(input.sessionId), input)
|
||||
setOption: StructuredAgentSessionAdapter['setOption'] = (input) =>
|
||||
|
||||
@@ -53,7 +53,11 @@ describe('Claude published session close lifecycle', () => {
|
||||
is_backgrounded: true
|
||||
})
|
||||
expect(backgroundStates).toEqual([
|
||||
{ state: 'monitoring', tasks: [{ id: 'background-1', kind: 'agent' }] }
|
||||
{
|
||||
state: 'monitoring',
|
||||
tasks: [{ id: 'background-1', kind: 'agent' }],
|
||||
supportsTaskStop: true
|
||||
}
|
||||
])
|
||||
const session = (
|
||||
adapter as unknown as {
|
||||
@@ -68,7 +72,11 @@ describe('Claude published session close lifecycle', () => {
|
||||
expect(events.filter((event) => event.type === 'handle')).toHaveLength(0)
|
||||
expect(disposeTranslator).toHaveBeenCalledOnce()
|
||||
expect(backgroundStates).toEqual([
|
||||
{ state: 'monitoring', tasks: [{ id: 'background-1', kind: 'agent' }] },
|
||||
{
|
||||
state: 'monitoring',
|
||||
tasks: [{ id: 'background-1', kind: 'agent' }],
|
||||
supportsTaskStop: true
|
||||
},
|
||||
null
|
||||
])
|
||||
|
||||
|
||||
@@ -0,0 +1,71 @@
|
||||
import { readRecord, readString } from './codex-item-field-readers'
|
||||
import type { CodexThreadItem } from './codex-thread-item-identity'
|
||||
|
||||
/**
|
||||
* Codex's own classification of a shell call: the tool name to show, and the
|
||||
* fields worth lifting into `input` for the shared label helper (a file target,
|
||||
* a search term, a scanned root). A `Map`, not an object — an object index
|
||||
* answers `__proto__` with a truthy non-string. Every other action type stays an
|
||||
* unclassified `shell` row.
|
||||
*
|
||||
* Nothing is invented for a field Codex sends as null: a stand-in path is a
|
||||
* claim about a target, and the label helper turns any path into a file link.
|
||||
*/
|
||||
type CommandActionClass = {
|
||||
name: string
|
||||
/** Action field to the `input` key it lifts to. A scan root and a listed
|
||||
* directory lift to `directory`, never `path`: the label helper reads `path`
|
||||
* as a file target, which mobile turns into a tappable open-file link. */
|
||||
keys: Readonly<Record<string, string>>
|
||||
}
|
||||
|
||||
const COMMAND_ACTION_CLASSES = new Map<string, CommandActionClass>([
|
||||
['read', { name: 'read', keys: { path: 'path' } }],
|
||||
['search', { name: 'search', keys: { query: 'query', path: 'directory' } }],
|
||||
['listFiles', { name: 'list', keys: { path: 'directory' } }]
|
||||
])
|
||||
|
||||
/** The one class every classified `commandActions` entry agrees on, with the
|
||||
* fields they all agree on; null leaves the row exactly as a Codex that sends no
|
||||
* classification renders it. `cat a.txt && ls src` classifies as two different
|
||||
* things, and naming that row after either would drop the other, so it stays a
|
||||
* `shell` row that shows the whole command. */
|
||||
export function commandActionFacts(
|
||||
item: CodexThreadItem
|
||||
): { name: string; fields: Record<string, string> } | null {
|
||||
const actions = item.commandActions
|
||||
if (!Array.isArray(actions)) {
|
||||
return null
|
||||
}
|
||||
let matched: { class: CommandActionClass; fields: Record<string, string> } | null = null
|
||||
for (const action of actions) {
|
||||
const record = readRecord(action)
|
||||
const type = readString(record, 'type')
|
||||
const classified = type === null ? undefined : COMMAND_ACTION_CLASSES.get(type)
|
||||
if (classified === undefined) {
|
||||
continue
|
||||
}
|
||||
if (matched === null) {
|
||||
const fields: Record<string, string> = {}
|
||||
for (const [source, lifted] of Object.entries(classified.keys)) {
|
||||
const value = readString(record, source)
|
||||
if (value !== null) {
|
||||
fields[lifted] = value
|
||||
}
|
||||
}
|
||||
matched = { class: classified, fields }
|
||||
continue
|
||||
}
|
||||
if (matched.class.name !== classified.name) {
|
||||
return null
|
||||
}
|
||||
// The same class twice keeps the class, but only a target both entries name.
|
||||
for (const [source, lifted] of Object.entries(matched.class.keys)) {
|
||||
const kept = matched.fields[lifted]
|
||||
if (kept !== undefined && readString(record, source) !== kept) {
|
||||
delete matched.fields[lifted]
|
||||
}
|
||||
}
|
||||
}
|
||||
return matched === null ? null : { name: matched.class.name, fields: matched.fields }
|
||||
}
|
||||
@@ -0,0 +1,45 @@
|
||||
// Field readers for the loosely-typed records Codex sends on thread items.
|
||||
|
||||
export function readRecord(value: unknown): Record<string, unknown> {
|
||||
return typeof value === 'object' && value !== null ? (value as Record<string, unknown>) : {}
|
||||
}
|
||||
|
||||
export function readString(source: Record<string, unknown>, key: string): string | null {
|
||||
const value = source[key]
|
||||
return typeof value === 'string' && value.length > 0 ? value : null
|
||||
}
|
||||
|
||||
export function readFirstString(
|
||||
source: Record<string, unknown>,
|
||||
keys: readonly string[]
|
||||
): string | null {
|
||||
for (const key of keys) {
|
||||
const value = readString(source, key)
|
||||
if (value !== null) {
|
||||
return value
|
||||
}
|
||||
}
|
||||
return null
|
||||
}
|
||||
|
||||
export function readTextContent(source: Record<string, unknown>, key: string): string | null {
|
||||
const direct = readString(source, key)
|
||||
if (direct) {
|
||||
return direct
|
||||
}
|
||||
const value = source[key]
|
||||
if (!Array.isArray(value)) {
|
||||
return null
|
||||
}
|
||||
const parts = value.flatMap((part) => {
|
||||
if (typeof part === 'string') {
|
||||
return part.length > 0 ? [part] : []
|
||||
}
|
||||
if (typeof part !== 'object' || part === null) {
|
||||
return []
|
||||
}
|
||||
const text = readString(part as Record<string, unknown>, 'text')
|
||||
return text ? [text] : []
|
||||
})
|
||||
return parts.length > 0 ? parts.join('\n') : null
|
||||
}
|
||||
@@ -1,6 +1,10 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key'
|
||||
import { createToolInputDisplay } from '../../shared/native-chat-tool-summary'
|
||||
import {
|
||||
briefToolArg,
|
||||
createToolInputDisplay,
|
||||
describeToolInput
|
||||
} from '../../shared/native-chat-tool-summary'
|
||||
import {
|
||||
codexItemBody,
|
||||
codexItemIdentity,
|
||||
@@ -14,6 +18,13 @@ import {
|
||||
type CodexThreadItem
|
||||
} from './codex-structured-item-translation'
|
||||
|
||||
/** The tool-call input a Codex item lands on, which is what the row label and
|
||||
* the collapsed run header are both derived from. */
|
||||
function toolCallInput(item: CodexThreadItem): unknown {
|
||||
const body = codexItemBody(item)
|
||||
return body !== null && body.kind === 'tool-call' ? body.input : null
|
||||
}
|
||||
|
||||
const THREAD_ID = 'thread-abc'
|
||||
const TURN_ID = 'turn-1'
|
||||
|
||||
@@ -572,10 +583,226 @@ describe('codex item bodies', () => {
|
||||
})
|
||||
expect(codexItemBody({ type: 'reasoning', id: 'r' })).toBeNull()
|
||||
expect(codexItemBody({ type: 'agentMessage', id: 'm', text: '' })).toBeNull()
|
||||
expect(codexItemBody({ type: 'webSearch', id: 'w' })).toMatchObject({
|
||||
expect(codexItemBody({ type: 'somethingCodexAddedLater', id: 'x' })).toMatchObject({
|
||||
kind: 'status',
|
||||
text: 'codex · item:webSearch',
|
||||
providerFrame: { provider: 'codex', kind: 'item:webSearch' }
|
||||
text: 'codex · item:somethingCodexAddedLater',
|
||||
providerFrame: { provider: 'codex', kind: 'item:somethingCodexAddedLater' }
|
||||
})
|
||||
})
|
||||
|
||||
it('gives an mcp tool call a typed body with its own arguments as input', () => {
|
||||
expect(
|
||||
codexItemBody({
|
||||
type: 'mcpToolCall',
|
||||
id: 'mcp-1',
|
||||
server: 'weather',
|
||||
tool: 'get_forecast',
|
||||
status: 'completed',
|
||||
arguments: { city: 'Oslo' },
|
||||
result: { content: [{ type: 'text', text: '12C' }] }
|
||||
})
|
||||
).toEqual({
|
||||
kind: 'tool-call',
|
||||
// Server-qualified, and the arguments stay top level so the row label can
|
||||
// read `query`/`command`/`file_path` out of them.
|
||||
name: 'weather/get_forecast',
|
||||
input: { city: 'Oslo' },
|
||||
state: 'completed',
|
||||
output: { head: '12C', byteLength: 3, truncated: false, digest: expect.any(String) }
|
||||
})
|
||||
})
|
||||
|
||||
it('passes an mcp tool name through with no casing transform', () => {
|
||||
// Downstream dispatch is exact-match on raw identifiers, so every shape —
|
||||
// bare snake_case included — has to survive byte-identical.
|
||||
for (const tool of ['get_forecast', 'mcp__server__tool', 'ns.tool', 'urn:tool', 'listTools']) {
|
||||
expect(
|
||||
codexItemBody({ type: 'mcpToolCall', id: 'm', tool, status: 'inProgress' }),
|
||||
tool
|
||||
).toMatchObject({ kind: 'tool-call', name: tool, state: 'running' })
|
||||
expect(
|
||||
codexItemBody({ type: 'mcpToolCall', id: 'm', server: 'srv', tool, status: 'inProgress' }),
|
||||
tool
|
||||
).toMatchObject({ kind: 'tool-call', name: `srv/${tool}`, state: 'running' })
|
||||
}
|
||||
})
|
||||
|
||||
it('falls back to the bare tool, then to `mcp`, when the item is under-specified', () => {
|
||||
expect(
|
||||
codexItemBody({ type: 'mcpToolCall', id: 'm', tool: 'get_forecast', status: 'inProgress' })
|
||||
).toMatchObject({ name: 'get_forecast' })
|
||||
expect(
|
||||
codexItemBody({ type: 'mcpToolCall', id: 'm', server: '', tool: 'ping', status: 'completed' })
|
||||
).toMatchObject({ name: 'ping' })
|
||||
expect(
|
||||
codexItemBody({ type: 'mcpToolCall', id: 'm', server: 'weather', status: 'completed' })
|
||||
).toMatchObject({ name: 'mcp' })
|
||||
})
|
||||
|
||||
it('keeps non-object mcp arguments addressable and empty ones off the label', () => {
|
||||
// `arguments` is arbitrary JSON upstream; a scalar or array must still reach
|
||||
// the row rather than being dropped or unwrapped into a bare value.
|
||||
expect(
|
||||
codexItemBody({ type: 'mcpToolCall', id: 'm', tool: 't', arguments: 'raw text' })
|
||||
).toMatchObject({ input: { arguments: 'raw text' } })
|
||||
expect(
|
||||
codexItemBody({ type: 'mcpToolCall', id: 'm', tool: 't', arguments: [1, 2] })
|
||||
).toMatchObject({ input: { arguments: [1, 2] } })
|
||||
// `arguments` is required on the wire, so `{}` — not an absent key — is what
|
||||
// an argument-less MCP tool sends, and passing it through labels the row `{}`.
|
||||
expect(codexItemBody({ type: 'mcpToolCall', id: 'm', tool: 't', arguments: {} })).toEqual({
|
||||
kind: 'tool-call',
|
||||
name: 't',
|
||||
input: null,
|
||||
state: 'running'
|
||||
})
|
||||
expect(codexItemBody({ type: 'mcpToolCall', id: 'm', tool: 't' })).toMatchObject({
|
||||
input: null
|
||||
})
|
||||
expect(
|
||||
codexItemBody({ type: 'mcpToolCall', id: 'm', tool: 't', arguments: null })
|
||||
).toMatchObject({ input: null })
|
||||
})
|
||||
|
||||
it('renders an argument-less mcp call as a bare server/tool row', () => {
|
||||
const input = toolCallInput({
|
||||
type: 'mcpToolCall',
|
||||
id: 'm',
|
||||
server: 'srv',
|
||||
tool: 'list_tools',
|
||||
arguments: {}
|
||||
})
|
||||
expect(describeToolInput(input)).toBe('')
|
||||
expect(briefToolArg(input)).toBe('')
|
||||
})
|
||||
|
||||
it('reports an mcp error as a failed call carrying the server message', () => {
|
||||
expect(
|
||||
codexItemBody({
|
||||
type: 'mcpToolCall',
|
||||
id: 'mcp-2',
|
||||
server: 's',
|
||||
tool: 'ping',
|
||||
status: 'completed',
|
||||
error: { message: 'server unreachable' }
|
||||
})
|
||||
).toMatchObject({
|
||||
kind: 'tool-call',
|
||||
name: 's/ping',
|
||||
state: 'failed',
|
||||
output: { head: 'server unreachable', truncated: false }
|
||||
})
|
||||
})
|
||||
|
||||
it('models a web search as a tool call that runs until codex sends the action', () => {
|
||||
// The start frame Codex actually emits: empty query, no action. Nothing is
|
||||
// labelable yet, so the input is absent rather than a hull of null keys.
|
||||
expect(codexItemBody({ type: 'webSearch', id: 'w', query: '', action: null })).toEqual({
|
||||
kind: 'tool-call',
|
||||
name: 'web_search',
|
||||
input: null,
|
||||
state: 'running'
|
||||
})
|
||||
expect(
|
||||
codexItemBody({
|
||||
type: 'webSearch',
|
||||
id: 'w',
|
||||
query: 'orca release notes',
|
||||
action: { type: 'search', query: 'orca release notes', queries: null },
|
||||
results: null
|
||||
})
|
||||
).toEqual({
|
||||
kind: 'tool-call',
|
||||
name: 'web_search',
|
||||
input: {
|
||||
query: 'orca release notes',
|
||||
description: 'search',
|
||||
action: { type: 'search', query: 'orca release notes', queries: null }
|
||||
},
|
||||
state: 'completed'
|
||||
})
|
||||
})
|
||||
|
||||
it('carries the web search hits as the call output', () => {
|
||||
const results = [{ title: 'Orca 1.0', url: 'https://example.com/notes' }]
|
||||
expect(
|
||||
codexItemBody({
|
||||
type: 'webSearch',
|
||||
id: 'w',
|
||||
query: 'orca release notes',
|
||||
action: { type: 'search', query: 'orca release notes', queries: null },
|
||||
results
|
||||
})
|
||||
).toMatchObject({
|
||||
kind: 'tool-call',
|
||||
name: 'web_search',
|
||||
state: 'completed',
|
||||
output: { head: JSON.stringify(results), truncated: false }
|
||||
})
|
||||
// Nothing to show is no output block at all, not an empty one.
|
||||
for (const empty of [undefined, null, []]) {
|
||||
expect(
|
||||
codexItemBody({
|
||||
type: 'webSearch',
|
||||
id: 'w',
|
||||
query: 'q',
|
||||
action: { type: 'search' },
|
||||
results: empty
|
||||
}),
|
||||
String(empty)
|
||||
).not.toHaveProperty('output')
|
||||
}
|
||||
})
|
||||
|
||||
it('labels every web search shape without falling back to raw JSON', () => {
|
||||
// Both the row label and the run header read top-level input keys only, so a
|
||||
// shape whose detail sits inside `action` renders as the input's raw JSON.
|
||||
const url = 'https://example.com/docs/page'
|
||||
const shapes: [string, unknown, string, string][] = [
|
||||
['started', null, '', ''],
|
||||
[
|
||||
'search',
|
||||
{ type: 'search', query: 'a sample query', queries: null },
|
||||
'a sample query',
|
||||
'a sample query'
|
||||
],
|
||||
['openPage', { type: 'openPage', url }, url, ''],
|
||||
[
|
||||
'findInPage',
|
||||
{ type: 'findInPage', url, pattern: 'a needle' },
|
||||
'a sample query',
|
||||
'a sample query'
|
||||
],
|
||||
['other', { type: 'other' }, 'other', '']
|
||||
]
|
||||
for (const [name, action, label, brief] of shapes) {
|
||||
// Codex leaves the item's own `query` empty on most completed searches.
|
||||
const query = name === 'search' || name === 'findInPage' ? 'a sample query' : ''
|
||||
const input = toolCallInput({ type: 'webSearch', id: 'w', query, action })
|
||||
expect(describeToolInput(input), name).toBe(label)
|
||||
expect(briefToolArg(input), name).toBe(brief)
|
||||
}
|
||||
})
|
||||
|
||||
it('leaves subagent items on the generic row until a real renderer exists', () => {
|
||||
expect(
|
||||
codexJournalItem({
|
||||
type: 'subAgentActivity',
|
||||
id: 'a-1',
|
||||
kind: 'started',
|
||||
agentThreadId: 'thread-child',
|
||||
agentPath: '/root/list_directory'
|
||||
})
|
||||
).toMatchObject({
|
||||
handled: false,
|
||||
body: { kind: 'status', providerFrame: { kind: 'item:subAgentActivity' } }
|
||||
})
|
||||
})
|
||||
|
||||
it('drops the sleep item, which codex itself renders as nothing', () => {
|
||||
expect(codexJournalItem({ type: 'sleep', id: 's-1', durationMs: 20_000 })).toEqual({
|
||||
body: null,
|
||||
handled: true
|
||||
})
|
||||
})
|
||||
|
||||
|
||||
@@ -1,7 +1,4 @@
|
||||
import type {
|
||||
AgentJournalItemBody,
|
||||
AgentJournalItemIdentity
|
||||
} from '../../shared/agent-session-journal-types'
|
||||
import type { AgentJournalItemBody } from '../../shared/agent-session-journal-types'
|
||||
import type { NativeChatBlock } from '../../shared/native-chat-types'
|
||||
import {
|
||||
boundInlineText,
|
||||
@@ -9,116 +6,27 @@ import {
|
||||
DEFAULT_JOURNAL_PAYLOAD_LIMITS
|
||||
} from '../native-chat/agent-session-journal/journal-payload-bounds'
|
||||
import { unhandledProviderFrameJournalItem } from '../native-chat/agent-session-wire/unhandled-provider-frame'
|
||||
import type { CodexTurnOrdinals } from './codex-turn-ordinals'
|
||||
import { commandActionFacts } from './codex-command-action-class'
|
||||
import {
|
||||
readFirstString,
|
||||
readRecord,
|
||||
readString,
|
||||
readTextContent
|
||||
} from './codex-item-field-readers'
|
||||
import type { CodexThreadItem } from './codex-thread-item-identity'
|
||||
export {
|
||||
codexItemIdentity,
|
||||
isCodexMessageItemType,
|
||||
readCodexThreadItem,
|
||||
type CodexThreadItem
|
||||
} from './codex-thread-item-identity'
|
||||
export {
|
||||
CodexTurnOrdinals,
|
||||
MAX_CODEX_TURN_ORDINAL_BYTES,
|
||||
MAX_CODEX_TURN_ORDINAL_ENTRIES
|
||||
} from './codex-turn-ordinals'
|
||||
|
||||
// Codex thread items → journal item bodies and durable identities.
|
||||
//
|
||||
// THE ORDINAL RULE, and why it is not "index within the turn". Codex renumbers
|
||||
// item ids positionally on resume (`item-1`…`item-N` across the whole thread),
|
||||
// and a resumed turn does NOT contain every item the live turn emitted —
|
||||
// reasoning and command execution are dropped from persisted history. Numbering
|
||||
// by live position would therefore shift every message after the first tool
|
||||
// call and hand the user a duplicate of the assistant's answer after a resume.
|
||||
//
|
||||
// So the ordinal counts MESSAGE items only, and the same projection is applied
|
||||
// to the live stream and to a resumed turn's item list. Any other item type —
|
||||
// including ones this build does not model — is skipped identically on both
|
||||
// sides, which is what makes the key survive a Codex release that adds one.
|
||||
|
||||
/** Only these carry a durable `(threadId, turnId, ordinal)` identity. */
|
||||
const CODEX_MESSAGE_ITEM_TYPES = new Set(['userMessage', 'agentMessage'])
|
||||
|
||||
export type CodexThreadItem = {
|
||||
type: string
|
||||
id: string
|
||||
[key: string]: unknown
|
||||
}
|
||||
|
||||
export function isCodexMessageItemType(type: string): boolean {
|
||||
return CODEX_MESSAGE_ITEM_TYPES.has(type)
|
||||
}
|
||||
|
||||
export function readCodexThreadItem(value: unknown): CodexThreadItem | null {
|
||||
if (typeof value !== 'object' || value === null) {
|
||||
return null
|
||||
}
|
||||
const record = value as Record<string, unknown>
|
||||
return typeof record.type === 'string' && typeof record.id === 'string'
|
||||
? (record as CodexThreadItem)
|
||||
: null
|
||||
}
|
||||
|
||||
function readRecord(value: unknown): Record<string, unknown> {
|
||||
return typeof value === 'object' && value !== null ? (value as Record<string, unknown>) : {}
|
||||
}
|
||||
|
||||
/**
|
||||
* Durable identity for a Codex item, or null for one that has none.
|
||||
*
|
||||
* Non-message items fall back to the `orca` namespace keyed by the Codex item
|
||||
* id. That id is unstable across resume, so those rows are live-session detail
|
||||
* that a recovered journal simply will not contain — which is correct: Codex
|
||||
* itself does not persist them either.
|
||||
*/
|
||||
export function codexItemIdentity(input: {
|
||||
threadId: string
|
||||
turnId: string | null
|
||||
item: CodexThreadItem
|
||||
ordinals: CodexTurnOrdinals
|
||||
}): AgentJournalItemIdentity {
|
||||
const { item, turnId } = input
|
||||
if (turnId && isCodexMessageItemType(item.type)) {
|
||||
return {
|
||||
provider: 'codex',
|
||||
threadId: input.threadId,
|
||||
turnId,
|
||||
ordinal: input.ordinals.ordinalFor(input.threadId, turnId, item.id)
|
||||
}
|
||||
}
|
||||
return { provider: 'orca', clientMessageId: `codex-item:${input.threadId}:${item.id}` }
|
||||
}
|
||||
|
||||
function readString(source: Record<string, unknown>, key: string): string | null {
|
||||
const value = source[key]
|
||||
return typeof value === 'string' && value.length > 0 ? value : null
|
||||
}
|
||||
|
||||
function readFirstString(source: Record<string, unknown>, keys: readonly string[]): string | null {
|
||||
for (const key of keys) {
|
||||
const value = readString(source, key)
|
||||
if (value !== null) {
|
||||
return value
|
||||
}
|
||||
}
|
||||
return null
|
||||
}
|
||||
|
||||
function readTextContent(source: Record<string, unknown>, key: string): string | null {
|
||||
const direct = readString(source, key)
|
||||
if (direct) {
|
||||
return direct
|
||||
}
|
||||
const value = source[key]
|
||||
if (!Array.isArray(value)) {
|
||||
return null
|
||||
}
|
||||
const parts = value.flatMap((part) => {
|
||||
if (typeof part === 'string') {
|
||||
return part.length > 0 ? [part] : []
|
||||
}
|
||||
if (typeof part !== 'object' || part === null) {
|
||||
return []
|
||||
}
|
||||
const text = readString(part as Record<string, unknown>, 'text')
|
||||
return text ? [text] : []
|
||||
})
|
||||
return parts.length > 0 ? parts.join('\n') : null
|
||||
}
|
||||
// Codex thread items → journal item bodies.
|
||||
|
||||
/** `userMessage` carries structured content parts; `agentMessage` a flat text. */
|
||||
export function codexMessageBlocks(item: CodexThreadItem): NativeChatBlock[] {
|
||||
@@ -175,75 +83,6 @@ export type CodexJournalItem = {
|
||||
handled: boolean
|
||||
}
|
||||
|
||||
/**
|
||||
* Codex's own classification of a shell call: the tool name to show, and the
|
||||
* fields worth lifting into `input` for the shared label helper (a file target,
|
||||
* a search term, a scanned root). A `Map`, not an object — an object index
|
||||
* answers `__proto__` with a truthy non-string. Every other action type stays an
|
||||
* unclassified `shell` row.
|
||||
*
|
||||
* Nothing is invented for a field Codex sends as null: a stand-in path is a
|
||||
* claim about a target, and the label helper turns any path into a file link.
|
||||
*/
|
||||
type CommandActionClass = {
|
||||
name: string
|
||||
/** Action field to the `input` key it lifts to. A scan root and a listed
|
||||
* directory lift to `directory`, never `path`: the label helper reads `path`
|
||||
* as a file target, which mobile turns into a tappable open-file link. */
|
||||
keys: Readonly<Record<string, string>>
|
||||
}
|
||||
|
||||
const COMMAND_ACTION_CLASSES = new Map<string, CommandActionClass>([
|
||||
['read', { name: 'read', keys: { path: 'path' } }],
|
||||
['search', { name: 'search', keys: { query: 'query', path: 'directory' } }],
|
||||
['listFiles', { name: 'list', keys: { path: 'directory' } }]
|
||||
])
|
||||
|
||||
/** The one class every classified `commandActions` entry agrees on, with the
|
||||
* fields they all agree on; null leaves the row exactly as a Codex that sends no
|
||||
* classification renders it. `cat a.txt && ls src` classifies as two different
|
||||
* things, and naming that row after either would drop the other, so it stays a
|
||||
* `shell` row that shows the whole command. */
|
||||
function commandActionFacts(
|
||||
item: CodexThreadItem
|
||||
): { name: string; fields: Record<string, string> } | null {
|
||||
const actions = item.commandActions
|
||||
if (!Array.isArray(actions)) {
|
||||
return null
|
||||
}
|
||||
let matched: { class: CommandActionClass; fields: Record<string, string> } | null = null
|
||||
for (const action of actions) {
|
||||
const record = readRecord(action)
|
||||
const type = readString(record, 'type')
|
||||
const classified = type === null ? undefined : COMMAND_ACTION_CLASSES.get(type)
|
||||
if (classified === undefined) {
|
||||
continue
|
||||
}
|
||||
if (matched === null) {
|
||||
const fields: Record<string, string> = {}
|
||||
for (const [source, lifted] of Object.entries(classified.keys)) {
|
||||
const value = readString(record, source)
|
||||
if (value !== null) {
|
||||
fields[lifted] = value
|
||||
}
|
||||
}
|
||||
matched = { class: classified, fields }
|
||||
continue
|
||||
}
|
||||
if (matched.class.name !== classified.name) {
|
||||
return null
|
||||
}
|
||||
// The same class twice keeps the class, but only a target both entries name.
|
||||
for (const [source, lifted] of Object.entries(matched.class.keys)) {
|
||||
const kept = matched.fields[lifted]
|
||||
if (kept !== undefined && readString(record, source) !== kept) {
|
||||
delete matched.fields[lifted]
|
||||
}
|
||||
}
|
||||
}
|
||||
return matched === null ? null : { name: matched.class.name, fields: matched.fields }
|
||||
}
|
||||
|
||||
function commandItem(item: CodexThreadItem): CodexJournalItem {
|
||||
const output = readFirstString(item, ['aggregatedOutput', 'aggregated_output'])
|
||||
const bounded = output === null ? null : boundInlineText(output, DEFAULT_JOURNAL_PAYLOAD_LIMITS)
|
||||
@@ -296,6 +135,85 @@ function fileChangeItem(item: CodexThreadItem): CodexJournalItem {
|
||||
}
|
||||
}
|
||||
|
||||
/** The tool name reaches the row verbatim — downstream dispatch (diff renderer,
|
||||
* question parsers, input previews) matches raw identifiers, so any casing
|
||||
* transform would silently miss them. `server/` qualifies it so two servers
|
||||
* exposing the same tool stay distinguishable and neither shadows a built-in. */
|
||||
function mcpToolCallName(item: CodexThreadItem): string {
|
||||
const tool = readString(item, 'tool')
|
||||
const server = readString(item, 'server')
|
||||
return tool === null ? 'mcp' : server === null ? tool : `${server}/${tool}`
|
||||
}
|
||||
|
||||
/** Row-label derivation only reads top-level keys, so the call's own arguments
|
||||
* have to be the input itself. `arguments` is arbitrary JSON upstream: a
|
||||
* non-object stays addressable under a key rather than being dropped, while a
|
||||
* no-argument call — `{}` on the wire, the shape every argument-less MCP tool
|
||||
* sends — becomes null so the row reads as a bare `server/tool` instead of a
|
||||
* literal `{}`. */
|
||||
function mcpToolArguments(value: unknown): unknown {
|
||||
if (typeof value !== 'object' || value === null) {
|
||||
return value === null || value === undefined ? null : { arguments: value }
|
||||
}
|
||||
return Array.isArray(value) ? { arguments: value } : Object.keys(value).length > 0 ? value : null
|
||||
}
|
||||
|
||||
function mcpToolCallItem(item: CodexThreadItem): CodexJournalItem {
|
||||
const failure = readString(readRecord(item.error), 'message')
|
||||
const text = failure ?? readTextContent(readRecord(item.result), 'content')
|
||||
const bounded = text === null ? null : boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS)
|
||||
return {
|
||||
body: {
|
||||
kind: 'tool-call',
|
||||
name: mcpToolCallName(item),
|
||||
input: boundToolInput(mcpToolArguments(item.arguments), DEFAULT_JOURNAL_PAYLOAD_LIMITS),
|
||||
state: failure === null ? commandState(item) : 'failed',
|
||||
...(bounded === null ? {} : { output: bounded.bounded })
|
||||
},
|
||||
handled: true
|
||||
}
|
||||
}
|
||||
|
||||
/** A row label is read off top-level keys only, so the action's own labelable
|
||||
* fields are hoisted beside the query while `action` stays whole for the
|
||||
* expanded detail. The action `type` lands on `description`, the lowest-ranked
|
||||
* label key, so it names only an action that carries nothing better. */
|
||||
function webSearchInput(item: CodexThreadItem): Record<string, unknown> | null {
|
||||
const action = readRecord(item.action)
|
||||
const fields: [string, unknown][] = [
|
||||
['url', readString(action, 'url')],
|
||||
['pattern', readString(action, 'pattern')],
|
||||
['description', readString(action, 'type')],
|
||||
['action', item.action ?? null]
|
||||
]
|
||||
const query = readString(item, 'query') ?? readString(action, 'query')
|
||||
const present = fields.filter(([, value]) => value !== null)
|
||||
// A blank `query` is the run header's "this call has no brief argument"
|
||||
// signal; drop the key and the header stands the row's raw JSON in for one.
|
||||
return query === null && present.length === 0
|
||||
? null
|
||||
: { query: query ?? '', ...Object.fromEntries(present) }
|
||||
}
|
||||
|
||||
/** `webSearch` carries no status: Codex starts it with an empty query and a null
|
||||
* action, then sends the action, so `action` is the completion signal — a
|
||||
* completed item's own `query` is routinely still empty. The hits arrive on
|
||||
* `results` and are the call's output. */
|
||||
function webSearchItem(item: CodexThreadItem): CodexJournalItem {
|
||||
const hits = Array.isArray(item.results) && item.results.length > 0 ? item.results : null
|
||||
const bounded = hits && boundInlineText(JSON.stringify(hits), DEFAULT_JOURNAL_PAYLOAD_LIMITS)
|
||||
return {
|
||||
body: {
|
||||
kind: 'tool-call',
|
||||
name: 'web_search',
|
||||
input: boundToolInput(webSearchInput(item), DEFAULT_JOURNAL_PAYLOAD_LIMITS),
|
||||
state: item.action === null || item.action === undefined ? 'running' : 'completed',
|
||||
...(bounded === null ? {} : { output: bounded.bounded })
|
||||
},
|
||||
handled: true
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Journal body for a Codex item, or null for one with nothing to render.
|
||||
*
|
||||
@@ -319,6 +237,12 @@ export function codexJournalItem(item: CodexThreadItem): CodexJournalItem {
|
||||
if (item.type === 'fileChange') {
|
||||
return fileChangeItem(item)
|
||||
}
|
||||
if (item.type === 'mcpToolCall') {
|
||||
return mcpToolCallItem(item)
|
||||
}
|
||||
if (item.type === 'webSearch') {
|
||||
return webSearchItem(item)
|
||||
}
|
||||
if (item.type === 'reasoning' || item.type === 'plan') {
|
||||
const text =
|
||||
readTextContent(item, 'text') ??
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types'
|
||||
import type { CodexTurnOrdinals } from './codex-turn-ordinals'
|
||||
|
||||
// Codex thread items → durable journal identities.
|
||||
//
|
||||
// THE ORDINAL RULE, and why it is not "index within the turn". Codex renumbers
|
||||
// item ids positionally on resume (`item-1`…`item-N` across the whole thread),
|
||||
// and a resumed turn does NOT contain every item the live turn emitted —
|
||||
// reasoning and command execution are dropped from persisted history. Numbering
|
||||
// by live position would therefore shift every message after the first tool
|
||||
// call and hand the user a duplicate of the assistant's answer after a resume.
|
||||
//
|
||||
// So the ordinal counts MESSAGE items only, and the same projection is applied
|
||||
// to the live stream and to a resumed turn's item list. Any other item type —
|
||||
// including ones this build does not model — is skipped identically on both
|
||||
// sides, which is what makes the key survive a Codex release that adds one.
|
||||
|
||||
/** Only these carry a durable `(threadId, turnId, ordinal)` identity. */
|
||||
const CODEX_MESSAGE_ITEM_TYPES = new Set(['userMessage', 'agentMessage'])
|
||||
|
||||
export type CodexThreadItem = {
|
||||
type: string
|
||||
id: string
|
||||
[key: string]: unknown
|
||||
}
|
||||
|
||||
export function isCodexMessageItemType(type: string): boolean {
|
||||
return CODEX_MESSAGE_ITEM_TYPES.has(type)
|
||||
}
|
||||
|
||||
export function readCodexThreadItem(value: unknown): CodexThreadItem | null {
|
||||
if (typeof value !== 'object' || value === null) {
|
||||
return null
|
||||
}
|
||||
const record = value as Record<string, unknown>
|
||||
return typeof record.type === 'string' && typeof record.id === 'string'
|
||||
? (record as CodexThreadItem)
|
||||
: null
|
||||
}
|
||||
|
||||
/**
|
||||
* Durable identity for a Codex item, or null for one that has none.
|
||||
*
|
||||
* Non-message items fall back to the `orca` namespace keyed by the Codex item
|
||||
* id. That id is unstable across resume, so those rows are live-session detail
|
||||
* that a recovered journal simply will not contain — which is correct: Codex
|
||||
* itself does not persist them either.
|
||||
*/
|
||||
export function codexItemIdentity(input: {
|
||||
threadId: string
|
||||
turnId: string | null
|
||||
item: CodexThreadItem
|
||||
ordinals: CodexTurnOrdinals
|
||||
}): AgentJournalItemIdentity {
|
||||
const { item, turnId } = input
|
||||
if (turnId && isCodexMessageItemType(item.type)) {
|
||||
return {
|
||||
provider: 'codex',
|
||||
threadId: input.threadId,
|
||||
turnId,
|
||||
ordinal: input.ordinals.ordinalFor(input.threadId, turnId, item.id)
|
||||
}
|
||||
}
|
||||
return { provider: 'orca', clientMessageId: `codex-item:${input.threadId}:${item.id}` }
|
||||
}
|
||||
@@ -21,7 +21,7 @@ describe('getBranchConflictKindViaExec', () => {
|
||||
|
||||
await expect(getBranchConflictKindViaExec(exec, 'feature/fix')).resolves.toBe('remote')
|
||||
expect(calls).toEqual([
|
||||
['rev-parse', '--verify', 'refs/heads/feature/fix'],
|
||||
['rev-parse', '--verify', '--quiet', 'refs/heads/feature/fix'],
|
||||
['remote'],
|
||||
['show-ref', '--verify', '--quiet', '--', 'refs/remotes/foo/bar/feature/fix'],
|
||||
['show-ref', '--verify', '--quiet', '--', 'refs/remotes/origin/feature/fix']
|
||||
@@ -41,7 +41,10 @@ describe('getBranchConflictKindViaExec', () => {
|
||||
await expect(
|
||||
getBranchConflictKindViaExec(exec, 'feature/fix', 'origin/feature/fix')
|
||||
).resolves.toBeNull()
|
||||
expect(calls).toEqual([['rev-parse', '--verify', 'refs/heads/feature/fix'], ['remote']])
|
||||
expect(calls).toEqual([
|
||||
['rev-parse', '--verify', '--quiet', 'refs/heads/feature/fix'],
|
||||
['remote']
|
||||
])
|
||||
})
|
||||
|
||||
it('keeps longest configured remote-name matching semantics', async () => {
|
||||
@@ -176,7 +179,7 @@ describe('getBranchConflictKindViaExec batched remote probe', () => {
|
||||
getBranchConflictKindViaExec(exec, 'feature', undefined, {}, batched)
|
||||
).resolves.toBeNull()
|
||||
expect(calls).toEqual([
|
||||
['rev-parse', '--verify', 'refs/heads/feature'],
|
||||
['rev-parse', '--verify', '--quiet', 'refs/heads/feature'],
|
||||
['remote'],
|
||||
['cat-file', '--batch-check']
|
||||
])
|
||||
@@ -254,3 +257,68 @@ describe('getBranchConflictKindViaExec batched remote probe', () => {
|
||||
expect(calls.filter((argv) => argv[0] === 'show-ref')).toHaveLength(3)
|
||||
})
|
||||
})
|
||||
|
||||
describe('branch conflict with existing-branch adoption', () => {
|
||||
const absent = () => Object.assign(new Error('missing'), { code: 1, stderr: '' })
|
||||
|
||||
it('skips adoption and its commit probe for a proven missing local ref', async () => {
|
||||
const exec = vi.fn(async (argv: string[]) => {
|
||||
if (argv[0] === 'rev-parse') {
|
||||
throw absent()
|
||||
}
|
||||
return { stdout: '' }
|
||||
})
|
||||
const adopt = vi.fn(async () => false)
|
||||
await expect(
|
||||
getBranchConflictKindViaExec(exec, 'new', undefined, {}, undefined, adopt)
|
||||
).resolves.toBeNull()
|
||||
expect(adopt).not.toHaveBeenCalled()
|
||||
expect(exec).toHaveBeenCalledTimes(2)
|
||||
})
|
||||
|
||||
it('allows an existing branch without querying remote refs', async () => {
|
||||
const exec = vi.fn(async () => ({ stdout: 'a'.repeat(40) }))
|
||||
const adopt = vi.fn(async () => true)
|
||||
await expect(
|
||||
getBranchConflictKindViaExec(exec, 'existing', undefined, {}, undefined, adopt)
|
||||
).resolves.toBeNull()
|
||||
expect(adopt).toHaveBeenCalledOnce()
|
||||
expect(exec).toHaveBeenCalledOnce()
|
||||
})
|
||||
|
||||
it('retains conflicts for refs whose objects cannot be adopted as commits', async () => {
|
||||
const exec = vi.fn(async () => ({ stdout: 'a'.repeat(40) }))
|
||||
const adopt = vi.fn(async () => false)
|
||||
await expect(
|
||||
getBranchConflictKindViaExec(exec, 'dangling', undefined, {}, undefined, adopt)
|
||||
).resolves.toBe('local')
|
||||
expect(adopt).toHaveBeenCalledOnce()
|
||||
expect(exec).toHaveBeenCalledTimes(2)
|
||||
})
|
||||
|
||||
it.each([
|
||||
Object.assign(new Error('transport'), { code: 1, stderr: 'transport failed' }),
|
||||
Object.assign(new Error('timeout'), { code: 'ETIMEDOUT' })
|
||||
])('still attempts adoption after an undecided ref probe: %s', async (error) => {
|
||||
const exec = vi.fn(async () => {
|
||||
throw error
|
||||
})
|
||||
const adopt = vi.fn(async () => true)
|
||||
await expect(
|
||||
getBranchConflictKindViaExec(exec, 'existing', undefined, {}, undefined, adopt)
|
||||
).resolves.toBeNull()
|
||||
expect(adopt).toHaveBeenCalledOnce()
|
||||
})
|
||||
|
||||
it('rechecks a ref that disappeared while adoption was running', async () => {
|
||||
const exec = vi
|
||||
.fn()
|
||||
.mockResolvedValueOnce({ stdout: 'a'.repeat(40) })
|
||||
.mockRejectedValueOnce(absent())
|
||||
.mockResolvedValueOnce({ stdout: '' })
|
||||
await expect(
|
||||
getBranchConflictKindViaExec(exec, 'removed', undefined, {}, undefined, async () => false)
|
||||
).resolves.toBeNull()
|
||||
expect(exec).toHaveBeenCalledTimes(3)
|
||||
})
|
||||
})
|
||||
|
||||
@@ -3,6 +3,7 @@ import { gitExecOptions, type LocalGitExecOptions } from './repo-default-base-re
|
||||
import { gitExecFileAsync } from './runner'
|
||||
import { isSafeGitRefName } from '../../shared/git-status-upstream-ref'
|
||||
import {
|
||||
isShowRefNoMatchError,
|
||||
probeAnyExactRef,
|
||||
probeAnyExactRefBatched,
|
||||
type ExactRefProbeExec,
|
||||
@@ -29,16 +30,17 @@ function canQueryRemoteBranchName(branchName: string): boolean {
|
||||
return !branchName.startsWith('-') && isSafeGitRefName(`refs/heads/${branchName}`)
|
||||
}
|
||||
|
||||
async function hasGitRefAsync(
|
||||
async function probeLocalBranchRef(
|
||||
exec: ExactRefProbeExec,
|
||||
ref: string,
|
||||
options: ExactRefProbeExecOptions
|
||||
): Promise<boolean> {
|
||||
): Promise<'present' | 'absent' | 'unknown'> {
|
||||
try {
|
||||
const { stdout } = await runGit(exec, ['rev-parse', '--verify', ref], options)
|
||||
return stdout.trim().length > 0
|
||||
} catch {
|
||||
return false
|
||||
// Quiet absence avoids retrying the WSL probe through a login shell.
|
||||
const { stdout } = await runGit(exec, ['rev-parse', '--verify', '--quiet', ref], options)
|
||||
return stdout.trim().length > 0 ? 'present' : 'unknown'
|
||||
} catch (error) {
|
||||
return isShowRefNoMatchError(error) ? 'absent' : 'unknown'
|
||||
}
|
||||
}
|
||||
|
||||
@@ -105,7 +107,8 @@ export async function getBranchConflictKindViaExec(
|
||||
branchName: string,
|
||||
allowedBaseRef?: string,
|
||||
options: ExactRefProbeExecOptions = {},
|
||||
batchedExec?: ExactRefProbeStdinExec
|
||||
batchedExec?: ExactRefProbeStdinExec,
|
||||
allowLocalBranch?: () => Promise<boolean>
|
||||
): Promise<BranchConflictKind | null> {
|
||||
if (!canQueryRemoteBranchName(branchName)) {
|
||||
return null
|
||||
@@ -114,7 +117,16 @@ export async function getBranchConflictKindViaExec(
|
||||
// are quiet, so introducing a smaller implicit cap would only make a large
|
||||
// remote configuration look like a missing conflict.
|
||||
const probeOptions: ExactRefProbeExecOptions = options
|
||||
if (await hasGitRefAsync(exec, `refs/heads/${branchName}`, probeOptions)) {
|
||||
const localRef = `refs/heads/${branchName}`
|
||||
let presence = await probeLocalBranchRef(exec, localRef, probeOptions)
|
||||
if (allowLocalBranch && presence !== 'absent') {
|
||||
if (await allowLocalBranch()) {
|
||||
return null
|
||||
}
|
||||
// Adoption can span ref changes; preserve the fresh conflict check after it fails.
|
||||
presence = await probeLocalBranchRef(exec, localRef, probeOptions)
|
||||
}
|
||||
if (presence === 'present') {
|
||||
return 'local'
|
||||
}
|
||||
|
||||
@@ -142,7 +154,8 @@ export function getBranchConflictKind(
|
||||
path: string,
|
||||
branchName: string,
|
||||
allowedBaseRef?: string,
|
||||
options: LocalGitExecOptions = {}
|
||||
options: LocalGitExecOptions = {},
|
||||
allowLocalBranch?: () => Promise<boolean>
|
||||
): Promise<BranchConflictKind | null> {
|
||||
const execOptions = gitExecOptions(path, options)
|
||||
const runLocalGit = (
|
||||
@@ -168,7 +181,8 @@ export function getBranchConflictKind(
|
||||
// one `show-ref` subprocess per remote -- the exact cost the batch exists to remove.
|
||||
// `show-ref --verify --quiet` prints nothing and is read by exit code, so it needs
|
||||
// no fence; the capture wrapper preserves the payload's exit status either way.
|
||||
(argv, commandOptions) => runLocalGit(argv, commandOptions, true)
|
||||
(argv, commandOptions) => runLocalGit(argv, commandOptions, true),
|
||||
allowLocalBranch
|
||||
)
|
||||
}
|
||||
|
||||
|
||||
@@ -17,6 +17,7 @@ vi.mock('../observability/instrumentation', () => ({
|
||||
}))
|
||||
vi.mock('../diagnostics/main-thread-churn-probe', () => ({ recordSubprocessSpawn: vi.fn() }))
|
||||
|
||||
import { getBranchConflictKind } from './repo-branch-conflict'
|
||||
import { pendingWslDirectGitReadEnvironment } from './command-runner/git-command-resolution'
|
||||
import { gitExecFileAsync, gitSpawn, gitStreamStdout } from './runner'
|
||||
import {
|
||||
@@ -584,6 +585,33 @@ describe('WSL direct Git reads', () => {
|
||||
})
|
||||
})
|
||||
|
||||
it('checks a missing branch conflict without retrying through a login shell', async () => {
|
||||
await withPlatform('win32', async () => {
|
||||
seedWslGitReadEnvironmentForTests(DISTRO, LOGIN_ENVIRONMENT)
|
||||
execFileMock.mockImplementation((_command, args: string[], _options, callback) => {
|
||||
const child = createMockChild()
|
||||
queueMicrotask(() => {
|
||||
const missingRef = args.join(' ').includes('rev-parse')
|
||||
const quiet = args.includes('--quiet')
|
||||
const code = missingRef ? (quiet ? 1 : 128) : 0
|
||||
callback?.(
|
||||
code ? Object.assign(new Error('missing ref'), { code }) : null,
|
||||
'',
|
||||
missingRef && !quiet ? 'fatal: Needed a single revision' : ''
|
||||
)
|
||||
child.emit('close', code, null)
|
||||
})
|
||||
return child
|
||||
})
|
||||
|
||||
await expect(
|
||||
getBranchConflictKind(String.raw`\\wsl.localhost\Ubuntu\repo`, 'new-feature')
|
||||
).resolves.toBeNull()
|
||||
expect(execFileMock).toHaveBeenCalledTimes(2)
|
||||
expect(execFileMock.mock.calls[0]?.[1]).toContain('--quiet')
|
||||
})
|
||||
})
|
||||
|
||||
it('keeps the fast path when direct and login Git both report an expected failure', async () => {
|
||||
await withPlatform('win32', async () => {
|
||||
seedWslGitReadEnvironmentForTests(DISTRO, LOGIN_ENVIRONMENT)
|
||||
|
||||
@@ -199,7 +199,7 @@ describe('addWorktree', () => {
|
||||
})
|
||||
|
||||
const worktreeAddCall = gitExecFileAsyncMock.mock.calls.find(
|
||||
([argv]) => Array.isArray(argv) && argv[0] === 'worktree' && argv[1] === 'add'
|
||||
([argv]) => Array.isArray(argv) && argv.includes('worktree') && argv.includes('add')
|
||||
)
|
||||
expect(worktreeAddCall?.[1]).toMatchObject({ timeout: WORKTREE_ADD_TIMEOUT_MS })
|
||||
expect(WORKTREE_ADD_TIMEOUT_MS).toBeGreaterThan(0)
|
||||
@@ -214,7 +214,7 @@ describe('addWorktree', () => {
|
||||
})
|
||||
|
||||
const worktreeAddCall = gitExecFileAsyncMock.mock.calls.find(
|
||||
([argv]) => Array.isArray(argv) && argv[0] === 'worktree' && argv[1] === 'add'
|
||||
([argv]) => Array.isArray(argv) && argv.includes('worktree') && argv.includes('add')
|
||||
)
|
||||
expect(worktreeAddCall?.[1]).toMatchObject({ timeout: 600_000 })
|
||||
})
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
// addWorktree: fast-forwarding the local base ref (reset --hard / update-ref) and its safety bailouts.
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
const {
|
||||
gitExecFileAsyncMock,
|
||||
@@ -32,11 +32,14 @@ import { registerWorktreeSuiteHooks } from './worktree-test-harness'
|
||||
registerWorktreeSuiteHooks()
|
||||
|
||||
describe('addWorktree', () => {
|
||||
afterEach(() => vi.restoreAllMocks())
|
||||
const resolveCreationBaseConfigWrite = () => {
|
||||
gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: '' }) // config --local --replace-all branch.<branch>.base
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
// These branch-safety assertions use POSIX argv; Windows flags have separate coverage.
|
||||
vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin')
|
||||
gitExecFileAsyncMock.mockReset()
|
||||
gitExecFileSyncMock.mockReset()
|
||||
translateWslOutputPathsMock.mockClear()
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
// addWorktree: advisory local-base-ref update suggestions when the refresh setting is off.
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
const {
|
||||
gitExecFileAsyncMock,
|
||||
@@ -32,7 +32,10 @@ import { registerWorktreeSuiteHooks } from './worktree-test-harness'
|
||||
registerWorktreeSuiteHooks()
|
||||
|
||||
describe('addWorktree', () => {
|
||||
afterEach(() => vi.restoreAllMocks())
|
||||
beforeEach(() => {
|
||||
// These branch-safety assertions use POSIX argv; Windows flags have separate coverage.
|
||||
vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin')
|
||||
gitExecFileAsyncMock.mockReset()
|
||||
gitExecFileSyncMock.mockReset()
|
||||
translateWslOutputPathsMock.mockClear()
|
||||
|
||||
@@ -12,7 +12,7 @@ import {
|
||||
getLocalBaseRefUpdateSuggestionForWorktreeCreate,
|
||||
refreshLocalBaseRefForWorktreeCreate
|
||||
} from './worktree-base-refresh'
|
||||
import { hasWorktreeBaseCommitRef } from './worktree-base-ref-probe'
|
||||
import { resolveWorktreeBaseCommitOid } from './worktree-base-ref-probe'
|
||||
import type {
|
||||
AddWorktreeOptions,
|
||||
AddWorktreeResult,
|
||||
@@ -23,6 +23,7 @@ import { bumpWorktreeScanGeneration } from './worktree-scan-cache'
|
||||
|
||||
export type WorktreeAddBaseContext = AddWorktreeResult & {
|
||||
effectiveBase: string
|
||||
effectiveBaseOid?: string
|
||||
}
|
||||
|
||||
export async function resolveWorktreeAddBaseContext(
|
||||
@@ -31,9 +32,11 @@ export async function resolveWorktreeAddBaseContext(
|
||||
refreshLocalBaseRef: boolean,
|
||||
options: AddWorktreeOptions
|
||||
): Promise<WorktreeAddBaseContext> {
|
||||
const effectiveBase = await resolveWorktreeAddBaseRef(baseBranch, (qualifiedRef) =>
|
||||
hasWorktreeBaseCommitRef(repoPath, qualifiedRef, options)
|
||||
)
|
||||
let effectiveBaseOid: string | null = null
|
||||
const effectiveBase = await resolveWorktreeAddBaseRef(baseBranch, async (qualifiedRef) => {
|
||||
effectiveBaseOid = await resolveWorktreeBaseCommitOid(repoPath, qualifiedRef, options)
|
||||
return effectiveBaseOid !== null
|
||||
})
|
||||
const localBaseRefRefresh = refreshLocalBaseRef
|
||||
? await refreshLocalBaseRefForWorktreeCreate(
|
||||
repoPath,
|
||||
@@ -55,6 +58,10 @@ export async function resolveWorktreeAddBaseContext(
|
||||
: undefined
|
||||
return {
|
||||
effectiveBase,
|
||||
// Refresh/suggestion work can span ref changes; only reuse the immediate resolution probe.
|
||||
...(!refreshLocalBaseRef && !options.suggestLocalBaseRefUpdate && effectiveBaseOid
|
||||
? { effectiveBaseOid }
|
||||
: {}),
|
||||
...(localBaseRefRefresh ? { localBaseRefRefresh } : {}),
|
||||
...(localBaseRefUpdateSuggestion ? { localBaseRefUpdateSuggestion } : {})
|
||||
}
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
import { mkdir, mkdtemp, readFile, rm, writeFile } from 'node:fs/promises'
|
||||
import { join } from 'node:path'
|
||||
import { expect, it } from 'vitest'
|
||||
import { createWorktreePreparationLockReason } from '../../shared/worktree/create-preparation'
|
||||
import { gitExecFileAsync } from './runner'
|
||||
import {
|
||||
discardPreparedWorktree,
|
||||
finalizePreparedWorktree,
|
||||
prepareWorktreeCreateCheckout
|
||||
} from './worktree-create-preparation'
|
||||
|
||||
// Opt in on Windows with a running distro; all Git commands use the production WSL router.
|
||||
const wslDistro = process.env.ORCA_TEST_WSL_DISTRO
|
||||
|
||||
it.skipIf(process.platform !== 'win32' || !wslDistro)(
|
||||
'prepares, retargets, moves and cleans up a real WSL checkout from Windows',
|
||||
async () => {
|
||||
const fixtureParent = process.env.ORCA_TEST_WSL_ROOT ?? `\\\\wsl.localhost\\${wslDistro}\\tmp`
|
||||
const root = await mkdtemp(join(fixtureParent, 'orca-create-route-'))
|
||||
const repoPath = join(root, 'repo')
|
||||
const preparedPath = join(root, 'prepared checkout')
|
||||
const finalPath = join(root, 'final checkout')
|
||||
const options = { wslDistro, timeout: 60_000 }
|
||||
const git = async (cwd: string, args: string[]): Promise<string> =>
|
||||
(await gitExecFileAsync(args, { cwd, ...options })).stdout.trim()
|
||||
|
||||
try {
|
||||
await mkdir(repoPath)
|
||||
await git(repoPath, ['init', '--quiet'])
|
||||
expect(await git(repoPath, ['rev-parse', '--show-toplevel'])).toMatch(/^\/(?!\/)/)
|
||||
await git(repoPath, ['symbolic-ref', 'HEAD', 'refs/heads/main'])
|
||||
await git(repoPath, ['config', 'user.name', 'Test User'])
|
||||
await git(repoPath, ['config', 'user.email', 'test@example.com'])
|
||||
await writeFile(join(repoPath, 'version.txt'), 'one\n')
|
||||
await git(repoPath, ['add', 'version.txt'])
|
||||
await git(repoPath, ['commit', '--quiet', '-m', 'initial'])
|
||||
await prepareWorktreeCreateCheckout(
|
||||
repoPath,
|
||||
preparedPath,
|
||||
'main',
|
||||
createWorktreePreparationLockReason('real-wsl-test'),
|
||||
options
|
||||
)
|
||||
expect(await git(repoPath, ['worktree', 'list', '--porcelain'])).toContain(
|
||||
'locked orca-create-preparation:v1:'
|
||||
)
|
||||
|
||||
await writeFile(join(repoPath, 'version.txt'), 'two\n')
|
||||
await git(repoPath, ['commit', '--quiet', '-am', 'advance base'])
|
||||
const target = await git(repoPath, ['rev-parse', 'HEAD'])
|
||||
await finalizePreparedWorktree(
|
||||
repoPath,
|
||||
preparedPath,
|
||||
finalPath,
|
||||
'feature/routed',
|
||||
'main',
|
||||
false,
|
||||
options
|
||||
)
|
||||
expect(await git(finalPath, ['rev-parse', 'HEAD'])).toBe(target)
|
||||
expect(await git(finalPath, ['symbolic-ref', '--short', 'HEAD'])).toBe('feature/routed')
|
||||
expect(await git(finalPath, ['status', '--porcelain'])).toBe('')
|
||||
expect(await readFile(join(finalPath, 'version.txt'), 'utf8')).toBe('two\n')
|
||||
expect(await git(finalPath, ['config', '--get', 'branch.feature/routed.base'])).toBe(
|
||||
'refs/heads/main'
|
||||
)
|
||||
expect(await git(repoPath, ['worktree', 'list', '--porcelain'])).not.toContain('locked ')
|
||||
await discardPreparedWorktree(repoPath, finalPath, options)
|
||||
expect(
|
||||
(await git(repoPath, ['worktree', 'list', '--porcelain'])).match(/^worktree /gm)
|
||||
).toHaveLength(1)
|
||||
} finally {
|
||||
await rm(root, { recursive: true, force: true })
|
||||
}
|
||||
},
|
||||
120_000
|
||||
)
|
||||
@@ -195,25 +195,38 @@ export async function finalizePreparedWorktree(
|
||||
}
|
||||
try {
|
||||
return await runWithGitReadCacheInvalidation(async () => {
|
||||
const baseContext = await resolveWorktreeAddBaseContext(
|
||||
repoPath,
|
||||
baseBranch,
|
||||
refreshLocalBaseRef,
|
||||
finalizeGitOptions
|
||||
)
|
||||
const [targetHeadResult, preparedHeadResult] = await Promise.all([
|
||||
gitExecFileAsync(
|
||||
['rev-parse', '--verify', `${baseContext.effectiveBase}^{commit}`],
|
||||
gitExecOptions(repoPath, finalizeGitOptions)
|
||||
),
|
||||
const [targetResult, preparedResult] = await Promise.allSettled([
|
||||
(async () => {
|
||||
const baseContext = await resolveWorktreeAddBaseContext(
|
||||
repoPath,
|
||||
baseBranch,
|
||||
refreshLocalBaseRef,
|
||||
finalizeGitOptions
|
||||
)
|
||||
const targetHead =
|
||||
baseContext.effectiveBaseOid ??
|
||||
(
|
||||
await gitExecFileAsync(
|
||||
['rev-parse', '--verify', `${baseContext.effectiveBase}^{commit}`],
|
||||
gitExecOptions(repoPath, finalizeGitOptions)
|
||||
)
|
||||
).stdout.trim()
|
||||
return { baseContext, targetHead }
|
||||
})(),
|
||||
gitExecFileAsync(
|
||||
['rev-parse', '--verify', 'HEAD'],
|
||||
gitExecOptions(preparedPath, finalizeGitOptions)
|
||||
)
|
||||
])
|
||||
const { stdout: targetHeadOutput } = targetHeadResult
|
||||
const targetHead = targetHeadOutput.trim()
|
||||
const { stdout: preparedHeadOutput } = preparedHeadResult
|
||||
// Settle both reads before failure cleanup can remove the prepared checkout.
|
||||
if (targetResult.status === 'rejected') {
|
||||
throw targetResult.reason
|
||||
}
|
||||
if (preparedResult.status === 'rejected') {
|
||||
throw preparedResult.reason
|
||||
}
|
||||
const { baseContext, targetHead } = targetResult.value
|
||||
const preparedHeadOutput = preparedResult.value.stdout
|
||||
if (preparedHeadOutput.trim() !== targetHead) {
|
||||
await gitExecFileAsync(
|
||||
[...windowsLongPathGitArgs(preparedPath), 'reset', '--hard', targetHead],
|
||||
|
||||
@@ -0,0 +1,98 @@
|
||||
import { beforeEach, expect, it, vi } from 'vitest'
|
||||
|
||||
const gitExec = vi.hoisted(() => vi.fn())
|
||||
vi.mock('./runner', () => ({ gitExecFileAsync: gitExec }))
|
||||
vi.mock('./worktree-base-refresh', () => ({
|
||||
refreshLocalBaseRefForWorktreeCreate: vi.fn(),
|
||||
getLocalBaseRefUpdateSuggestionForWorktreeCreate: vi.fn()
|
||||
}))
|
||||
vi.mock('./status', () => ({ runWithGitReadCacheInvalidation: (run: () => unknown) => run() }))
|
||||
vi.mock('./wsl-linked-worktree-git-routing', () => ({
|
||||
invalidateWslLinkedWorktreeGitRouting: vi.fn()
|
||||
}))
|
||||
|
||||
import { finalizePreparedWorktree } from './worktree-create-preparation'
|
||||
|
||||
const originalOid = '1'.repeat(40)
|
||||
const refreshedOid = '2'.repeat(40)
|
||||
|
||||
beforeEach(() => {
|
||||
gitExec.mockReset().mockImplementation(async (args: string[]) => ({
|
||||
stdout:
|
||||
args[0] === 'rev-parse'
|
||||
? args.includes('--quiet') || args.at(-1) === 'HEAD'
|
||||
? originalOid
|
||||
: refreshedOid
|
||||
: ''
|
||||
}))
|
||||
})
|
||||
|
||||
it('reuses the current base-resolution oid and preserves WSL routing', async () => {
|
||||
await finalizePreparedWorktree('/repo', '/prepared', '/final', 'feature', 'main', false, {
|
||||
wslDistro: 'Ubuntu',
|
||||
timeout: 8000
|
||||
})
|
||||
const revisions = gitExec.mock.calls.filter(([args]) => args[0] === 'rev-parse')
|
||||
expect(revisions.map(([args]) => args)).toEqual([
|
||||
['rev-parse', '--verify', '--quiet', 'refs/heads/main^{commit}'],
|
||||
['rev-parse', '--verify', 'HEAD']
|
||||
])
|
||||
expect(gitExec.mock.calls.find(([args]) => args.includes('checkout'))?.[0]).toContain(originalOid)
|
||||
expect(gitExec.mock.calls.some(([args]) => args.includes('reset'))).toBe(false)
|
||||
for (const [, options] of gitExec.mock.calls) {
|
||||
expect(options).toMatchObject({ wslDistro: 'Ubuntu', timeout: 8000 })
|
||||
}
|
||||
})
|
||||
|
||||
it.each([
|
||||
{ base: 'refs/heads/main', refresh: false, options: {} },
|
||||
{ base: 'main', refresh: true, options: {} },
|
||||
{ base: 'main', refresh: false, options: { suggestLocalBaseRefUpdate: true } }
|
||||
])('re-reads the target for $base, refresh=$refresh, options=$options', async (test) => {
|
||||
await finalizePreparedWorktree(
|
||||
'/repo',
|
||||
'/prepared',
|
||||
'/final',
|
||||
'feature',
|
||||
test.base,
|
||||
test.refresh,
|
||||
test.options
|
||||
)
|
||||
expect(gitExec).toHaveBeenCalledWith(
|
||||
['rev-parse', '--verify', 'refs/heads/main^{commit}'],
|
||||
expect.objectContaining({ cwd: '/repo' })
|
||||
)
|
||||
expect(gitExec.mock.calls.find(([args]) => args.includes('reset'))?.[0]).toContain(refreshedOid)
|
||||
expect(gitExec.mock.calls.find(([args]) => args.includes('checkout'))?.[0]).toContain(
|
||||
refreshedOid
|
||||
)
|
||||
})
|
||||
|
||||
it('starts both independent probes before either resolves and settles them before failure', async () => {
|
||||
let resolveBase!: (value: { stdout: string }) => void
|
||||
let rejectPrepared!: (reason: Error) => void
|
||||
gitExec.mockImplementation((args: string[]) => {
|
||||
if (args.includes('--quiet')) {
|
||||
return new Promise((resolve) => (resolveBase = resolve))
|
||||
}
|
||||
if (args.at(-1) === 'HEAD') {
|
||||
return new Promise((_, reject) => (rejectPrepared = reject))
|
||||
}
|
||||
return Promise.resolve({ stdout: '' })
|
||||
})
|
||||
let settled = false
|
||||
const error = new Error('prepared HEAD unreadable')
|
||||
const result = finalizePreparedWorktree('/repo', '/prepared', '/final', 'feature', 'main')
|
||||
const checked = expect(result).rejects.toBe(error)
|
||||
void result.then(
|
||||
() => (settled = true),
|
||||
() => (settled = true)
|
||||
)
|
||||
await vi.waitFor(() => expect(gitExec).toHaveBeenCalledTimes(2))
|
||||
rejectPrepared(error)
|
||||
await Promise.resolve()
|
||||
expect(settled).toBe(false)
|
||||
resolveBase({ stdout: originalOid })
|
||||
await checked
|
||||
expect(gitExec.mock.calls.some(([args]) => args.includes('move'))).toBe(false)
|
||||
})
|
||||
@@ -16,6 +16,7 @@ vi.mock('../wsl', () => ({
|
||||
import {
|
||||
computeWorktreePath,
|
||||
computeWorktreePathAsync,
|
||||
computeWorkspaceRootAsync,
|
||||
getWorktreePathSettings
|
||||
} from './worktree-logic'
|
||||
import {
|
||||
@@ -32,6 +33,26 @@ describe('computeWorktreePath WSL layout', () => {
|
||||
parseWslPathMock.mockReset()
|
||||
})
|
||||
|
||||
it('reuses an asynchronously resolved root for every name candidate without a sync probe', async () => {
|
||||
parseWslPathMock.mockReturnValue({ distro: 'Ubuntu', linuxPath: '/home/jin/repo' })
|
||||
const repoPath = String.raw`\\wsl.localhost\Ubuntu\home\jin\repo`
|
||||
const home = String.raw`\\wsl.localhost\Ubuntu\home\jin`
|
||||
const settings = { workspaceDir: 'C:\\workspaces', nestWorkspaces: true }
|
||||
let resolveHome!: (home: string) => void
|
||||
getWslHomeAsyncMock.mockReturnValue(new Promise<string>((resolve) => (resolveHome = resolve)))
|
||||
const pendingRoot = computeWorkspaceRootAsync(repoPath, settings)
|
||||
expect(getWslHomeMock).not.toHaveBeenCalled()
|
||||
resolveHome(home)
|
||||
const root = await pendingRoot
|
||||
for (const name of ['feature', 'feature-2', 'feature-3']) {
|
||||
expect(computeWorktreePath(name, repoPath, settings, root)).toBe(
|
||||
win32.join(home, 'orca', 'workspaces', 'repo', name)
|
||||
)
|
||||
}
|
||||
expect(getWslHomeAsyncMock).toHaveBeenCalledExactlyOnceWith('Ubuntu')
|
||||
expect(getWslHomeMock).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('places WSL repo worktrees under the distro home workspace root', () => {
|
||||
parseWslPathMock.mockReturnValue({
|
||||
distro: 'Ubuntu',
|
||||
|
||||
@@ -103,12 +103,13 @@ export function ensurePathWithinWorkspace(targetPath: string, workspaceDir: stri
|
||||
export function computeWorktreePath(
|
||||
sanitizedName: string,
|
||||
repoPath: string,
|
||||
settings: WorktreePathSettings
|
||||
settings: WorktreePathSettings,
|
||||
workspaceRoot?: string
|
||||
): string {
|
||||
return computeWorktreePathFromWorkspaceRoot(
|
||||
sanitizedName,
|
||||
repoPath,
|
||||
computeWorkspaceRoot(repoPath, settings),
|
||||
workspaceRoot ?? computeWorkspaceRoot(repoPath, settings),
|
||||
settings.nestWorkspaces
|
||||
)
|
||||
}
|
||||
@@ -130,7 +131,7 @@ function computeWorktreePathFromWorkspaceRoot(
|
||||
}
|
||||
|
||||
/** Async twin of computeWorktreePath. Same result; resolves the WSL home without blocking the main
|
||||
* thread, so callers off the create path never freeze the app on a stopped distro. */
|
||||
* thread, so callers never freeze the app on a stopped distro. */
|
||||
export async function computeWorktreePathAsync(
|
||||
sanitizedName: string,
|
||||
repoPath: string,
|
||||
@@ -147,7 +148,7 @@ export async function computeWorktreePathAsync(
|
||||
/** Async twin of computeWorkspaceRoot. Same result; the WSL home probe spawns `wsl.exe`, so
|
||||
* background preparation uses this variant rather than blocking the Electron main thread for up
|
||||
* to the probe timeout. The sync twin below still serves callers that cannot await (allowed-roots
|
||||
* resolution, the create click, CLI create, watch targets, worktree trash). */
|
||||
* resolution, CLI create, watch targets, worktree trash). */
|
||||
export async function computeWorkspaceRootAsync(
|
||||
repoPath: string,
|
||||
settings: { workspaceDir: string; wslMirrorDistro?: string }
|
||||
|
||||
@@ -87,7 +87,7 @@ import {
|
||||
computeValidatedBranchName,
|
||||
computeWorktreePath,
|
||||
computeRemoteWorktreePath,
|
||||
computeWorkspaceRoot,
|
||||
computeWorkspaceRootAsync,
|
||||
ensurePathWithinWorkspace,
|
||||
getWorktreeCreationLayout,
|
||||
getWorktreePathSettings,
|
||||
@@ -2444,7 +2444,7 @@ export async function createLocalWorktree(
|
||||
emitCreateWorktreeProgress(mainWindow, 'fetching', args.creationId)
|
||||
}
|
||||
}
|
||||
const workspaceRoot = computeWorkspaceRoot(repo.path, worktreePathSettings)
|
||||
const workspaceRoot = await computeWorkspaceRootAsync(repo.path, worktreePathSettings)
|
||||
|
||||
// Why: this validation doesn't depend on remote refs, so it can overlap a required remote-tracking base refresh.
|
||||
const primarySetupScript = getEffectiveHooks(repo)?.scripts.setup
|
||||
@@ -2530,19 +2530,33 @@ export async function createLocalWorktree(
|
||||
username,
|
||||
localWorktreeGitOptions
|
||||
)
|
||||
checkoutExistingBranch = await canCheckoutExistingLocalBranch(
|
||||
repo.path,
|
||||
branchName,
|
||||
baseBranch,
|
||||
localWorktreeGitOptions
|
||||
)
|
||||
if (checkoutExistingBranch && !selectedExistingLocalBranchName) {
|
||||
// Why: suffix retries may need a new path, but an existing-branch checkout must keep the user-selected branch, not a sibling.
|
||||
selectedExistingLocalBranchName = branchName
|
||||
const tryExistingBranch = async (): Promise<boolean> => {
|
||||
checkoutExistingBranch = await canCheckoutExistingLocalBranch(
|
||||
repo.path,
|
||||
branchName,
|
||||
baseBranch,
|
||||
localWorktreeGitOptions
|
||||
)
|
||||
return checkoutExistingBranch
|
||||
}
|
||||
// Explicit branch selections retain the adoption-first path.
|
||||
const preferExistingBranch = Boolean(
|
||||
args.branchNameOverride || selectedExistingLocalBranchName
|
||||
)
|
||||
checkoutExistingBranch = preferExistingBranch && (await tryExistingBranch())
|
||||
lastBranchConflictKind = checkoutExistingBranch
|
||||
? null
|
||||
: await getBranchConflictKind(repo.path, branchName, baseBranch, localWorktreeGitOptions)
|
||||
: await getBranchConflictKind(
|
||||
repo.path,
|
||||
branchName,
|
||||
baseBranch,
|
||||
localWorktreeGitOptions,
|
||||
preferExistingBranch ? undefined : tryExistingBranch
|
||||
)
|
||||
if (checkoutExistingBranch && !selectedExistingLocalBranchName) {
|
||||
// Path retries must retain the adopted branch.
|
||||
selectedExistingLocalBranchName = branchName
|
||||
}
|
||||
const allowedPushTargetRemoteConflict =
|
||||
lastBranchConflictKind &&
|
||||
isAllowedPushTargetRemoteConflict(lastBranchConflictKind, branchName, args)
|
||||
@@ -2604,7 +2618,7 @@ export async function createLocalWorktree(
|
||||
}
|
||||
|
||||
worktreePath = ensurePathWithinWorkspace(
|
||||
computeWorktreePath(effectiveSanitizedName, repo.path, worktreePathSettings),
|
||||
computeWorktreePath(effectiveSanitizedName, repo.path, worktreePathSettings, workspaceRoot),
|
||||
workspaceRoot
|
||||
)
|
||||
if (existsSync(worktreePath)) {
|
||||
@@ -3010,6 +3024,8 @@ export async function createLocalWorktree(
|
||||
}
|
||||
})
|
||||
|
||||
// Startup resolves the new id before lifecycle notifications invalidate runtime caches.
|
||||
runtime?.invalidateWorktreeCatalog?.(repo.id)
|
||||
const stagedStartup = await timing.time('spawn_startup_terminal', () =>
|
||||
spawnLocalStartupAndSetupTerminals({
|
||||
runtime,
|
||||
|
||||
@@ -2,6 +2,8 @@ import { beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
import { resolve } from 'node:path'
|
||||
import type { CreateWorktreeResult } from '../../shared/worktree/create-types'
|
||||
import { resolveRegisteredWorktreePath } from './registered-worktree-roots-cache'
|
||||
import { computeWorkspaceRootAsync } from './worktree-logic'
|
||||
import type * as WorktreeLogic from './worktree-logic'
|
||||
import {
|
||||
listWorktreesMock,
|
||||
describeCreatedWorktreeMock,
|
||||
@@ -78,11 +80,13 @@ vi.mock('../setup-hook-env-vars', async (importOriginal) =>
|
||||
(await importOriginal()) as Record<string, unknown>
|
||||
)
|
||||
)
|
||||
vi.mock('./worktree-logic', async (importOriginal) =>
|
||||
(await import('./worktrees-test-module-mocks')).worktreeLogicModuleMock(
|
||||
(await importOriginal()) as Record<string, unknown>
|
||||
)
|
||||
)
|
||||
vi.mock('./worktree-logic', async (importOriginal) => {
|
||||
const actual = await importOriginal<typeof WorktreeLogic>()
|
||||
return {
|
||||
...(await import('./worktrees-test-module-mocks')).worktreeLogicModuleMock(actual),
|
||||
computeWorkspaceRootAsync: vi.fn(actual.computeWorkspaceRootAsync)
|
||||
}
|
||||
})
|
||||
vi.mock('../terminal-history-deletion', async () =>
|
||||
(await import('./worktrees-test-module-mocks')).terminalHistoryDeletionModuleMock()
|
||||
)
|
||||
@@ -409,15 +413,23 @@ describe('registerWorktreeHandlers', () => {
|
||||
}
|
||||
])
|
||||
|
||||
await handlers['worktrees:create'](null, {
|
||||
const root = Promise.withResolvers<string>()
|
||||
vi.mocked(computeWorkspaceRootAsync).mockReturnValueOnce(root.promise)
|
||||
const create = handlers['worktrees:create'](null, {
|
||||
repoId: 'repo-1',
|
||||
name: 'feature'
|
||||
})
|
||||
|
||||
expect(computeWorktreePathMock).toHaveBeenCalledWith('feature', '/workspace/repo', {
|
||||
nestWorkspaces: false,
|
||||
workspaceDir: '../worktrees'
|
||||
})
|
||||
await vi.waitFor(() => expect(computeWorkspaceRootAsync).toHaveBeenCalled())
|
||||
expect(addWorktreeMock).not.toHaveBeenCalled()
|
||||
root.resolve('/workspace/worktrees')
|
||||
await create
|
||||
expect(computeWorktreePathMock).toHaveBeenCalledWith(
|
||||
'feature',
|
||||
'/workspace/repo',
|
||||
{ nestWorkspaces: false, workspaceDir: '../worktrees' },
|
||||
'/workspace/worktrees'
|
||||
)
|
||||
expect(addWorktreeMock).toHaveBeenCalledWith(
|
||||
'/workspace/repo',
|
||||
'../worktrees/feature',
|
||||
@@ -689,6 +701,10 @@ describe('registerWorktreeHandlers', () => {
|
||||
expect(setupCommand).toBe('bash /workspace/repo/.git/orca/setup-runner.sh')
|
||||
expect(result.setup).toBeUndefined()
|
||||
expect(result.startupTerminal).toEqual({ spawned: true, surface: 'visible' })
|
||||
expect(runtimeStub.invalidateWorktreeCatalog).toHaveBeenCalledWith('repo-1')
|
||||
expect(runtimeStub.invalidateWorktreeCatalog.mock.invocationCallOrder[0]).toBeLessThan(
|
||||
runtimeStub.createTerminal.mock.invocationCallOrder[0]
|
||||
)
|
||||
expect(result.timing?.phases.map((phase) => phase.phase)).toEqual(
|
||||
expect.arrayContaining([
|
||||
'git_worktree_add',
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { type Mock, vi } from 'vitest'
|
||||
import type { computeWorktreePath } from './worktree-logic'
|
||||
import type { HandlerMap } from './worktrees-test-ipc-surface'
|
||||
|
||||
/** Loose signature: one mock stands in for many unrelated module exports. */
|
||||
@@ -78,13 +79,7 @@ export const resolveSetupRunnerShellMock: ModuleMock = vi.fn()
|
||||
export const runHookMock: ModuleMock = vi.fn()
|
||||
export const hasHooksFileMock: ModuleMock = vi.fn()
|
||||
export const loadHooksMock: ModuleMock = vi.fn()
|
||||
export const computeWorktreePathMock: Mock<
|
||||
(
|
||||
sanitizedName: string,
|
||||
repoPath: string,
|
||||
settings: { nestWorkspaces: boolean; workspaceDir: string }
|
||||
) => string
|
||||
> = vi.fn()
|
||||
export const computeWorktreePathMock: Mock<typeof computeWorktreePath> = vi.fn()
|
||||
export const ensurePathWithinWorkspaceMock: StringArgMock = vi.fn()
|
||||
export const gitExecFileAsyncMock: GitArgvMock = vi.fn()
|
||||
export const getSshGitProviderMock: StringArgMock = vi.fn()
|
||||
@@ -138,7 +133,19 @@ export const gitRepoModuleMock = () => ({
|
||||
resolveDefaultBaseRefWithLocalGit: resolveDefaultBaseRefWithLocalGitMock,
|
||||
resolveDefaultBaseRefViaExec: resolveDefaultBaseRefViaExecMock,
|
||||
getDefaultRemote: getDefaultRemoteMock,
|
||||
getBranchConflictKind: getBranchConflictKindMock
|
||||
getBranchConflictKind: async (
|
||||
repoPath: string,
|
||||
branch: string,
|
||||
base?: string,
|
||||
options?: { wslDistro?: string },
|
||||
allowLocalBranch?: () => Promise<boolean>
|
||||
) => {
|
||||
// These handler tests stub ref presence; policy tests cover the absent-ref fast path.
|
||||
if (allowLocalBranch && (await allowLocalBranch())) {
|
||||
return null
|
||||
}
|
||||
return getBranchConflictKindMock(repoPath, branch, base, options)
|
||||
}
|
||||
})
|
||||
|
||||
export const githubClientModuleMock = () => ({
|
||||
|
||||
@@ -12,6 +12,7 @@ export type WorktreeRuntimeStub = {
|
||||
clearOptimisticReconcileToken: ReturnType<typeof vi.fn>
|
||||
resolveManagedMrBase: ReturnType<typeof vi.fn>
|
||||
createTerminal: ReturnType<typeof vi.fn>
|
||||
invalidateWorktreeCatalog: ReturnType<typeof vi.fn>
|
||||
splitTerminal: ReturnType<typeof vi.fn>
|
||||
notifyWorktreesChangedForRemoteClients: ReturnType<typeof vi.fn>
|
||||
closeFileWatchersForRemoval: ReturnType<typeof vi.fn>
|
||||
@@ -38,6 +39,7 @@ export function createWorktreeRuntimeStub(): WorktreeRuntimeStub {
|
||||
title: null,
|
||||
surface: 'visible'
|
||||
}),
|
||||
invalidateWorktreeCatalog: vi.fn(),
|
||||
splitTerminal: vi.fn().mockResolvedValue({
|
||||
handle: 'term-setup',
|
||||
tabId: 'tab-startup',
|
||||
|
||||
@@ -0,0 +1,196 @@
|
||||
// An empty chat beside a pre-SQLite journal explains itself.
|
||||
//
|
||||
// The SQLite move shipped no importer, so a session whose history is a
|
||||
// `log.jsonl` founds a fresh empty journal beside it and looks exactly like a
|
||||
// chat created seconds ago. One status row is the difference.
|
||||
|
||||
import { mkdtemp, rm, writeFile } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, beforeEach, describe, expect, it } from 'vitest'
|
||||
import Database from '../../sqlite/sync-database'
|
||||
import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key'
|
||||
import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types'
|
||||
import { projectStructuredItemsToNativeChat } from '../../../shared/structured-agent-session-projection'
|
||||
import { openJournalDatabase } from './journal-database'
|
||||
import { JOURNAL_DB_SCHEMA_VERSION } from './journal-database-schema'
|
||||
import { JOURNAL_FILE_FORMAT_REMNANT_DISCLOSURE_IDENTITY } from './journal-file-format-remnant'
|
||||
import { loadJournal } from './journal-open'
|
||||
import { journalDatabaseFile } from './journal-paths'
|
||||
import type { AgentSessionJournal } from './journal-store'
|
||||
import type { openAgentSessionJournal } from './journal-store-factory'
|
||||
import { createTrackedJournalOpener } from './journal-store-test-open'
|
||||
|
||||
const IDENTITY: AgentSessionJournalIdentity = {
|
||||
sessionId: 'session-1',
|
||||
workspaceId: 'ws-1',
|
||||
hostId: 'host-1',
|
||||
agent: 'codex',
|
||||
providerHandle: { kind: 'codex', threadId: 'thread-1' }
|
||||
}
|
||||
|
||||
const DISCLOSURE_ITEM_ID = agentJournalItemKey(JOURNAL_FILE_FORMAT_REMNANT_DISCLOSURE_IDENTITY)
|
||||
|
||||
let root: string
|
||||
let clock = 1_000
|
||||
const journals = createTrackedJournalOpener()
|
||||
|
||||
function open(overrides: Partial<Parameters<typeof openAgentSessionJournal>[0]> = {}) {
|
||||
return journals.open({
|
||||
identity: IDENTITY,
|
||||
journalDir: root,
|
||||
now: () => (clock += 1),
|
||||
mintEpoch: () => `epoch-${clock}`,
|
||||
...overrides
|
||||
})
|
||||
}
|
||||
|
||||
function writeRemnant(name = 'log.jsonl'): Promise<void> {
|
||||
return writeFile(join(root, name), '{"kind":"epoch","v":1,"seq":1}\n', 'utf8')
|
||||
}
|
||||
|
||||
function disclosure(journal: AgentSessionJournal): string | null {
|
||||
const row = journal.snapshot().items.find((entry) => entry.itemId === DISCLOSURE_ITEM_ID)
|
||||
return row?.body.kind === 'status' ? row.body.text : null
|
||||
}
|
||||
|
||||
beforeEach(async () => {
|
||||
root = await mkdtemp(join(tmpdir(), 'orca-journal-remnant-'))
|
||||
clock = 1_000
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
await journals.closeAll()
|
||||
await rm(root, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
describe('a chat whose history is still in the pre-SQLite format', () => {
|
||||
it('says how to carry on, and where the transcript is', async () => {
|
||||
await writeRemnant()
|
||||
|
||||
const journal = await open()
|
||||
|
||||
expect(disclosure(journal)).toContain('send a message to pick up where you left off')
|
||||
expect(disclosure(journal)).toContain(join(root, 'log.jsonl'))
|
||||
expect(disclosure(journal)).toContain('Codex')
|
||||
})
|
||||
|
||||
// Both files is the normal shape of a pre-SQLite directory: every epoch roll
|
||||
// staged a snapshot whether or not anything compacted into it, so preferring
|
||||
// the snapshot would name an empty file for ~every affected chat.
|
||||
it('names the log, not the snapshot staged beside it', async () => {
|
||||
await writeRemnant('log.jsonl')
|
||||
await writeRemnant('snapshot.json')
|
||||
|
||||
const journal = await open()
|
||||
|
||||
expect(disclosure(journal)).toContain(join(root, 'log.jsonl'))
|
||||
expect(disclosure(journal)).not.toContain('snapshot.json')
|
||||
})
|
||||
|
||||
it('falls back to the snapshot when a chat has no log beside it', async () => {
|
||||
await writeRemnant('snapshot.json')
|
||||
|
||||
const journal = await open()
|
||||
|
||||
expect(disclosure(journal)).toContain(join(root, 'snapshot.json'))
|
||||
})
|
||||
|
||||
it('says nothing to a chat that is genuinely new', async () => {
|
||||
const journal = await open()
|
||||
|
||||
expect(journal.snapshot().items).toEqual([])
|
||||
})
|
||||
|
||||
// Counting rows proves nothing here — the append upserts by identity, so a
|
||||
// second append would still leave exactly one. The revision is what moves.
|
||||
it('does not re-append the row on a later open', async () => {
|
||||
await writeRemnant()
|
||||
const first = await open()
|
||||
const firstRevision = first
|
||||
.snapshot()
|
||||
.items.find((e) => e.itemId === DISCLOSURE_ITEM_ID)?.revision
|
||||
await first.close()
|
||||
|
||||
const reopened = await open()
|
||||
|
||||
const row = reopened.snapshot().items.find((e) => e.itemId === DISCLOSURE_ITEM_ID)
|
||||
expect(firstRevision).toBe(1)
|
||||
expect(row?.revision).toBe(1)
|
||||
expect(reopened.cursor().sequence).toBe(2)
|
||||
})
|
||||
|
||||
// The epoch commit and this append are separate transactions; if the append is
|
||||
// lost the epoch exists but holds nothing, and every later open takes the
|
||||
// adopt branch. The offer has to survive that.
|
||||
it('offers the message again when a committed epoch holds nothing', async () => {
|
||||
const founded = await open()
|
||||
await founded.close()
|
||||
await writeRemnant()
|
||||
|
||||
const reopened = await open()
|
||||
|
||||
expect(disclosure(reopened)).toContain(join(root, 'log.jsonl'))
|
||||
})
|
||||
|
||||
// A repair's epoch is the marker that history was deleted and never rebuilt,
|
||||
// and any row that is not the repair's own disclosure retires it. Appending
|
||||
// here would silently stop the session ever asking the provider for that
|
||||
// history — with the journal still holding none.
|
||||
it('stays out of a journal this open just repaired', async () => {
|
||||
const journal = await open()
|
||||
await journal.appendItem(
|
||||
{ provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 0 },
|
||||
{ kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'history' }] },
|
||||
{ fence: 1 }
|
||||
)
|
||||
await journal.close()
|
||||
// Deleting the anchor leaves every row unanchored: replay keeps nothing, so
|
||||
// the repair publishes an empty `unreconcilable_prefix` epoch and — costing
|
||||
// no malformed row — appends no disclosure of its own. That is the one state
|
||||
// where this branch and a repair meet.
|
||||
const opened = openJournalDatabase(journalDatabaseFile(root))
|
||||
try {
|
||||
opened.db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(1)
|
||||
} finally {
|
||||
opened.db.close()
|
||||
}
|
||||
await writeRemnant()
|
||||
|
||||
const repaired = await open()
|
||||
|
||||
expect(disclosure(repaired)).toBeNull()
|
||||
// Still asking the provider for the history the repair dropped.
|
||||
expect(loadJournal(root, IDENTITY.sessionId)).toMatchObject({ corrupt: true })
|
||||
})
|
||||
|
||||
// A latched journal loads empty, so it reaches the same branch — and an append
|
||||
// into one throws, which would make the session unopenable rather than read-only.
|
||||
it('writes nothing into a journal latched by a newer schema', async () => {
|
||||
const founded = await open()
|
||||
await founded.close()
|
||||
const db = new Database(journalDatabaseFile(root))
|
||||
try {
|
||||
db.pragma(`user_version = ${JOURNAL_DB_SCHEMA_VERSION + 1}`)
|
||||
} finally {
|
||||
db.close()
|
||||
}
|
||||
await writeRemnant()
|
||||
|
||||
const latched = await open()
|
||||
|
||||
expect(latched.isReadOnly).toBe(true)
|
||||
expect(disclosure(latched)).toBeNull()
|
||||
})
|
||||
|
||||
// A row nothing projects is a row nobody reads.
|
||||
it('renders in the transcript as a system line', async () => {
|
||||
await writeRemnant()
|
||||
|
||||
const journal = await open()
|
||||
|
||||
const messages = projectStructuredItemsToNativeChat(journal.snapshot().items)
|
||||
expect(messages).toHaveLength(1)
|
||||
expect(messages[0]?.role).toBe('system')
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,60 @@
|
||||
// A journal directory left behind by the pre-SQLite file format.
|
||||
//
|
||||
// Not `journal-legacy-import.ts`, which reads the PROVIDER's own transcript.
|
||||
// This is Orca's own `log.jsonl`, which no build after the SQLite move reads.
|
||||
// Nothing imports it, so the session it belonged to opens empty and is
|
||||
// indistinguishable from a chat created seconds ago — same `session_created`
|
||||
// epoch, same empty timeline. The remnant is the one durable fact that tells
|
||||
// them apart, so the empty session says where its history went and how to carry
|
||||
// on instead of silently claiming it never had any.
|
||||
|
||||
import { existsSync } from 'node:fs'
|
||||
import { join } from 'node:path'
|
||||
import type { AgentJournalItemIdentity } from '../../../shared/agent-session-journal-types'
|
||||
import { boundJournalStatusText } from './journal-prompt-body-bounds'
|
||||
import { formatAgentTypeLabel } from '../../../shared/agent-type-label'
|
||||
import type { AgentType } from '../../../shared/agent-status-types'
|
||||
|
||||
/** The remnant's transcript, or null when the directory never held one.
|
||||
*
|
||||
* `log.jsonl` first, and the order matters: every epoch roll staged a
|
||||
* `snapshot.json` whether or not anything was ever compacted into it, so the
|
||||
* file's existence says nothing about where the history lives. Measured across
|
||||
* a real profile, `compactedThrough` was 0 in all 80 — the log holds the
|
||||
* transcript and the snapshot is the fallback for a session that has no log. */
|
||||
export function findJournalFileFormatRemnant(journalDir: string): string | null {
|
||||
for (const name of ['log.jsonl', 'snapshot.json']) {
|
||||
const path = join(journalDir, name)
|
||||
if (existsSync(path)) {
|
||||
return path
|
||||
}
|
||||
}
|
||||
return null
|
||||
}
|
||||
|
||||
/** One stable identity, so a reopen upserts the same row instead of adding one. */
|
||||
export const JOURNAL_FILE_FORMAT_REMNANT_DISCLOSURE_IDENTITY: AgentJournalItemIdentity = {
|
||||
provider: 'orca',
|
||||
clientMessageId: 'journal-file-format-remnant'
|
||||
}
|
||||
|
||||
/** How to carry on. The session attaches on the record's own provider handle, so it
|
||||
* still names the conversation the transcript no longer shows — whether the provider
|
||||
* itself still holds that thread is its own business, hence "points at". */
|
||||
export function journalFileFormatRemnantDisclosure(input: {
|
||||
transcriptPath: string
|
||||
agent: AgentType
|
||||
}): { identity: AgentJournalItemIdentity; body: { kind: 'status'; text: string } } {
|
||||
return {
|
||||
identity: JOURNAL_FILE_FORMAT_REMNANT_DISCLOSURE_IDENTITY,
|
||||
body: {
|
||||
kind: 'status',
|
||||
text: boundJournalStatusText(
|
||||
`This chat's history was saved in an older format Orca no longer reads, so it starts ` +
|
||||
`empty. The session still points at the same ${formatAgentTypeLabel(input.agent)} ` +
|
||||
`conversation — send a message to pick up where you left off. The original ` +
|
||||
`transcript is on the session's host at \`${input.transcriptPath}\``
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,7 +1,16 @@
|
||||
import { mkdir } from 'node:fs/promises'
|
||||
import type { AgentType } from '../../../shared/agent-status-types'
|
||||
import {
|
||||
findJournalFileFormatRemnant,
|
||||
journalFileFormatRemnantDisclosure
|
||||
} from './journal-file-format-remnant'
|
||||
import type { JournalLoad } from './journal-open'
|
||||
import { journalRepairDisclosure, type JournalRepairDisclosure } from './journal-repair-disclosure'
|
||||
|
||||
/** What any of this file's disclosures hands the store — a repair's, or the
|
||||
* pre-SQLite notice's. Same shape, and neither is only a repair. */
|
||||
type JournalDisclosure = JournalRepairDisclosure
|
||||
|
||||
export async function ensureJournalDir(journalDir: string): Promise<void> {
|
||||
await mkdir(journalDir, { recursive: true })
|
||||
}
|
||||
@@ -32,6 +41,7 @@ export async function openJournalStoreState(input: {
|
||||
body: JournalRepairDisclosure['body'],
|
||||
fence: number
|
||||
) => Promise<unknown>
|
||||
agent: AgentType
|
||||
highestFence: () => number
|
||||
malformedRows: () => number
|
||||
setMalformedRows: (count: number) => void
|
||||
@@ -40,6 +50,7 @@ export async function openJournalStoreState(input: {
|
||||
const loaded = input.loaded !== undefined ? input.loaded : input.replay()
|
||||
if (!loaded) {
|
||||
input.start()
|
||||
await discloseFileFormatRemnant(input)
|
||||
return
|
||||
}
|
||||
input.adopt(loaded)
|
||||
@@ -59,4 +70,43 @@ export async function openJournalStoreState(input: {
|
||||
const disclosure = journalRepairDisclosure({ malformedRows: input.malformedRows() })
|
||||
await input.appendDisclosure(disclosure.identity, disclosure.body, input.highestFence())
|
||||
}
|
||||
// Founding the epoch and appending the row are two transactions, and a
|
||||
// committed epoch sends every later open down this branch instead. Anything
|
||||
// that interrupts between them — a quit during startup restore, a failed
|
||||
// append — would otherwise lose the message for good. An epoch holding nothing
|
||||
// is exactly the state that append was owed, so offer it again.
|
||||
//
|
||||
// Never onto a repair, though: `loaded.state` is the PRE-repair load, so a
|
||||
// journal this open just emptied looks identical. The repair's epoch is the
|
||||
// marker that its history was deleted and never rebuilt, and any row that is
|
||||
// not the repair's own disclosure retires it — this row would silently stop
|
||||
// the session ever asking the provider for that history again.
|
||||
if (!loaded.corrupt && loaded.state.items.size === 0 && loaded.state.submissions.size === 0) {
|
||||
await discloseFileFormatRemnant(input)
|
||||
}
|
||||
}
|
||||
|
||||
/** Says what happened to a chat whose history is in the abandoned file format.
|
||||
* Upserts by a constant identity, so the offer above is exactly-once in effect:
|
||||
* once the row exists the epoch is no longer empty. */
|
||||
async function discloseFileFormatRemnant(input: {
|
||||
journalDir: string
|
||||
agent: AgentType
|
||||
appendDisclosure: (
|
||||
identity: JournalDisclosure['identity'],
|
||||
body: JournalDisclosure['body'],
|
||||
fence: number
|
||||
) => Promise<unknown>
|
||||
highestFence: () => number
|
||||
readOnly: () => boolean
|
||||
}): Promise<void> {
|
||||
if (input.readOnly()) {
|
||||
return
|
||||
}
|
||||
const transcriptPath = findJournalFileFormatRemnant(input.journalDir)
|
||||
if (!transcriptPath) {
|
||||
return
|
||||
}
|
||||
const disclosure = journalFileFormatRemnantDisclosure({ transcriptPath, agent: input.agent })
|
||||
await input.appendDisclosure(disclosure.identity, disclosure.body, input.highestFence())
|
||||
}
|
||||
|
||||
@@ -41,6 +41,7 @@ export function restoreJournalStore(
|
||||
adopt: host.adopt,
|
||||
appendDisclosure: (identity, body, fence) =>
|
||||
host.journal().appendItem(identity, body, { fence }),
|
||||
agent: host.identity.agent,
|
||||
highestFence: () => host.state().highestFence,
|
||||
malformedRows: host.malformedRows,
|
||||
setMalformedRows: host.setMalformedRows,
|
||||
|
||||
@@ -112,4 +112,52 @@ describe('provider frame classification catalog', () => {
|
||||
// An item type nobody has dispositioned still falls through visibly.
|
||||
expect(classifyProviderFrame('codex', 'item:futureThing', {})).toBe('timeline-substantive')
|
||||
})
|
||||
|
||||
it('chromes the one unmodelled codex item type that carries no content', () => {
|
||||
expect(classifyProviderFrame('codex', 'item:sleep', { id: 's', durationMs: 20_000 })).toBe(
|
||||
'status-chrome'
|
||||
)
|
||||
// Payload inspection still outranks the item catalog, so chroming a type
|
||||
// cannot swallow one that reports a failure.
|
||||
expect(classifyProviderFrame('codex', 'item:sleep', { id: 's', status: 'failed' })).toBe(
|
||||
'error-surface'
|
||||
)
|
||||
})
|
||||
|
||||
it('keeps subagent items visible — the only evidence a spawned agent is working', () => {
|
||||
expect(
|
||||
classifyProviderFrame('codex', 'item:subAgentActivity', {
|
||||
id: 'a-1',
|
||||
kind: 'started',
|
||||
agentThreadId: 'thread-child',
|
||||
agentPath: '/root/list_directory'
|
||||
})
|
||||
).toBe('timeline-substantive')
|
||||
expect(
|
||||
classifyProviderFrame('codex', 'item:collabAgentToolCall', {
|
||||
id: 'c-1',
|
||||
tool: 'spawn',
|
||||
status: 'inProgress',
|
||||
senderThreadId: 'thread-root',
|
||||
receiverThreadIds: ['thread-child'],
|
||||
agentsStates: {}
|
||||
})
|
||||
).toBe('timeline-substantive')
|
||||
})
|
||||
|
||||
it('leaves content-bearing codex item types on the visible fallback', () => {
|
||||
// Each carries text or a path a user would want: review output, the image
|
||||
// the agent looked at or generated, injected hook prompt text.
|
||||
for (const type of [
|
||||
'imageView',
|
||||
'imageGeneration',
|
||||
'enteredReviewMode',
|
||||
'exitedReviewMode',
|
||||
'hookPrompt'
|
||||
]) {
|
||||
expect(classifyProviderFrame('codex', `item:${type}`, { id: 'i' }), type).toBe(
|
||||
'timeline-substantive'
|
||||
)
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
@@ -197,7 +197,12 @@ function hasProviderError(payload: unknown): boolean {
|
||||
const CODEX_ITEM_CLASSIFICATIONS: Record<string, ProviderFrameClassification> = {
|
||||
// The `thread/compacted` notification is already chrome; its item form is the
|
||||
// same event and must not read as a mysterious opcode row.
|
||||
contextCompaction: 'status-chrome'
|
||||
contextCompaction: 'status-chrome',
|
||||
// `{id, durationMs}` and nothing else — Codex's own transcript renders it as
|
||||
// nothing at all. Every other item type this build does not model carries text
|
||||
// a user would want (review output, an image path, hook prompt text, subagent
|
||||
// progress), so those keep their visible fallback row.
|
||||
sleep: 'status-chrome'
|
||||
}
|
||||
|
||||
function notificationKind(kind: string): string {
|
||||
|
||||
@@ -137,7 +137,11 @@ export type StructuredAgentSessionAdapter = {
|
||||
turnId: string
|
||||
fence: number
|
||||
}): Promise<{ cancelled: boolean }>
|
||||
stopBackgroundTasks?(input: { sessionId: string; fence: number }): Promise<{ cancelled: boolean }>
|
||||
stopBackgroundTasks?(input: {
|
||||
sessionId: string
|
||||
fence: number
|
||||
taskId?: string
|
||||
}): Promise<{ cancelled: boolean }>
|
||||
backgroundTaskState?(sessionId: string): AgentSessionBackgroundTaskState | null | undefined
|
||||
/** Fires the provider callback for an approval or a question. The wire calls
|
||||
* this only after the durable compare-and-set won, so it runs exactly once. */
|
||||
|
||||
@@ -8,7 +8,7 @@ import type {
|
||||
import type { AgentSessionJournal } from '../agent-session-journal/journal-store'
|
||||
import { readAgentSessionHistory } from './agent-session-history-page'
|
||||
|
||||
function providerSessionMetadata(
|
||||
export function structuredAgentSessionProviderSessionMetadata(
|
||||
record: AgentSessionRecord | null
|
||||
): AgentProviderSessionMetadata | undefined {
|
||||
const head = record ? agentSessionProviderHandleChainHead(record.providerHandleChain) : null
|
||||
@@ -27,7 +27,7 @@ export function readStructuredAgentSessionHistoryResult(input: {
|
||||
}): AgentSessionHistoryResult {
|
||||
const result = readAgentSessionHistory(input.journal, input.request)
|
||||
const fence = input.record?.lease.runtimeFence
|
||||
const providerSession = providerSessionMetadata(input.record)
|
||||
const providerSession = structuredAgentSessionProviderSessionMetadata(input.record)
|
||||
if (fence === undefined) {
|
||||
return providerSession ? { ...result, providerSession } : result
|
||||
}
|
||||
|
||||
@@ -19,6 +19,8 @@ import type {
|
||||
StructuredAgentSessionHostSession
|
||||
} from './structured-agent-session-host-types'
|
||||
import { releaseStoredStructuredAgentSessionOwner } from './structured-agent-session-lease-release'
|
||||
import { resumeHeldStructuredAgentSession } from './structured-agent-session-hold-resume'
|
||||
import type { AgentSessionWireRefusal } from '../../../shared/agent-session-wire'
|
||||
|
||||
export type StructuredAgentSessionLifetimeContext = {
|
||||
deps: StructuredAgentSessionHostDeps
|
||||
@@ -67,6 +69,27 @@ export async function evictHeldStructuredAgentSession(
|
||||
)
|
||||
}
|
||||
|
||||
/** The first hold on a childless session: reconcile the lease, settle recovery, then attach. */
|
||||
export async function resumeStructuredAgentSessionForHold(
|
||||
context: StructuredAgentSessionLifetimeContext & {
|
||||
reconcileLeases: (sessionId: string) => Promise<AgentSessionWireRefusal | null>
|
||||
},
|
||||
sessionId: string,
|
||||
attach: Parameters<typeof resumeHeldStructuredAgentSession>[0]['attach']
|
||||
): Promise<void> {
|
||||
const unreconciled = await context.reconcileLeases(sessionId)
|
||||
if (unreconciled) {
|
||||
throw new Error(unreconciled.code)
|
||||
}
|
||||
await context.runtimeState.resolveRecovery(sessionId)
|
||||
await resumeHeldStructuredAgentSession({
|
||||
sessionId,
|
||||
deps: context.deps,
|
||||
now: context.now,
|
||||
attach
|
||||
})
|
||||
}
|
||||
|
||||
export function createStructuredAgentSessionHolds(
|
||||
context: StructuredAgentSessionLifetimeContext,
|
||||
input: {
|
||||
|
||||
@@ -78,6 +78,7 @@ export function cancelStructuredAgentSessionTurn(
|
||||
envelope: AgentSessionMutationEnvelope
|
||||
turnId: string
|
||||
scope?: 'background-tasks'
|
||||
taskId?: string
|
||||
}
|
||||
): Promise<AgentSessionMutationResult<AgentSessionCancelResult>> {
|
||||
return mutate(context, caller, params.envelope, cancelPlan(params))
|
||||
|
||||
@@ -2,17 +2,7 @@
|
||||
// Mutations share one durable admission path and serialize per session.
|
||||
|
||||
import type { AgentSessionExecutionLocation } from '../../../shared/agent-session-record'
|
||||
import type {
|
||||
AgentSessionAttachResult,
|
||||
AgentSessionHistoryRequest,
|
||||
AgentSessionHistoryResult,
|
||||
AgentSessionHandoffRequest,
|
||||
AgentSessionHandoffResult,
|
||||
AgentSessionHandoffStatus,
|
||||
AgentSessionMutationResult,
|
||||
AgentSessionOptionsResult,
|
||||
AgentSessionWireRefusal
|
||||
} from '../../../shared/agent-session-wire'
|
||||
import type * as SessionWire from '../../../shared/agent-session-wire'
|
||||
import type { AgentSessionAttachParams } from './structured-agent-session-attach'
|
||||
import { AGENT_SESSION_NOT_ATTACHED } from './structured-agent-session-mutation-admission'
|
||||
import { createRestartReconciler } from './structured-agent-session-restart-reconcile'
|
||||
@@ -33,13 +23,13 @@ import { attachStructuredAgentSession } from './structured-agent-session-attach-
|
||||
import {
|
||||
createStructuredAgentSessionHolds,
|
||||
evictHeldStructuredAgentSession,
|
||||
resumeStructuredAgentSessionForHold,
|
||||
type StructuredAgentSessionLifetimeContext
|
||||
} from './structured-agent-session-host-lifetime'
|
||||
import type {
|
||||
StructuredAgentSessionHolds,
|
||||
StructuredAgentSessionHoldOptions
|
||||
} from './structured-agent-session-holds'
|
||||
import { resumeHeldStructuredAgentSession } from './structured-agent-session-hold-resume'
|
||||
import type { StructuredAgentSessionAttachContext } from './structured-agent-session-attach-context'
|
||||
import { listStructuredAgentSessionTabs } from './structured-agent-session-host-tabs'
|
||||
import {
|
||||
@@ -57,6 +47,7 @@ import type {
|
||||
StructuredAgentSessionHostDeps,
|
||||
StructuredAgentSessionHostSession
|
||||
} from './structured-agent-session-host-types'
|
||||
import { StructuredAgentSessionStatusFeed } from './structured-agent-session-status-feed'
|
||||
import { StructuredAgentSessionEventRecovery } from './structured-agent-session-event-recovery'
|
||||
import { StructuredAgentSessionBackgroundTaskChannel } from './structured-agent-session-background-task-channel'
|
||||
import { withTimeout } from '../../../shared/promise-timeout-fallback'
|
||||
@@ -66,10 +57,19 @@ const HANDOFF_DRAIN_TIMEOUT_MS = 5_000
|
||||
|
||||
export class StructuredAgentSessionHost {
|
||||
private readonly sessions = new Map<string, StructuredAgentSessionHostSession>()
|
||||
private readonly subscribers = new AgentSessionSubscribers()
|
||||
private readonly statusFeed = new StructuredAgentSessionStatusFeed({
|
||||
sessions: this.sessions,
|
||||
getRecord: (sessionId) => this.deps.store.getRecord(sessionId),
|
||||
now: () => this.now()
|
||||
})
|
||||
private readonly subscribers = new AgentSessionSubscribers({
|
||||
onJournalPublished: (sessionId, journal) => this.statusFeed.publish(sessionId, journal)
|
||||
})
|
||||
private readonly tasks = new StructuredAgentSessionTaskQueue()
|
||||
private readonly runtimeState: StructuredAgentSessionHostRuntimeState
|
||||
private readonly reconcileLeases: (sessionId: string) => Promise<AgentSessionWireRefusal | null>
|
||||
private readonly reconcileLeases: (
|
||||
sessionId: string
|
||||
) => Promise<SessionWire.AgentSessionWireRefusal | null>
|
||||
private readonly handoffs: StructuredAgentSessionHostHandoff
|
||||
private readonly readableRestorer: StructuredAgentSessionReadableRestorer
|
||||
private readonly restartRestore = new StructuredAgentSessionRestartRestoreGate()
|
||||
@@ -112,7 +112,12 @@ export class StructuredAgentSessionHost {
|
||||
now: this.now
|
||||
})
|
||||
this.holds = createStructuredAgentSessionHolds(this.lifetimeContext(), {
|
||||
resume: (sessionId) => this.resumeForHold(sessionId),
|
||||
resume: (sessionId) =>
|
||||
resumeStructuredAgentSessionForHold(
|
||||
{ ...this.lifetimeContext(), reconcileLeases: this.reconcileLeases },
|
||||
sessionId,
|
||||
(params) => this.attach({ callerKey: 'trusted-local:surface-hold' }, params)
|
||||
),
|
||||
evict: (sessionId) => this.close(sessionId)
|
||||
})
|
||||
this.readableRestorer = new StructuredAgentSessionReadableRestorer({
|
||||
@@ -125,7 +130,10 @@ export class StructuredAgentSessionHost {
|
||||
hasSession: (sessionId) => this.sessions.has(sessionId),
|
||||
// Site 10: cannot overwrite a live entry — the restorer returns early on
|
||||
// `hasSession` inside the same serialized step as this `set`.
|
||||
onReadable: (sessionId, restored) => this.sessions.set(sessionId, restored),
|
||||
onReadable: (sessionId, restored) => {
|
||||
this.sessions.set(sessionId, restored)
|
||||
this.statusFeed.publish(sessionId)
|
||||
},
|
||||
restoreHandoff: (sessionId) => this.handoffs.restore(sessionId)
|
||||
})
|
||||
this.eventRecovery = new StructuredAgentSessionEventRecovery({
|
||||
@@ -160,20 +168,6 @@ export class StructuredAgentSessionHost {
|
||||
/** That surface is gone. The child outlives it by the release grace, and by any running turn. */
|
||||
release = (sessionId: string, holderId: string): void => this.holds.release(sessionId, holderId)
|
||||
|
||||
private async resumeForHold(sessionId: string): Promise<void> {
|
||||
const unreconciled = await this.reconcileLeases(sessionId)
|
||||
if (unreconciled) {
|
||||
throw new Error(unreconciled.code)
|
||||
}
|
||||
await this.runtimeState.resolveRecovery(sessionId)
|
||||
await resumeHeldStructuredAgentSession({
|
||||
sessionId,
|
||||
deps: this.deps,
|
||||
now: () => this.now(),
|
||||
attach: (params) => this.attach({ callerKey: 'trusted-local:surface-hold' }, params)
|
||||
})
|
||||
}
|
||||
|
||||
handleAdapterEvent = (event: Parameters<StructuredAgentSessionEventRecovery['handle']>[0]) =>
|
||||
this.eventRecovery.handle(event)
|
||||
|
||||
@@ -250,7 +244,7 @@ export class StructuredAgentSessionHost {
|
||||
attach(
|
||||
caller: StructuredAgentSessionCaller,
|
||||
params: AgentSessionAttachParams
|
||||
): Promise<AgentSessionMutationResult<AgentSessionAttachResult>> {
|
||||
): Promise<SessionWire.AgentSessionMutationResult<SessionWire.AgentSessionAttachResult>> {
|
||||
return attachStructuredAgentSession(this.attachContext(), caller.callerKey, params)
|
||||
}
|
||||
|
||||
@@ -316,22 +310,23 @@ export class StructuredAgentSessionHost {
|
||||
|
||||
requestHandoff = (
|
||||
caller: StructuredAgentSessionCaller,
|
||||
params: AgentSessionHandoffRequest
|
||||
): Promise<AgentSessionMutationResult<AgentSessionHandoffResult>> =>
|
||||
params: SessionWire.AgentSessionHandoffRequest
|
||||
): Promise<SessionWire.AgentSessionMutationResult<SessionWire.AgentSessionHandoffResult>> =>
|
||||
this.handoffs.request(caller.callerKey, params)
|
||||
|
||||
readOptions = (sessionId: string): Promise<AgentSessionOptionsResult> =>
|
||||
readOptions = (sessionId: string): Promise<SessionWire.AgentSessionOptionsResult> =>
|
||||
readStructuredAgentSessionOptions(this.mutationContext(), sessionId)
|
||||
|
||||
async handoffStatus(sessionId: string): Promise<AgentSessionHandoffStatus> {
|
||||
async handoffStatus(sessionId: string): Promise<SessionWire.AgentSessionHandoffStatus> {
|
||||
this.requireSession(sessionId)
|
||||
return this.serialize(sessionId, () =>
|
||||
refreshRecoverableStructuredHandoffStatus(this.handoffs, this.deps.store, sessionId)
|
||||
)
|
||||
}
|
||||
|
||||
history = (request: AgentSessionHistoryRequest): AgentSessionHistoryResult =>
|
||||
this.backgroundTasks.history(request)
|
||||
history = (
|
||||
request: SessionWire.AgentSessionHistoryRequest
|
||||
): SessionWire.AgentSessionHistoryResult => this.backgroundTasks.history(request)
|
||||
|
||||
subscribe = (input: AgentSessionSubscribeInput): (() => void) =>
|
||||
this.backgroundTasks.subscribe(input)
|
||||
@@ -342,6 +337,11 @@ export class StructuredAgentSessionHost {
|
||||
) => this.backgroundTasks.publish(sessionId, state)
|
||||
unsubscribe = (sessionId: string, id: string): void => this.subscribers.close(sessionId, id)
|
||||
|
||||
/** Every session's projected status for session lists; unlike `subscribe`, retains nothing. */
|
||||
subscribeStatus = (
|
||||
subscriber: Parameters<StructuredAgentSessionStatusFeed['subscribe']>[0]
|
||||
): (() => void) => this.statusFeed.subscribe(subscriber)
|
||||
|
||||
private requireSession(sessionId: string): StructuredAgentSessionHostSession {
|
||||
const session = this.sessions.get(sessionId)
|
||||
if (!session) {
|
||||
|
||||
@@ -76,15 +76,21 @@ export function cancelPlan(params: {
|
||||
envelope: AgentSessionMutationEnvelope
|
||||
turnId: string
|
||||
scope?: 'background-tasks'
|
||||
taskId?: string
|
||||
}): MutationPlan<AgentSessionCancelResult> {
|
||||
return {
|
||||
method: 'agentSession.cancel',
|
||||
fields: { turnId: params.turnId, ...(params.scope ? { scope: params.scope } : {}) },
|
||||
fields: {
|
||||
turnId: params.turnId,
|
||||
...(params.scope ? { scope: params.scope } : {}),
|
||||
...(params.taskId ? { taskId: params.taskId } : {})
|
||||
},
|
||||
run: (ctx) =>
|
||||
performCancel(ctx, {
|
||||
clientOperationId: params.envelope.clientOperationId,
|
||||
turnId: params.turnId,
|
||||
...(params.scope ? { scope: params.scope } : {})
|
||||
...(params.scope ? { scope: params.scope } : {}),
|
||||
...(params.taskId ? { taskId: params.taskId } : {})
|
||||
}),
|
||||
// Interrupting twice would kill a turn the client never asked to stop, so a
|
||||
// replay reports the turn as already handled instead.
|
||||
|
||||
+110
@@ -0,0 +1,110 @@
|
||||
// Read restore decides whether a session comes back at all.
|
||||
//
|
||||
// A chat still in the pre-SQLite format has no `journal.db`, so the probe that
|
||||
// loads one reports nothing. Reading that as "no session" is what removed these
|
||||
// chats: an unpublished session is also what prunes its tab out of the saved
|
||||
// workspace, so the tab is gone before anything can explain itself.
|
||||
|
||||
import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, beforeEach, describe, expect, it } from 'vitest'
|
||||
import type { AgentSessionRecord } from '../../../shared/agent-session-record'
|
||||
import type { AgentSessionRecordStore } from '../../runtime/agent-session-record-store'
|
||||
import { journalDirectoryFor } from '../agent-session-journal/journal-paths'
|
||||
import type { AgentSessionJournal } from '../agent-session-journal/journal-store'
|
||||
import { restoreStructuredAgentSessionRead } from './structured-agent-session-read-restore'
|
||||
|
||||
const SESSION_ID = 'codex_read_restore_fixture'
|
||||
const WORKSPACE_ID = 'repo-1::/tmp/workspace'
|
||||
|
||||
const RECORD = {
|
||||
schemaVersion: 2,
|
||||
sessionId: SESSION_ID,
|
||||
location: {
|
||||
executionHostId: 'local',
|
||||
wslDistro: null,
|
||||
workspaceId: WORKSPACE_ID,
|
||||
workspaceKind: 'git-worktree'
|
||||
},
|
||||
provider: 'codex',
|
||||
providerHandleChain: [
|
||||
{
|
||||
linkId: 'codex-1-thread-1',
|
||||
handle: { provider: 'codex', threadId: 'thread-1' },
|
||||
origin: 'created',
|
||||
mintedAtFence: 1,
|
||||
observedAt: 1
|
||||
}
|
||||
],
|
||||
accountHome: { variable: 'CODEX_HOME', path: '/tmp/codex-home' },
|
||||
createdAt: 1,
|
||||
updatedAt: 2,
|
||||
lease: { sessionId: SESSION_ID, runtimeKind: 'native', runtimeFence: 1 }
|
||||
} as unknown as AgentSessionRecord
|
||||
|
||||
const store = {
|
||||
getRecord: (sessionId: string) => (sessionId === SESSION_ID ? RECORD : null)
|
||||
} as unknown as AgentSessionRecordStore
|
||||
|
||||
let journalRoot: string
|
||||
const opened: AgentSessionJournal[] = []
|
||||
|
||||
async function writeRemnant(name: string): Promise<string> {
|
||||
const dir = journalDirectoryFor(journalRoot, {
|
||||
workspaceId: WORKSPACE_ID,
|
||||
sessionId: SESSION_ID
|
||||
})
|
||||
await mkdir(dir, { recursive: true })
|
||||
await writeFile(join(dir, name), '{"kind":"epoch","v":1,"seq":1}\n', 'utf8')
|
||||
return join(dir, name)
|
||||
}
|
||||
|
||||
beforeEach(async () => {
|
||||
journalRoot = await mkdtemp(join(tmpdir(), 'orca-read-restore-'))
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
await Promise.allSettled(opened.splice(0).map((journal) => journal.close()))
|
||||
await rm(journalRoot, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
describe('a session whose journal is still the pre-SQLite format', () => {
|
||||
it('is published, carrying the message that explains it', async () => {
|
||||
const transcript = await writeRemnant('log.jsonl')
|
||||
|
||||
const restored = await restoreStructuredAgentSessionRead(store, journalRoot, SESSION_ID)
|
||||
|
||||
expect(restored).not.toBeNull()
|
||||
opened.push(restored!.journal)
|
||||
const disclosed = restored!.journal
|
||||
.snapshot()
|
||||
.items.map((entry) => (entry.body.kind === 'status' ? entry.body.text : ''))
|
||||
expect(disclosed.join('')).toContain(transcript)
|
||||
// Publishing it costs no agent process; acquisition still waits for the user.
|
||||
expect(restored!.hasProviderChild).toBe(false)
|
||||
})
|
||||
|
||||
it('is published for a remnant whose log is gone', async () => {
|
||||
await writeRemnant('snapshot.json')
|
||||
|
||||
const restored = await restoreStructuredAgentSessionRead(store, journalRoot, SESSION_ID)
|
||||
|
||||
expect(restored).not.toBeNull()
|
||||
opened.push(restored!.journal)
|
||||
})
|
||||
|
||||
it('still drops a session with neither a journal nor a remnant', async () => {
|
||||
const restored = await restoreStructuredAgentSessionRead(store, journalRoot, SESSION_ID)
|
||||
|
||||
expect(restored).toBeNull()
|
||||
})
|
||||
|
||||
it('still drops a session with no record', async () => {
|
||||
await writeRemnant('log.jsonl')
|
||||
|
||||
const restored = await restoreStructuredAgentSessionRead(store, journalRoot, 'unknown-session')
|
||||
|
||||
expect(restored).toBeNull()
|
||||
})
|
||||
})
|
||||
@@ -3,6 +3,7 @@ import type {
|
||||
AgentSessionRecord
|
||||
} from '../../../shared/agent-session-record'
|
||||
import type { AgentSessionRecordStore } from '../../runtime/agent-session-record-store'
|
||||
import { findJournalFileFormatRemnant } from '../agent-session-journal/journal-file-format-remnant'
|
||||
import { loadJournal } from '../agent-session-journal/journal-open'
|
||||
import { journalDirectoryFor } from '../agent-session-journal/journal-paths'
|
||||
import type { AgentSessionJournal } from '../agent-session-journal/journal-store'
|
||||
@@ -40,15 +41,27 @@ export async function restoreStructuredAgentSessionRead(
|
||||
sessionId
|
||||
})
|
||||
const loaded = loadJournal(journalDir, sessionId)
|
||||
if (!loaded || loaded.corrupt) {
|
||||
if (loaded?.corrupt) {
|
||||
return null
|
||||
}
|
||||
// A session still in the pre-SQLite format has no `journal.db` to load. Dropping
|
||||
// it here leaves it unpublished, which is also what prunes its tab out of the
|
||||
// saved workspace — so the chat disappears with nowhere to explain itself.
|
||||
if (!loaded && !findJournalFileFormatRemnant(journalDir)) {
|
||||
return null
|
||||
}
|
||||
const journal = await openAgentSessionJournal({
|
||||
identity: journalIdentityFor(record, params),
|
||||
journalDir,
|
||||
loaded
|
||||
// Omitted, not `null`: the store reads `null` as "replay already ran and
|
||||
// found nothing" and founds a fresh epoch. In process the probe above is the
|
||||
// previous statement, so the window is zero-width; this holds the line for a
|
||||
// database another process creates in between.
|
||||
...(loaded ? { loaded } : {})
|
||||
})
|
||||
// Read restore opens the journal and nothing else: no adapter call, so no provider child.
|
||||
// Read restore opens the journal and nothing else: no adapter call, so no
|
||||
// provider child. Opening it can still write — a session whose history is in
|
||||
// the old format founds its epoch and commits the row explaining that here.
|
||||
return {
|
||||
journal,
|
||||
params,
|
||||
|
||||
+149
@@ -0,0 +1,149 @@
|
||||
// Startup restore has to publish status, not just index the session.
|
||||
//
|
||||
// A tab nobody reopens after a restart still owes the sidebar a row. The host restores such a
|
||||
// session read-only, without a provider child, so the only thing that can surface its state is
|
||||
// the status publication the restore wiring makes.
|
||||
|
||||
import { mkdtemp, rm } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, describe, expect, it } from 'vitest'
|
||||
import type {
|
||||
AgentSessionMutationEnvelope,
|
||||
AgentSessionStatusEvent
|
||||
} from '../../../shared/agent-session-wire'
|
||||
import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope'
|
||||
import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store'
|
||||
import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter'
|
||||
import { StructuredAgentSessionHost } from './structured-agent-session-host'
|
||||
import {
|
||||
HOST_TEST_NOW as NOW,
|
||||
HOST_TEST_SESSION as SESSION,
|
||||
HOST_TEST_THREAD as THREAD,
|
||||
hostTestAttachParams,
|
||||
hostTestMessage,
|
||||
hostTestOperationId,
|
||||
resetHostTestOperationIds
|
||||
} from './structured-agent-session-host-test-data'
|
||||
|
||||
const CALLER = { callerKey: 'client-1' }
|
||||
|
||||
const hosts: StructuredAgentSessionHost[] = []
|
||||
let root = ''
|
||||
|
||||
function adapter(): StructuredAgentSessionAdapter {
|
||||
return {
|
||||
acquire: async ({ fence, spawnToken }) => ({
|
||||
process: { hostId: 'local', pid: 4242, processStartTimeMs: 1_700_000_000_000, spawnToken },
|
||||
link: {
|
||||
linkId: `link-${fence}`,
|
||||
handle: { provider: 'codex', threadId: THREAD },
|
||||
origin: 'created',
|
||||
mintedAtFence: fence,
|
||||
observedAt: NOW
|
||||
}
|
||||
}),
|
||||
dispatch: async () => ({
|
||||
state: 'accepted',
|
||||
providerIdentity: { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal: 1 }
|
||||
}),
|
||||
cancelTurn: async () => ({ cancelled: true }),
|
||||
answerPrompt: async () => undefined,
|
||||
setOption: async () => undefined
|
||||
}
|
||||
}
|
||||
|
||||
function createHost(store: AgentSessionRecordStore): StructuredAgentSessionHost {
|
||||
const host = new StructuredAgentSessionHost({
|
||||
store,
|
||||
adapter: adapter(),
|
||||
journalRoot: root,
|
||||
claimKeyId: 'key-1',
|
||||
mintSpawnToken: () => 'spawn-a',
|
||||
probeOwner: async () => ({
|
||||
outcome: 'indeterminate',
|
||||
reason: 'read does not need ownership'
|
||||
}),
|
||||
now: () => NOW
|
||||
})
|
||||
hosts.push(host)
|
||||
return host
|
||||
}
|
||||
|
||||
function sendEnvelope(
|
||||
store: AgentSessionRecordStore,
|
||||
fields: Record<string, unknown>
|
||||
): AgentSessionMutationEnvelope {
|
||||
return {
|
||||
sessionId: SESSION,
|
||||
clientOperationId: hostTestOperationId(),
|
||||
expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? 1,
|
||||
payloadFingerprint: computeAgentSessionPayloadFingerprint({
|
||||
method: 'agentSession.send',
|
||||
sessionId: SESSION,
|
||||
fields
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/** Persists one turn, then hands back a restarted host over the same directories. */
|
||||
async function restartWithPersistedTurn(): Promise<StructuredAgentSessionHost> {
|
||||
root = await mkdtemp(join(tmpdir(), 'orca-restart-status-'))
|
||||
resetHostTestOperationIds()
|
||||
const directory = join(root, 'store')
|
||||
const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' })
|
||||
const host = createHost(store)
|
||||
expect(await host.attach(CALLER, hostTestAttachParams(null))).toMatchObject({ ok: true })
|
||||
const body = hostTestMessage('persisted conversation')
|
||||
await host.send(CALLER, { envelope: sendEnvelope(store, { body }), body })
|
||||
await host.flushAllStreamedEvents()
|
||||
return createHost(await AgentSessionRecordStore.open({ directory, hostId: 'local' }))
|
||||
}
|
||||
|
||||
afterEach(async () => {
|
||||
await Promise.all(hosts.splice(0).map((host) => host.flushAllStreamedEvents()))
|
||||
await rm(root, { recursive: true, force: true })
|
||||
root = ''
|
||||
})
|
||||
|
||||
describe('structured session restart status publication', () => {
|
||||
// Served by the subscribe-time re-projection rather than the restore's own publish, so this
|
||||
// covers what a restored journal projects — not the restore wiring. The test below pins that.
|
||||
it('projects the persisted turn of a session restored without a provider', async () => {
|
||||
const restarted = await restartWithPersistedTurn()
|
||||
|
||||
await restarted.restoreReadableSessions()
|
||||
const events: AgentSessionStatusEvent[] = []
|
||||
restarted.subscribeStatus({ id: 'session-list', emit: (event) => events.push(event) })
|
||||
|
||||
expect(events).toEqual([
|
||||
{
|
||||
type: 'snapshot',
|
||||
sessions: [
|
||||
expect.objectContaining({
|
||||
sessionId: SESSION,
|
||||
workspaceId: 'workspace-1',
|
||||
agent: 'codex',
|
||||
status: 'idle',
|
||||
latestPrompt: 'persisted conversation'
|
||||
})
|
||||
]
|
||||
}
|
||||
])
|
||||
})
|
||||
|
||||
it('publishes a restored session to a list already sitting on the stream', async () => {
|
||||
const restarted = await restartWithPersistedTurn()
|
||||
const events: AgentSessionStatusEvent[] = []
|
||||
restarted.subscribeStatus({ id: 'session-list', emit: (event) => events.push(event) })
|
||||
expect(events).toEqual([{ type: 'snapshot', sessions: [] }])
|
||||
|
||||
await restarted.restoreReadableSessions()
|
||||
|
||||
// The restore wiring publishes; without it this list never hears about the session at all.
|
||||
expect(events.at(-1)).toEqual({
|
||||
type: 'status',
|
||||
session: expect.objectContaining({ sessionId: SESSION, status: 'idle' })
|
||||
})
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,242 @@
|
||||
import { mkdtemp, rm } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, beforeEach, describe, expect, it } from 'vitest'
|
||||
import type { AgentSessionStatusEvent } from '../../../shared/agent-session-wire'
|
||||
import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open'
|
||||
import { StructuredAgentSessionStatusFeed } from './structured-agent-session-status-feed'
|
||||
|
||||
const SESSION = 'status-session'
|
||||
const TURN_IDENTITY = {
|
||||
provider: 'codex',
|
||||
threadId: 'thread-1',
|
||||
turnId: 'turn-1',
|
||||
ordinal: 0
|
||||
} as const
|
||||
const USER_IDENTITY = {
|
||||
provider: 'codex',
|
||||
threadId: 'thread-1',
|
||||
turnId: 'turn-1',
|
||||
ordinal: 1
|
||||
} as const
|
||||
|
||||
let root: string
|
||||
const journals = createTrackedJournalOpener()
|
||||
|
||||
beforeEach(async () => {
|
||||
root = await mkdtemp(join(tmpdir(), 'orca-agent-status-feed-'))
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
await journals.closeAll()
|
||||
await rm(root, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
async function openJournal(sessionId = SESSION) {
|
||||
return journals.open({
|
||||
identity: {
|
||||
sessionId,
|
||||
workspaceId: 'workspace-1',
|
||||
hostId: 'local',
|
||||
agent: 'codex',
|
||||
providerHandle: { kind: 'codex', threadId: 'thread-1' }
|
||||
},
|
||||
journalDir: join(root, sessionId)
|
||||
})
|
||||
}
|
||||
|
||||
function indexed(session: { journal: Awaited<ReturnType<typeof openJournal>> }) {
|
||||
return {
|
||||
journal: session.journal,
|
||||
params: { location: { workspaceId: 'workspace-1' }, provider: 'codex' as const }
|
||||
}
|
||||
}
|
||||
|
||||
function feedFor(sessions: Map<string, { journal: Awaited<ReturnType<typeof openJournal>> }>) {
|
||||
let now = 1_000
|
||||
const feed = new StructuredAgentSessionStatusFeed({
|
||||
sessions: {
|
||||
get: (sessionId: string) => {
|
||||
const session = sessions.get(sessionId)
|
||||
return session ? indexed(session) : undefined
|
||||
},
|
||||
[Symbol.iterator]: function* () {
|
||||
for (const [sessionId, session] of sessions) {
|
||||
yield [sessionId, indexed(session)] as const
|
||||
}
|
||||
}
|
||||
} as unknown as ReadonlyMap<string, ReturnType<typeof indexed>>,
|
||||
getRecord: () => null,
|
||||
now: () => (now += 1)
|
||||
})
|
||||
const events: AgentSessionStatusEvent[] = []
|
||||
const dispose = feed.subscribe({ id: 'list-1', emit: (event) => events.push(event) })
|
||||
return { feed, events, dispose }
|
||||
}
|
||||
|
||||
describe('StructuredAgentSessionStatusFeed', () => {
|
||||
it('opens with every readable session and reports no status before a persisted turn', async () => {
|
||||
const journal = await openJournal()
|
||||
const { events } = feedFor(new Map([[SESSION, { journal }]]))
|
||||
|
||||
expect(events).toEqual([
|
||||
{
|
||||
type: 'snapshot',
|
||||
sessions: [
|
||||
{
|
||||
sessionId: SESSION,
|
||||
workspaceId: 'workspace-1',
|
||||
agent: 'codex',
|
||||
status: null,
|
||||
latestPrompt: '',
|
||||
updatedAt: expect.any(Number)
|
||||
}
|
||||
]
|
||||
}
|
||||
])
|
||||
})
|
||||
|
||||
it('publishes working, then idle once the running marker is tombstoned, and never a repeat', async () => {
|
||||
const journal = await openJournal()
|
||||
const { feed, events } = feedFor(new Map([[SESSION, { journal }]]))
|
||||
await journal.appendItem(
|
||||
USER_IDENTITY,
|
||||
{ kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'write a poem' }] },
|
||||
{ fence: 1 }
|
||||
)
|
||||
await journal.appendItem(
|
||||
TURN_IDENTITY,
|
||||
{ kind: 'status', text: 'Working', turnLifecycle: { turnId: 'turn-1', state: 'running' } },
|
||||
{ fence: 1 }
|
||||
)
|
||||
|
||||
feed.publish(SESSION)
|
||||
feed.publish(SESSION)
|
||||
expect(events.slice(1)).toEqual([
|
||||
{
|
||||
type: 'status',
|
||||
session: expect.objectContaining({
|
||||
sessionId: SESSION,
|
||||
status: 'working',
|
||||
latestPrompt: 'write a poem'
|
||||
})
|
||||
}
|
||||
])
|
||||
|
||||
await journal.appendTombstone(TURN_IDENTITY, { fence: 1 })
|
||||
feed.publish(SESSION)
|
||||
expect(events.at(-1)).toEqual({
|
||||
type: 'status',
|
||||
session: expect.objectContaining({ sessionId: SESSION, status: 'idle' })
|
||||
})
|
||||
expect(events).toHaveLength(3)
|
||||
})
|
||||
|
||||
it('reports a pending approval as attention', async () => {
|
||||
const journal = await openJournal()
|
||||
const { feed, events } = feedFor(new Map([[SESSION, { journal }]]))
|
||||
await journal.appendItem(
|
||||
USER_IDENTITY,
|
||||
{ kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'run it' }] },
|
||||
{ fence: 1 }
|
||||
)
|
||||
await journal.appendItem(
|
||||
TURN_IDENTITY,
|
||||
{
|
||||
kind: 'approval',
|
||||
title: 'Run command?',
|
||||
detail: null,
|
||||
options: [{ id: 'yes', label: 'Allow' }],
|
||||
resolution: { state: 'pending', selectedOptionId: null, resolvedBy: null, resolvedAt: null }
|
||||
},
|
||||
{ fence: 1 }
|
||||
)
|
||||
|
||||
feed.publish(SESSION)
|
||||
expect(events.at(-1)).toEqual({
|
||||
type: 'status',
|
||||
session: expect.objectContaining({ status: 'attention' })
|
||||
})
|
||||
})
|
||||
|
||||
it('keeps the last projection for an evicted session and serves it to a new subscriber', async () => {
|
||||
const journal = await openJournal()
|
||||
const sessions = new Map([[SESSION, { journal }]])
|
||||
const { feed, events } = feedFor(sessions)
|
||||
await journal.appendItem(
|
||||
USER_IDENTITY,
|
||||
{ kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hello' }] },
|
||||
{ fence: 1 }
|
||||
)
|
||||
feed.publish(SESSION)
|
||||
expect(events.at(-1)).toEqual({
|
||||
type: 'status',
|
||||
session: expect.objectContaining({ status: 'idle', latestPrompt: 'hello' })
|
||||
})
|
||||
|
||||
// Eviction drops the host's index entry; the projection it already made stays true.
|
||||
sessions.delete(SESSION)
|
||||
feed.publish(SESSION)
|
||||
const late: AgentSessionStatusEvent[] = []
|
||||
feed.subscribe({ id: 'list-late', emit: (event) => late.push(event) })
|
||||
|
||||
expect(events).toHaveLength(2)
|
||||
expect(late).toEqual([
|
||||
{
|
||||
type: 'snapshot',
|
||||
sessions: [expect.objectContaining({ sessionId: SESSION, status: 'idle' })]
|
||||
}
|
||||
])
|
||||
})
|
||||
|
||||
it('tells the sitting subscribers about a change a new subscriber re-projected', async () => {
|
||||
const journal = await openJournal()
|
||||
const { feed, events } = feedFor(new Map([[SESSION, { journal }]]))
|
||||
// Journal appends and the feed's publish are separate queue submissions, so the journal
|
||||
// can already hold the turn when a second client connects and re-projects it.
|
||||
await journal.appendItem(
|
||||
USER_IDENTITY,
|
||||
{ kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hello' }] },
|
||||
{ fence: 1 }
|
||||
)
|
||||
|
||||
const late: AgentSessionStatusEvent[] = []
|
||||
feed.subscribe({ id: 'list-late', emit: (event) => late.push(event) })
|
||||
|
||||
expect(events.at(-1)).toEqual({
|
||||
type: 'status',
|
||||
session: expect.objectContaining({ status: 'idle', latestPrompt: 'hello' })
|
||||
})
|
||||
// The arriving subscriber reads that same state once, from its snapshot.
|
||||
expect(late).toEqual([
|
||||
{
|
||||
type: 'snapshot',
|
||||
sessions: [expect.objectContaining({ status: 'idle', latestPrompt: 'hello' })]
|
||||
}
|
||||
])
|
||||
// The cache is not left holding a value nobody was told about.
|
||||
feed.publish(SESSION)
|
||||
expect(events).toHaveLength(2)
|
||||
})
|
||||
|
||||
it('ends a closed subscriber and keeps publishing to the rest', async () => {
|
||||
const journal = await openJournal()
|
||||
const { feed, events, dispose } = feedFor(new Map([[SESSION, { journal }]]))
|
||||
const others: AgentSessionStatusEvent[] = []
|
||||
feed.subscribe({ id: 'list-2', emit: (event) => others.push(event) })
|
||||
|
||||
dispose()
|
||||
await journal.appendItem(
|
||||
USER_IDENTITY,
|
||||
{ kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hello' }] },
|
||||
{ fence: 1 }
|
||||
)
|
||||
feed.publish(SESSION)
|
||||
|
||||
expect(events.at(-1)).toEqual({ type: 'end' })
|
||||
expect(others.at(-1)).toEqual({
|
||||
type: 'status',
|
||||
session: expect.objectContaining({ status: 'idle' })
|
||||
})
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,130 @@
|
||||
// The host's answer to "what is every structured session doing", fanned out to session lists.
|
||||
//
|
||||
// A client used to learn whether a turn was running by replaying the journal through its own
|
||||
// reducer, which tied the answer to whichever surface happened to hold a reader open: hide the
|
||||
// chat and the sidebar froze on the last thing it had heard. The host always has the journal, so
|
||||
// it projects the status once per journal publication and sends only the changes.
|
||||
//
|
||||
// The last projection is kept after the session's provider child is evicted: an idle session is
|
||||
// still idle without a process, and a renderer that reloads must not lose every settled row until
|
||||
// each chat is reopened. Restart is the one boundary that forgets, and restoring readable sessions
|
||||
// republishes them.
|
||||
|
||||
import { agentProviderSessionsEqual } from '../../../shared/agent-session-resume'
|
||||
import type { AgentSessionRecord } from '../../../shared/agent-session-record'
|
||||
import type {
|
||||
AgentSessionStatusEvent,
|
||||
AgentSessionStatusSummary
|
||||
} from '../../../shared/agent-session-wire'
|
||||
import { projectStructuredAgentSessionStatusSummary } from '../../../shared/structured-agent-session-projection'
|
||||
import type { AgentSessionJournal } from '../agent-session-journal/journal-store'
|
||||
import { structuredAgentSessionProviderSessionMetadata } from './structured-agent-session-history-result'
|
||||
|
||||
export type StructuredAgentSessionStatusSubscriber = {
|
||||
id: string
|
||||
emit: (event: AgentSessionStatusEvent) => void
|
||||
}
|
||||
|
||||
type StatusFeedSession = {
|
||||
journal: AgentSessionJournal
|
||||
params: { location: { workspaceId: string }; provider: AgentSessionRecord['provider'] }
|
||||
}
|
||||
|
||||
export type StructuredAgentSessionStatusFeedDeps = {
|
||||
sessions: ReadonlyMap<string, StatusFeedSession>
|
||||
getRecord: (sessionId: string) => AgentSessionRecord | null
|
||||
now: () => number
|
||||
}
|
||||
|
||||
function summariesEqual(a: AgentSessionStatusSummary, b: AgentSessionStatusSummary): boolean {
|
||||
return (
|
||||
a.workspaceId === b.workspaceId &&
|
||||
a.agent === b.agent &&
|
||||
a.status === b.status &&
|
||||
a.latestPrompt === b.latestPrompt &&
|
||||
agentProviderSessionsEqual(undefined, a.providerSession, b.providerSession)
|
||||
)
|
||||
}
|
||||
|
||||
export class StructuredAgentSessionStatusFeed {
|
||||
private readonly subscribers = new Map<string, StructuredAgentSessionStatusSubscriber>()
|
||||
private readonly published = new Map<string, AgentSessionStatusSummary>()
|
||||
|
||||
constructor(private readonly deps: StructuredAgentSessionStatusFeedDeps) {}
|
||||
|
||||
/** Opens with every session this host has projected, live ones re-read, then only changes. */
|
||||
subscribe(subscriber: StructuredAgentSessionStatusSubscriber): () => void {
|
||||
// Re-project before registering: a change found here has to reach the subscribers that
|
||||
// already read the old value, and the arriving one carries it in its snapshot instead.
|
||||
for (const [sessionId] of this.deps.sessions) {
|
||||
this.publish(sessionId)
|
||||
}
|
||||
this.subscribers.set(subscriber.id, subscriber)
|
||||
this.emit(subscriber, { type: 'snapshot', sessions: [...this.published.values()] })
|
||||
return () => this.unsubscribe(subscriber.id)
|
||||
}
|
||||
|
||||
unsubscribe(id: string): void {
|
||||
const subscriber = this.subscribers.get(id)
|
||||
if (!subscriber) {
|
||||
return
|
||||
}
|
||||
this.subscribers.delete(id)
|
||||
try {
|
||||
subscriber.emit({ type: 'end' })
|
||||
} catch {
|
||||
// The transport is already gone; teardown must remain idempotent.
|
||||
}
|
||||
}
|
||||
|
||||
/** Re-projects one session after its journal changed; equal projections are not re-sent. */
|
||||
publish(sessionId: string, journal?: AgentSessionJournal): void {
|
||||
const session = this.deps.sessions.get(sessionId)
|
||||
if (!session) {
|
||||
return
|
||||
}
|
||||
const summary = this.summaryFor(sessionId, session, journal ?? session.journal)
|
||||
const previous = this.published.get(sessionId)
|
||||
if (previous && summariesEqual(previous, summary)) {
|
||||
return
|
||||
}
|
||||
this.published.set(sessionId, summary)
|
||||
this.broadcast({ type: 'status', session: summary })
|
||||
}
|
||||
|
||||
private summaryFor(
|
||||
sessionId: string,
|
||||
session: StatusFeedSession,
|
||||
journal: AgentSessionJournal
|
||||
): AgentSessionStatusSummary {
|
||||
// An unreadable journal projects as "no turn": the chat itself shows the reset.
|
||||
const items = journal.isReadOnly ? [] : journal.snapshot().items
|
||||
const providerSession = structuredAgentSessionProviderSessionMetadata(
|
||||
this.deps.getRecord(sessionId)
|
||||
)
|
||||
return {
|
||||
sessionId,
|
||||
workspaceId: session.params.location.workspaceId,
|
||||
agent: session.params.provider,
|
||||
...projectStructuredAgentSessionStatusSummary(items),
|
||||
...(providerSession ? { providerSession } : {}),
|
||||
updatedAt: this.deps.now()
|
||||
}
|
||||
}
|
||||
|
||||
private broadcast(event: AgentSessionStatusEvent): void {
|
||||
// A Map skips entries deleted mid-iteration, so a failing subscriber can drop itself here.
|
||||
for (const subscriber of this.subscribers.values()) {
|
||||
this.emit(subscriber, event)
|
||||
}
|
||||
}
|
||||
|
||||
/** A dead transport must not poison every later publication. */
|
||||
private emit(subscriber: StructuredAgentSessionStatusSubscriber, event: AgentSessionStatusEvent) {
|
||||
try {
|
||||
subscriber.emit(event)
|
||||
} catch {
|
||||
this.subscribers.delete(subscriber.id)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -5,6 +5,7 @@ import { afterEach, beforeEach, describe, expect, it } from 'vitest'
|
||||
import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types'
|
||||
import type {
|
||||
AgentSessionHandoffStatus,
|
||||
AgentSessionStatusEvent,
|
||||
AgentSessionSubscribeEvent
|
||||
} from '../../../shared/agent-session-wire'
|
||||
import {
|
||||
@@ -16,6 +17,7 @@ import { journalDatabaseFile } from '../agent-session-journal/journal-paths'
|
||||
import { insertJournalRow } from '../agent-session-journal/journal-row-table'
|
||||
import type { JournalRow } from '../agent-session-journal/journal-row-schema'
|
||||
import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open'
|
||||
import { StructuredAgentSessionStatusFeed } from './structured-agent-session-status-feed'
|
||||
import { AgentSessionSubscribers } from './structured-agent-session-subscribers'
|
||||
|
||||
const SESSION = 'subscriber-session'
|
||||
@@ -70,6 +72,96 @@ describe('AgentSessionSubscribers', () => {
|
||||
])
|
||||
})
|
||||
|
||||
it('reports every content publication to the journal hook, subscribed or not', async () => {
|
||||
const journal = await journals.open({
|
||||
identity: {
|
||||
sessionId: SESSION,
|
||||
workspaceId: 'workspace-1',
|
||||
hostId: 'local',
|
||||
agent: 'codex',
|
||||
providerHandle: { kind: 'codex', threadId: 'thread-1' }
|
||||
},
|
||||
journalDir: join(root, 'hook-journal')
|
||||
})
|
||||
const published: string[] = []
|
||||
const subscribers = new AgentSessionSubscribers({
|
||||
onJournalPublished: (sessionId, published_journal) => {
|
||||
expect(published_journal).toBe(journal)
|
||||
published.push(sessionId)
|
||||
}
|
||||
})
|
||||
|
||||
subscribers.publish(SESSION, journal)
|
||||
subscribers.reset(SESSION, journal, 'epoch_changed', 1)
|
||||
subscribers.snapshot(SESSION, journal, 1)
|
||||
subscribers.handoff(SESSION, 1, {
|
||||
owner: 'native',
|
||||
direction: null,
|
||||
phase: 'idle',
|
||||
stage: null,
|
||||
operationId: null
|
||||
})
|
||||
|
||||
expect(published).toEqual([SESSION, SESSION, SESSION])
|
||||
})
|
||||
|
||||
it('settles a session nobody is reading, from running to idle', async () => {
|
||||
// The defect this whole feed exists for: status used to come from a transcript reader, so a
|
||||
// session with no open pane had no reader and froze on whatever it last said. Nothing here
|
||||
// ever calls `subscribers.open`.
|
||||
const journal = await journals.open({
|
||||
identity: {
|
||||
sessionId: SESSION,
|
||||
workspaceId: 'workspace-1',
|
||||
hostId: 'local',
|
||||
agent: 'codex',
|
||||
providerHandle: { kind: 'codex', threadId: 'thread-1' }
|
||||
},
|
||||
journalDir: join(root, 'unread-journal')
|
||||
})
|
||||
const statusFeed = new StructuredAgentSessionStatusFeed({
|
||||
sessions: new Map([
|
||||
[
|
||||
SESSION,
|
||||
{ journal, params: { location: { workspaceId: 'workspace-1' }, provider: 'codex' } }
|
||||
]
|
||||
]),
|
||||
getRecord: () => null,
|
||||
now: () => 1_000
|
||||
})
|
||||
const subscribers = new AgentSessionSubscribers({
|
||||
onJournalPublished: (sessionId, published) => statusFeed.publish(sessionId, published)
|
||||
})
|
||||
const statuses: AgentSessionStatusEvent[] = []
|
||||
statusFeed.subscribe({ id: 'session-list', emit: (event) => statuses.push(event) })
|
||||
const turn = { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 0 } as const
|
||||
|
||||
await journal.appendItem(
|
||||
{ ...turn, ordinal: 1 },
|
||||
{ kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'write a poem' }] },
|
||||
{ fence: 1 }
|
||||
)
|
||||
await journal.appendItem(
|
||||
turn,
|
||||
{ kind: 'status', text: 'Working', turnLifecycle: { turnId: 'turn-1', state: 'running' } },
|
||||
{ fence: 1 }
|
||||
)
|
||||
subscribers.publish(SESSION, journal)
|
||||
|
||||
expect(statuses.at(-1)).toEqual({
|
||||
type: 'status',
|
||||
session: expect.objectContaining({ status: 'working', latestPrompt: 'write a poem' })
|
||||
})
|
||||
|
||||
await journal.appendTombstone(turn, { fence: 1 })
|
||||
subscribers.publish(SESSION, journal)
|
||||
|
||||
expect(statuses.at(-1)).toEqual({
|
||||
type: 'status',
|
||||
session: expect.objectContaining({ status: 'idle' })
|
||||
})
|
||||
})
|
||||
|
||||
it('publishes handoff-only changes without serializing a transcript snapshot', async () => {
|
||||
const journal = await journals.open({
|
||||
identity: {
|
||||
|
||||
@@ -36,9 +36,17 @@ type Subscriber = {
|
||||
fence: number
|
||||
}
|
||||
|
||||
export type AgentSessionSubscribersHooks = {
|
||||
/** Fires after any publication that can change journal content, whether or not anyone
|
||||
* is subscribed to the transcript: session lists project status from this same edge. */
|
||||
onJournalPublished?: (sessionId: string, journal: AgentSessionJournal) => void
|
||||
}
|
||||
|
||||
export class AgentSessionSubscribers {
|
||||
private readonly bySession = new Map<string, Map<string, Subscriber>>()
|
||||
|
||||
constructor(private readonly hooks: AgentSessionSubscribersHooks = {}) {}
|
||||
|
||||
/** Opens the stream with a bounded tail page or, when the client's cursor
|
||||
* still resolves, with the rows it missed. Returns the disposer. */
|
||||
open(input: {
|
||||
@@ -99,6 +107,7 @@ export class AgentSessionSubscribers {
|
||||
for (const subscriber of this.subscribers(sessionId)) {
|
||||
this.deliver(subscriber, journal)
|
||||
}
|
||||
this.hooks.onJournalPublished?.(sessionId, journal)
|
||||
}
|
||||
|
||||
/** Force every subscriber back to a bounded tail page — recovery, epoch
|
||||
@@ -123,6 +132,7 @@ export class AgentSessionSubscribers {
|
||||
subscriber.cursor = page.liveCursor ?? page.window.nextCursor
|
||||
subscriber.fence = fence
|
||||
}
|
||||
this.hooks.onJournalPublished?.(sessionId, journal)
|
||||
}
|
||||
|
||||
snapshot(
|
||||
@@ -143,6 +153,7 @@ export class AgentSessionSubscribers {
|
||||
subscriber.cursor = page.liveCursor ?? page.window.nextCursor
|
||||
subscriber.fence = fence
|
||||
}
|
||||
this.hooks.onJournalPublished?.(sessionId, journal)
|
||||
}
|
||||
|
||||
handoff(sessionId: string, fence: number, handoff: AgentSessionHandoffStatus): void {
|
||||
|
||||
@@ -104,4 +104,40 @@ describe('performCancel', () => {
|
||||
expect(cancelTurn).not.toHaveBeenCalled()
|
||||
expect(journal.snapshot().items).toEqual([])
|
||||
})
|
||||
|
||||
it('routes one background task id without interrupting the foreground turn or writing a row', async () => {
|
||||
root = await mkdtemp(join(tmpdir(), 'orca-background-task-targeted-cancel-'))
|
||||
const journal = await journals.open({ identity: IDENTITY, journalDir: root })
|
||||
const cancelTurn = vi.fn(async () => ({ cancelled: true }))
|
||||
const stopBackgroundTasks = vi.fn(async () => ({ cancelled: true }))
|
||||
const ctx: AgentSessionTurnContext = {
|
||||
sessionId: 'session-1',
|
||||
journal,
|
||||
fence: 1,
|
||||
adapter: { cancelTurn, stopBackgroundTasks } as unknown as StructuredAgentSessionAdapter,
|
||||
persistOptions: async () => undefined,
|
||||
resolvedBy: 'client-1',
|
||||
publish: vi.fn(),
|
||||
now: () => 1
|
||||
}
|
||||
|
||||
const result = await performCancel(ctx, {
|
||||
clientOperationId: 'cancel-background-task-2',
|
||||
turnId: 'background-tasks',
|
||||
scope: 'background-tasks',
|
||||
taskId: 'task-2'
|
||||
})
|
||||
|
||||
expect(result).toEqual({
|
||||
ok: true,
|
||||
value: { turnId: 'background-tasks', cancelled: true }
|
||||
})
|
||||
expect(stopBackgroundTasks).toHaveBeenCalledWith({
|
||||
sessionId: 'session-1',
|
||||
fence: 1,
|
||||
taskId: 'task-2'
|
||||
})
|
||||
expect(cancelTurn).not.toHaveBeenCalled()
|
||||
expect(journal.snapshot().items).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
@@ -150,6 +150,7 @@ export async function performCancel(
|
||||
clientOperationId: string
|
||||
turnId: string
|
||||
scope?: 'background-tasks'
|
||||
taskId?: string
|
||||
}
|
||||
): Promise<TurnOutcome<AgentSessionCancelResult>> {
|
||||
let cancelled = false
|
||||
@@ -159,7 +160,8 @@ export async function performCancel(
|
||||
? (
|
||||
await ctx.adapter.stopBackgroundTasks?.({
|
||||
sessionId: ctx.sessionId,
|
||||
fence: ctx.fence
|
||||
fence: ctx.fence,
|
||||
...(input.taskId ? { taskId: input.taskId } : {})
|
||||
})
|
||||
)?.cancelled === true
|
||||
: (
|
||||
|
||||
@@ -172,6 +172,7 @@ describe('LocalPtyProvider', () => {
|
||||
|
||||
expect(second).toEqual({
|
||||
id: 'serve-session-1',
|
||||
incarnationId: first.incarnationId,
|
||||
pid: 12345,
|
||||
isReattach: true,
|
||||
// Why published: this attach really moved the PTY, unlike daemon/relay attach, so main
|
||||
@@ -220,7 +221,11 @@ describe('LocalPtyProvider', () => {
|
||||
attachOnly: true
|
||||
})
|
||||
|
||||
expect(result).toMatchObject({ id: first.id, isReattach: true })
|
||||
expect(result).toMatchObject({
|
||||
id: first.id,
|
||||
incarnationId: first.incarnationId,
|
||||
isReattach: true
|
||||
})
|
||||
expect(spawnMock).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import type { PtySpawnResult } from './types'
|
||||
import {
|
||||
pendingLocalPtySpawns,
|
||||
ptyIncarnations,
|
||||
ptyProcesses,
|
||||
ptyWslDistroById,
|
||||
type PendingLocalPtySpawn
|
||||
@@ -60,6 +61,7 @@ export function reattachLocalPty(id: string, cols: number, rows: number): PtySpa
|
||||
}
|
||||
return {
|
||||
id,
|
||||
...(ptyIncarnations.has(id) ? { incarnationId: ptyIncarnations.get(id) } : {}),
|
||||
pid: existing.pid,
|
||||
...(ptyWslDistroById.has(id) ? { wslDistro: ptyWslDistroById.get(id) ?? null } : {}),
|
||||
isReattach: true,
|
||||
|
||||
@@ -599,6 +599,46 @@ describe('a structured Claude session over agentSession.*', () => {
|
||||
`claude:${PROVIDER_SESSION}:assistant-leaf`
|
||||
)
|
||||
|
||||
claude.live().handlers.onMessage?.({
|
||||
type: 'system',
|
||||
subtype: 'background_tasks_changed',
|
||||
session_id: PROVIDER_SESSION,
|
||||
uuid: 'background-roster',
|
||||
tasks: [
|
||||
{ task_id: 'task-one', task_type: 'local_agent', description: 'First task' },
|
||||
{ task_id: 'task-two', task_type: 'local_bash', description: 'Second task' }
|
||||
]
|
||||
})
|
||||
const itemsBeforeTaskStop = itemsOf(stream)
|
||||
const targetedStopFields = {
|
||||
turnId: 'background-tasks',
|
||||
scope: 'background-tasks',
|
||||
taskId: 'task-two'
|
||||
}
|
||||
await expect(
|
||||
ok('agentSession.cancel', {
|
||||
envelope: envelope('agentSession.cancel', targetedStopFields, created.fence),
|
||||
...targetedStopFields
|
||||
})
|
||||
).resolves.toMatchObject({ turnId: 'background-tasks', cancelled: true })
|
||||
expect(claude.live().calls.filter((entry) => entry.subtype === 'stop_task')).toEqual([
|
||||
{ subtype: 'stop_task', params: { taskId: 'task-two' } }
|
||||
])
|
||||
expect(itemsOf(stream)).toEqual(itemsBeforeTaskStop)
|
||||
|
||||
const staleStopFields = {
|
||||
turnId: 'background-tasks',
|
||||
scope: 'background-tasks',
|
||||
taskId: 'task-stale'
|
||||
}
|
||||
await expect(
|
||||
ok('agentSession.cancel', {
|
||||
envelope: envelope('agentSession.cancel', staleStopFields, created.fence),
|
||||
...staleStopFields
|
||||
})
|
||||
).resolves.toMatchObject({ turnId: 'background-tasks', cancelled: false })
|
||||
expect(claude.live().calls.filter((entry) => entry.subtype === 'stop_task')).toHaveLength(1)
|
||||
|
||||
const answeredPermission = Promise.resolve(
|
||||
claude.live().handlers.canUseTool?.('Bash', { command: 'ls' }, {
|
||||
requestId: 'permission-1',
|
||||
|
||||
@@ -176,8 +176,17 @@ describe('OrcaRuntimeService.fetchRemoteWithCache', () => {
|
||||
expect(caches.fetchLastCompletedAt.has('/repo/cache-0::origin')).toBe(false)
|
||||
})
|
||||
|
||||
it.each(['main', 'a'.repeat(40), 'refs/remotes/main', ''])(
|
||||
'does not launch Git for a base without a remote/branch separator: %s',
|
||||
it.each([
|
||||
'main',
|
||||
'a'.repeat(40),
|
||||
'refs/remotes/main',
|
||||
'',
|
||||
'origin/',
|
||||
'/main',
|
||||
'refs/remotes/origin/',
|
||||
'refs/remotes//main'
|
||||
])(
|
||||
'does not launch Git for a base without both remote and branch components: %s',
|
||||
async (base) => {
|
||||
const runtime = new OrcaRuntimeService(null)
|
||||
await expect(runtime.resolveRemoteTrackingBase('/repo/e', base)).resolves.toBeNull()
|
||||
|
||||
@@ -153,6 +153,11 @@ export class OrcaRuntimeWithRefreshRepoWorktreeScan extends OrcaRuntimeWithListK
|
||||
}
|
||||
}
|
||||
|
||||
invalidateWorktreeCatalog(repoId: string): void {
|
||||
this.invalidateResolvedWorktreeCache()
|
||||
this.invalidateWorktreeScanCacheForRepo(repoId)
|
||||
}
|
||||
|
||||
protected invalidateSshWorktreeScanCacheInternal(targetId: string): void {
|
||||
const repos = this.store?.getRepos() ?? []
|
||||
const affectedRepos = repos.filter((repo) => getRepoSshConnectionId(repo) === targetId)
|
||||
|
||||
@@ -50,7 +50,13 @@ describe('OrcaRuntimeService', () => {
|
||||
pushTarget: { remoteName: 'origin', branchName: 'feature/fix' }
|
||||
})
|
||||
|
||||
expect(getBranchConflictKind).toHaveBeenCalledWith(TEST_REPO_PATH, 'feature/fix', 'abc123')
|
||||
expect(getBranchConflictKind).toHaveBeenCalledWith(
|
||||
TEST_REPO_PATH,
|
||||
'feature/fix',
|
||||
'abc123',
|
||||
{},
|
||||
undefined
|
||||
)
|
||||
expect(getPRForBranchMock).toHaveBeenCalledWith(TEST_REPO_PATH, 'feature/fix')
|
||||
expect(addWorktree).toHaveBeenCalledWith(
|
||||
TEST_REPO_PATH,
|
||||
@@ -165,7 +171,9 @@ describe('OrcaRuntimeService', () => {
|
||||
expect(getBranchConflictKind).toHaveBeenCalledWith(
|
||||
TEST_REPO_PATH,
|
||||
'feature/bitbucket',
|
||||
'abc123'
|
||||
'abc123',
|
||||
{},
|
||||
undefined
|
||||
)
|
||||
expect(getHostedReviewForBranchMock).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
|
||||
@@ -533,10 +533,14 @@ describe('OrcaRuntimeService', () => {
|
||||
branchNameOverride: 'feature/something'
|
||||
})
|
||||
|
||||
// Why: an explicit branch override adopts the local branch before the conflict
|
||||
// probe, so no lazy adoption callback is handed to getBranchConflictKind.
|
||||
expect(getBranchConflictKind).toHaveBeenCalledWith(
|
||||
TEST_REPO_PATH,
|
||||
'feature/something',
|
||||
'origin/feature/something'
|
||||
'origin/feature/something',
|
||||
{},
|
||||
undefined
|
||||
)
|
||||
expect(addWorktree).toHaveBeenCalledWith(
|
||||
TEST_REPO_PATH,
|
||||
|
||||
@@ -376,7 +376,18 @@ describe('OrcaRuntimeService', () => {
|
||||
TEST_REPO_PATH,
|
||||
'runtime-wsl',
|
||||
'origin/main',
|
||||
{ wslDistro: 'Ubuntu' }
|
||||
{ wslDistro: 'Ubuntu' },
|
||||
expect.any(Function)
|
||||
)
|
||||
// Why: the lazy adoption callback is only invoked when the conflict probe
|
||||
// sees a local ref, so drive it here to prove adoption also routes via WSL.
|
||||
const adoptLocalBranch = vi
|
||||
.mocked(getBranchConflictKind)
|
||||
.mock.calls.findLast((call) => call[1] === 'runtime-wsl')?.[4]
|
||||
await expect(adoptLocalBranch?.()).resolves.toBe(false)
|
||||
expect(gitSpy).toHaveBeenCalledWith(
|
||||
['rev-parse', '--verify', '--quiet', 'refs/heads/runtime-wsl^{commit}'],
|
||||
{ cwd: TEST_REPO_PATH, wslDistro: 'Ubuntu' }
|
||||
)
|
||||
expect(getPRForBranchMock).toHaveBeenCalledWith(
|
||||
TEST_REPO_PATH,
|
||||
|
||||
@@ -8,6 +8,7 @@ import { claimDispatchContextRow } from '../dispatch-row-writer'
|
||||
import { recordedCreatorIdentity, type DispatchCreator } from '../dispatch-depth'
|
||||
import type { OrchestrationDb } from '../orchestration-db'
|
||||
import { transitionLifecycleWithDb } from '../lifecycle-transition'
|
||||
import { taskNotFoundError, taskNotStartableError } from '../../task-dispatch-refusal'
|
||||
|
||||
export function createDispatchContext(
|
||||
this: OrchestrationDb,
|
||||
@@ -27,10 +28,14 @@ export function createDispatchContext(
|
||||
const depth = this.resolveChildDispatchDepth(params.creator, params.maxDepth)
|
||||
const task = this.getTask(taskId)
|
||||
if (!task) {
|
||||
throw new Error(`Task not found: ${taskId}`)
|
||||
throw taskNotFoundError(`Task not found: ${taskId}`, { taskId })
|
||||
}
|
||||
if (task.status !== 'ready') {
|
||||
throw new Error(`Task ${taskId} is ${task.status}; only ready tasks can be dispatched`)
|
||||
throw taskNotStartableError(
|
||||
this,
|
||||
`Task ${taskId} is ${task.status}; only ready tasks can be dispatched`,
|
||||
task
|
||||
)
|
||||
}
|
||||
|
||||
// Why: lock on pane identity too, so a reminted handle can't open a second concurrent dispatch on the same pane.
|
||||
@@ -76,9 +81,12 @@ export function createDispatchContext(
|
||||
`Terminal ${assigneeHandle} already has an active dispatch (${occupied.id} for task ${occupied.task_id})`
|
||||
)
|
||||
}
|
||||
throw new Error(
|
||||
`Task ${taskId} is ${current?.status ?? 'missing'}; only ready tasks can be dispatched`
|
||||
)
|
||||
// Why: the atomic claim lost to a concurrent status change; report it with the same
|
||||
// typed receipt as the precheck so the loser can recover instead of reading runtime_error.
|
||||
const message = `Task ${taskId} is ${current?.status ?? 'missing'}; only ready tasks can be dispatched`
|
||||
throw current
|
||||
? taskNotStartableError(this, message, current)
|
||||
: taskNotFoundError(message, { taskId })
|
||||
}
|
||||
transitionLifecycleWithDb(this.db, {
|
||||
entity: 'task',
|
||||
|
||||
@@ -7,6 +7,7 @@ import type { OrchestrationDb } from '../orchestration-db'
|
||||
import { insertStartingDispatchContextRow } from '../dispatch-row-writer'
|
||||
import { recordedCreatorIdentity, type DispatchCreator } from '../dispatch-depth'
|
||||
import { transitionLifecycleWithDb } from '../lifecycle-transition'
|
||||
import { taskNotFoundError, taskNotStartableError } from '../../task-dispatch-refusal'
|
||||
|
||||
export function createStartingWorkerDispatch(
|
||||
this: OrchestrationDb,
|
||||
@@ -84,7 +85,9 @@ export function createStartingWorkerDispatch(
|
||||
})
|
||||
: undefined
|
||||
if (!task) {
|
||||
throw new OrchestrationError('task_not_found', `Task ${params.taskId} was not found.`)
|
||||
// Why: `--spec` creates the Task inline, so a missing row here always names an explicit id.
|
||||
const taskId = params.taskId ?? ''
|
||||
throw taskNotFoundError(`Task ${taskId} was not found.`, { taskId })
|
||||
}
|
||||
if (params.retryOf) {
|
||||
const prior = this.getDispatchContextById(params.retryOf)
|
||||
@@ -101,15 +104,18 @@ export function createStartingWorkerDispatch(
|
||||
!priorSettled ||
|
||||
!['failed', 'blocked'].includes(task.status)
|
||||
) {
|
||||
throw new OrchestrationError(
|
||||
'task_not_startable',
|
||||
`Task ${task.id} cannot retry from Dispatch ${params.retryOf}.`
|
||||
throw taskNotStartableError(
|
||||
this,
|
||||
`Task ${task.id} cannot retry from Dispatch ${params.retryOf}.`,
|
||||
task,
|
||||
params.retryOf
|
||||
)
|
||||
}
|
||||
} else if (task.status !== 'ready') {
|
||||
throw new OrchestrationError(
|
||||
'task_not_startable',
|
||||
`Task ${task.id} is ${task.status}; only a ready Task can start.`
|
||||
throw taskNotStartableError(
|
||||
this,
|
||||
`Task ${task.id} is ${task.status}; only a ready Task can start.`,
|
||||
task
|
||||
)
|
||||
}
|
||||
|
||||
|
||||
@@ -196,7 +196,12 @@ describe('OrchestrationDb worker Dispatch state', () => {
|
||||
payloadHash: 'payload_hash'
|
||||
}
|
||||
})
|
||||
).toThrow('was not found')
|
||||
).toThrowError(
|
||||
expect.objectContaining({
|
||||
code: 'task_not_found',
|
||||
message: 'Task task_missing was not found.'
|
||||
})
|
||||
)
|
||||
expect(d.getMutationReceipt('caller_fingerprint', 'invalid_worker_start')).toBeUndefined()
|
||||
})
|
||||
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
import type { OrchestrationDb } from './db'
|
||||
import { OrchestrationError } from './orchestration-error'
|
||||
import type { TaskRow } from './types'
|
||||
import {
|
||||
injectRejectedRefusal,
|
||||
taskNotFoundRefusal,
|
||||
taskNotStartableRefusal,
|
||||
type DispatchRefusalReceipt,
|
||||
type InjectRejectionReason
|
||||
} from '../../../shared/orchestration-dispatch-refusal-contract'
|
||||
|
||||
// Why: each site keeps the exact message it published before; only the code and data are shared.
|
||||
|
||||
export function taskNotFoundError(
|
||||
message: string,
|
||||
detail: { taskId: string; runId?: string }
|
||||
): OrchestrationError {
|
||||
return toError(taskNotFoundRefusal(message, detail))
|
||||
}
|
||||
|
||||
export function taskNotStartableError(
|
||||
db: OrchestrationDb,
|
||||
message: string,
|
||||
task: TaskRow,
|
||||
retryOf?: string
|
||||
): OrchestrationError {
|
||||
return toError(
|
||||
taskNotStartableRefusal(message, {
|
||||
taskId: task.id,
|
||||
status: task.status,
|
||||
unmetDependencies: unmetTaskDependencies(db, task),
|
||||
...(retryOf ? { retryOf } : {})
|
||||
})
|
||||
)
|
||||
}
|
||||
|
||||
export function injectRejectedError(
|
||||
terminal: string,
|
||||
reason: InjectRejectionReason
|
||||
): OrchestrationError {
|
||||
return toError(injectRejectedRefusal(terminal, reason))
|
||||
}
|
||||
|
||||
function toError(receipt: DispatchRefusalReceipt): OrchestrationError {
|
||||
return new OrchestrationError(receipt.code, receipt.message, receipt.data)
|
||||
}
|
||||
|
||||
function unmetTaskDependencies(db: OrchestrationDb, task: TaskRow): string[] {
|
||||
let deps: unknown
|
||||
try {
|
||||
deps = JSON.parse(task.deps)
|
||||
} catch {
|
||||
return []
|
||||
}
|
||||
if (!Array.isArray(deps)) {
|
||||
return []
|
||||
}
|
||||
return deps.filter(
|
||||
(dep): dep is string => typeof dep === 'string' && db.getTask(dep)?.status !== 'completed'
|
||||
)
|
||||
}
|
||||
@@ -86,6 +86,7 @@ const STRUCTURED_RUNTIME_PASSTHROUGH_CODES: ReadonlySet<string> = new Set([
|
||||
'consumer_fenced',
|
||||
'task_not_found',
|
||||
'task_not_startable',
|
||||
'inject_rejected',
|
||||
'dispatch_not_found',
|
||||
'dispatch_run_mismatch',
|
||||
'terminal_not_found',
|
||||
|
||||
@@ -0,0 +1,242 @@
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import { ORCHESTRATION_CONTRACT_VERSION } from '../../../../shared/protocol-version'
|
||||
import {
|
||||
buildInjectRejectionMessage,
|
||||
injectRejectedRefusal,
|
||||
taskNotFoundRefusal,
|
||||
taskNotStartableRefusal
|
||||
} from '../../../../shared/orchestration-dispatch-refusal-contract'
|
||||
import { OrcaRuntimeService } from '../../orca-runtime'
|
||||
import { OrchestrationDb } from '../../orchestration/db'
|
||||
import type { RpcFailure, RpcRequest, RpcResponse } from '../core'
|
||||
import { RpcDispatcher } from '../dispatcher'
|
||||
import { ORCHESTRATION_METHODS } from './orchestration'
|
||||
|
||||
const COORDINATOR_HANDLE = 'term_codes_coordinator'
|
||||
const COORDINATOR_PANE = 'tab_coord:cccccccc-cccc-4ccc-8ccc-cccccccccccc'
|
||||
const WORKER_HANDLE = 'term_codes_worker'
|
||||
const WORKER_PANE = 'tab_worker:dddddddd-dddd-4ddd-8ddd-dddddddddddd'
|
||||
|
||||
type Harness = { db: OrchestrationDb; runtime: OrcaRuntimeService; dispatcher: RpcDispatcher }
|
||||
|
||||
const harnesses: Harness[] = []
|
||||
let requestSequence = 0
|
||||
|
||||
afterEach(() => {
|
||||
for (const harness of harnesses.splice(0)) {
|
||||
harness.db.close()
|
||||
}
|
||||
vi.restoreAllMocks()
|
||||
})
|
||||
|
||||
// Why: an agent reads the receipt code to pick a recovery; every case is driven from the real
|
||||
// RPC dispatcher and checked against the shared contract the CLI-side test formats.
|
||||
describe('orchestration dispatch failure codes through RpcDispatcher', () => {
|
||||
it('reports task_not_found for a task id that does not exist', async () => {
|
||||
const harness = createHarness()
|
||||
|
||||
const response = await dispatch(harness, { task: 'task_missing', to: WORKER_HANDLE })
|
||||
|
||||
expect(expectFailure(response).error).toEqual(
|
||||
taskNotFoundRefusal('Task not found: task_missing', { taskId: 'task_missing' })
|
||||
)
|
||||
})
|
||||
|
||||
it('reports task_not_startable with the unmet dependencies for a pending task', async () => {
|
||||
const harness = createHarness()
|
||||
const parent = harness.db.createTask({ spec: 'parent' })
|
||||
const child = harness.db.createTask({ spec: 'child', deps: [parent.id] })
|
||||
|
||||
const response = await dispatch(harness, { task: child.id, to: WORKER_HANDLE })
|
||||
|
||||
expect(expectFailure(response).error).toEqual(
|
||||
taskNotStartableRefusal(`Task ${child.id} is pending; only ready tasks can be dispatched`, {
|
||||
taskId: child.id,
|
||||
status: 'pending',
|
||||
unmetDependencies: [parent.id]
|
||||
})
|
||||
)
|
||||
expect(harness.db.getTask(child.id)?.status).toBe('pending')
|
||||
})
|
||||
|
||||
it('reports task_not_startable with the status for a completed task', async () => {
|
||||
const harness = createHarness()
|
||||
const task = harness.db.createTask({ spec: 'done' })
|
||||
harness.db.updateTaskStatus(task.id, 'completed')
|
||||
|
||||
const response = await dispatch(harness, { task: task.id, to: WORKER_HANDLE })
|
||||
|
||||
expect(expectFailure(response).error).toEqual(
|
||||
taskNotStartableRefusal(`Task ${task.id} is completed; only ready tasks can be dispatched`, {
|
||||
taskId: task.id,
|
||||
status: 'completed',
|
||||
unmetDependencies: []
|
||||
})
|
||||
)
|
||||
})
|
||||
|
||||
it('reports inject_rejected when the target terminal runs no recognized agent', async () => {
|
||||
const harness = createHarness()
|
||||
const task = harness.db.createTask({ spec: 'work' })
|
||||
vi.spyOn(harness.runtime, 'isTerminalRunningAgent').mockResolvedValue(false)
|
||||
|
||||
const response = await dispatch(harness, { task: task.id, to: WORKER_HANDLE, inject: true })
|
||||
|
||||
expect(expectFailure(response).error).toEqual(
|
||||
injectRejectedRefusal(WORKER_HANDLE, 'no_agent_detected')
|
||||
)
|
||||
expect(expectFailure(response).error.message).toBe(buildInjectRejectionMessage(WORKER_HANDLE))
|
||||
expect(harness.db.getTask(task.id)?.status).toBe('ready')
|
||||
expect(harness.db.getDispatchContext(task.id)).toBeUndefined()
|
||||
})
|
||||
|
||||
it('reports task_not_startable with dependency detail from worker-start', async () => {
|
||||
const harness = createHarness()
|
||||
const parent = harness.db.createTask({ spec: 'parent' })
|
||||
const child = harness.db.createTask({ spec: 'child', deps: [parent.id] })
|
||||
mockWorkerStartTopology(harness.runtime)
|
||||
|
||||
const response = await harness.dispatcher.dispatch(
|
||||
request('orchestration.workerStart', {
|
||||
task: child.id,
|
||||
from: COORDINATOR_HANDLE,
|
||||
agent: 'claude'
|
||||
})
|
||||
)
|
||||
|
||||
expect(expectFailure(response).error).toEqual(
|
||||
taskNotStartableRefusal(`Task ${child.id} is pending; only a ready Task can start.`, {
|
||||
taskId: child.id,
|
||||
status: 'pending',
|
||||
unmetDependencies: [parent.id]
|
||||
})
|
||||
)
|
||||
expect(harness.db.getTask(child.id)?.status).toBe('pending')
|
||||
})
|
||||
|
||||
it('reports task_not_startable with retry detail for an invalid --retry-of', async () => {
|
||||
const harness = createHarness()
|
||||
const task = harness.db.createTask({ spec: 'work' })
|
||||
mockWorkerStartTopology(harness.runtime)
|
||||
|
||||
const response = await harness.dispatcher.dispatch(
|
||||
request('orchestration.workerStart', {
|
||||
task: task.id,
|
||||
from: COORDINATOR_HANDLE,
|
||||
agent: 'claude',
|
||||
retryOf: 'ctx_missing'
|
||||
})
|
||||
)
|
||||
|
||||
expect(expectFailure(response).error).toEqual(
|
||||
taskNotStartableRefusal(`Task ${task.id} cannot retry from Dispatch ctx_missing.`, {
|
||||
taskId: task.id,
|
||||
status: 'ready',
|
||||
unmetDependencies: [],
|
||||
retryOf: 'ctx_missing'
|
||||
})
|
||||
)
|
||||
})
|
||||
|
||||
it('types the atomic claim loser when the task changes after the ready precheck', async () => {
|
||||
const harness = createHarness()
|
||||
const task = harness.db.createTask({ spec: 'raced' })
|
||||
// Why: the pane lookup runs after the ready precheck and before the DB claim, so failing the
|
||||
// task there is the interleaving a concurrent status change produces. The loser's DB refusal
|
||||
// must carry the same typed receipt instead of the bare Error it used to throw.
|
||||
vi.mocked(harness.runtime.getTerminalPaneKey).mockImplementation((handle) => {
|
||||
if (handle === WORKER_HANDLE) {
|
||||
harness.db.updateTaskStatus(task.id, 'failed', 'raced out')
|
||||
return WORKER_PANE
|
||||
}
|
||||
return handle === COORDINATOR_HANDLE ? COORDINATOR_PANE : null
|
||||
})
|
||||
|
||||
const response = await dispatch(harness, { task: task.id, to: WORKER_HANDLE })
|
||||
|
||||
expect(expectFailure(response).error).toEqual(
|
||||
taskNotStartableRefusal(`Task ${task.id} is failed; only ready tasks can be dispatched`, {
|
||||
taskId: task.id,
|
||||
status: 'failed',
|
||||
unmetDependencies: []
|
||||
})
|
||||
)
|
||||
expect(harness.db.getDispatchContext(task.id)).toBeUndefined()
|
||||
})
|
||||
|
||||
it('keeps runtime_error for a genuinely unexpected dispatch failure', async () => {
|
||||
const harness = createHarness()
|
||||
const task = harness.db.createTask({ spec: 'work' })
|
||||
vi.spyOn(harness.runtime, 'isTerminalRunningAgent').mockRejectedValue(
|
||||
new Error('probe exploded')
|
||||
)
|
||||
|
||||
const response = await dispatch(harness, { task: task.id, to: WORKER_HANDLE, inject: true })
|
||||
|
||||
expect(expectFailure(response).error).toMatchObject({
|
||||
code: 'runtime_error',
|
||||
message: 'probe exploded'
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
function expectFailure(response: RpcResponse): RpcFailure {
|
||||
if (response.ok) {
|
||||
throw new Error(`Expected a failure, got ${JSON.stringify(response.result)}`)
|
||||
}
|
||||
return response
|
||||
}
|
||||
|
||||
function createHarness(): Harness {
|
||||
const db = new OrchestrationDb(':memory:')
|
||||
const runtime = new OrcaRuntimeService()
|
||||
runtime.setOrchestrationDb(db)
|
||||
vi.spyOn(runtime, 'getTerminalPaneKey').mockImplementation((handle) =>
|
||||
handle === COORDINATOR_HANDLE ? COORDINATOR_PANE : handle === WORKER_HANDLE ? WORKER_PANE : null
|
||||
)
|
||||
vi.spyOn(runtime, 'getTerminalProcessIncarnation').mockImplementation((handle) =>
|
||||
handle === WORKER_HANDLE ? 'pty-worker:incarnation-1' : null
|
||||
)
|
||||
const runId = db.createRun({
|
||||
objective: 'Typed dispatch failures',
|
||||
coordinatorHandle: COORDINATOR_HANDLE,
|
||||
coordinatorPaneKey: COORDINATOR_PANE
|
||||
}).id
|
||||
const createTask = db.createTask.bind(db)
|
||||
db.createTask = (task) => createTask({ ...task, runId: task.runId ?? runId })
|
||||
const harness = {
|
||||
db,
|
||||
runtime,
|
||||
dispatcher: new RpcDispatcher({ runtime, methods: ORCHESTRATION_METHODS })
|
||||
}
|
||||
harnesses.push(harness)
|
||||
return harness
|
||||
}
|
||||
|
||||
function mockWorkerStartTopology(runtime: OrcaRuntimeService): void {
|
||||
vi.spyOn(runtime, 'validateOrchestrationAgentLauncher').mockImplementation(() => {})
|
||||
vi.spyOn(runtime, 'showTerminal').mockImplementation(
|
||||
async (handle) => ({ handle, worktreeId: 'repo::worktree', status: 'running' }) as never
|
||||
)
|
||||
vi.spyOn(runtime, 'showManagedTerminalWorkspace').mockResolvedValue({
|
||||
id: 'repo::worktree'
|
||||
} as never)
|
||||
}
|
||||
|
||||
function dispatch(harness: Harness, params: Record<string, unknown>): Promise<RpcResponse> {
|
||||
return harness.dispatcher.dispatch(
|
||||
request('orchestration.dispatch', { from: COORDINATOR_HANDLE, ...params })
|
||||
)
|
||||
}
|
||||
|
||||
function request(method: string, params: Record<string, unknown>): RpcRequest {
|
||||
requestSequence += 1
|
||||
return {
|
||||
id: `rpc_dispatch_error_codes_${requestSequence}`,
|
||||
authToken: 'test-token',
|
||||
method,
|
||||
params,
|
||||
orchestrationContractVersion: ORCHESTRATION_CONTRACT_VERSION,
|
||||
orchestrationRequestId: `dispatch_error_codes_${requestSequence}`
|
||||
}
|
||||
}
|
||||
@@ -4,7 +4,7 @@ import { OptionalFiniteNumber, OptionalString, requiredString } from '../../../s
|
||||
import type { GateStatus } from '../../../../orchestration/db'
|
||||
import { Coordinator } from '../../../../orchestration/coordinator'
|
||||
import { resolveRunScope } from '../runs/run-scope'
|
||||
import { OrchestrationError } from '../../../../orchestration/orchestration-error'
|
||||
import { taskNotFoundError } from '../../../../orchestration/task-dispatch-refusal'
|
||||
|
||||
// Why: the coordinator instance is stored at module scope so orchestration.runStop
|
||||
// can signal it to halt. Only one coordinator can run at a time (enforced by
|
||||
@@ -137,10 +137,10 @@ export const ORCHESTRATION_GATE_METHODS: RpcMethod[] = [
|
||||
callerEvidence: orchestrationCompatibilityEvidence
|
||||
})
|
||||
if (task.run_id !== run.id) {
|
||||
throw new OrchestrationError(
|
||||
'task_not_found',
|
||||
`Task ${params.task} was not found in Run ${run.id}.`
|
||||
)
|
||||
throw taskNotFoundError(`Task ${params.task} was not found in Run ${run.id}.`, {
|
||||
taskId: params.task,
|
||||
runId: run.id
|
||||
})
|
||||
}
|
||||
const gate = db.createGate({
|
||||
taskId: params.task,
|
||||
|
||||
@@ -1,31 +0,0 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { buildInjectRejectionMessage } from './inject-rejection-message'
|
||||
import { TUI_AGENT_CONFIG } from '../../../../../../shared/tui-agent-config'
|
||||
import { recognizeAgentProcess } from '../../../../../../shared/agent-process-recognition'
|
||||
|
||||
describe('buildInjectRejectionMessage', () => {
|
||||
const message = buildInjectRejectionMessage('term_a')
|
||||
|
||||
it('keeps the substring callers and scripts match on', () => {
|
||||
expect(message).toContain('Cannot dispatch --inject to terminal term_a')
|
||||
expect(message).toContain('no recognized agent detected')
|
||||
})
|
||||
|
||||
it('names every agent Orca recognizes, including agy', () => {
|
||||
expect(message).toMatch(/\bagy\b/)
|
||||
for (const config of Object.values(TUI_AGENT_CONFIG)) {
|
||||
expect(message).toContain(config.expectedProcess)
|
||||
}
|
||||
})
|
||||
|
||||
it('lists only names detection actually resolves, deduped and sorted', () => {
|
||||
const listed = (/\(([^)]+)\)/.exec(message)?.[1] ?? '').split(', ')
|
||||
|
||||
expect(listed.length).toBeGreaterThan(0)
|
||||
expect(new Set(listed).size).toBe(listed.length)
|
||||
expect([...listed].sort()).toEqual(listed)
|
||||
for (const name of listed) {
|
||||
expect(recognizeAgentProcess(name)).not.toBeNull()
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -1,16 +0,0 @@
|
||||
import { TUI_AGENT_CONFIG } from '../../../../../../shared/tui-agent-config'
|
||||
|
||||
// Why: the old five-name example read as an allowlist (#15125); derive from the field detection keys on so it cannot drift.
|
||||
// Not filtered by `disabledTuiAgents` — that gates Orca's launchers, not detection, so a hand-started disabled agent still injects.
|
||||
const RECOGNIZED_AGENT_PROCESS_NAMES = [
|
||||
...new Set(Object.values(TUI_AGENT_CONFIG).map((config) => config.expectedProcess))
|
||||
].sort()
|
||||
|
||||
export function buildInjectRejectionMessage(terminal: string): string {
|
||||
return (
|
||||
`Cannot dispatch --inject to terminal ${terminal}: no recognized agent detected. ` +
|
||||
`Orca detects these agent CLIs (${RECOGNIZED_AGENT_PROCESS_NAMES.join(', ')}). ` +
|
||||
'Start one in the terminal and let it finish launching, ' +
|
||||
'or dispatch without --inject and send the prompt manually.'
|
||||
)
|
||||
}
|
||||
@@ -2,7 +2,11 @@ import { defineMethod, type RpcMethod } from '../../../core'
|
||||
import { OrchestrationError } from '../../../../orchestration/orchestration-error'
|
||||
import { buildDispatchPreamble } from '../../../../orchestration/preamble'
|
||||
import { resolveDispatchCreator } from './dispatch-creator'
|
||||
import { buildInjectRejectionMessage } from '../messaging/inject-rejection-message'
|
||||
import {
|
||||
injectRejectedError,
|
||||
taskNotFoundError,
|
||||
taskNotStartableError
|
||||
} from '../../../../orchestration/task-dispatch-refusal'
|
||||
import { resolveRunScope } from './run-scope'
|
||||
import { DispatchParams, DispatchShowParams } from '../schemas'
|
||||
|
||||
@@ -23,7 +27,7 @@ export const ORCHESTRATION_DISPATCH_METHODS: RpcMethod[] = [
|
||||
const db = runtime.getOrchestrationDb()
|
||||
const task = db.getTask(params.task)
|
||||
if (!task) {
|
||||
throw new Error(`Task not found: ${params.task}`)
|
||||
throw taskNotFoundError(`Task not found: ${params.task}`, { taskId: params.task })
|
||||
}
|
||||
const run = resolveRunScope(runtime, {
|
||||
runId: params.run,
|
||||
@@ -33,10 +37,10 @@ export const ORCHESTRATION_DISPATCH_METHODS: RpcMethod[] = [
|
||||
callerEvidence: orchestrationCompatibilityEvidence
|
||||
})
|
||||
if (task.run_id !== run.id) {
|
||||
throw new OrchestrationError(
|
||||
'task_not_found',
|
||||
`Task ${task.id} was not found in Run ${run.id}.`
|
||||
)
|
||||
throw taskNotFoundError(`Task ${task.id} was not found in Run ${run.id}.`, {
|
||||
taskId: task.id,
|
||||
runId: run.id
|
||||
})
|
||||
}
|
||||
|
||||
// Why: dry-run previews the preamble without mutating state, so it skips the ready-status check and uses a placeholder dispatchId.
|
||||
@@ -67,7 +71,11 @@ export const ORCHESTRATION_DISPATCH_METHODS: RpcMethod[] = [
|
||||
const to = params.to
|
||||
|
||||
if (task.status !== 'ready') {
|
||||
throw new Error(`Task ${params.task} is ${task.status}; only ready tasks can be dispatched`)
|
||||
throw taskNotStartableError(
|
||||
db,
|
||||
`Task ${params.task} is ${task.status}; only ready tasks can be dispatched`,
|
||||
task
|
||||
)
|
||||
}
|
||||
|
||||
const dispatchAuthority = runtime.getOrchestrationDispatchAuthority(to)
|
||||
@@ -102,7 +110,7 @@ export const ORCHESTRATION_DISPATCH_METHODS: RpcMethod[] = [
|
||||
if (params.inject) {
|
||||
const hasAgent = await runtime.isTerminalRunningAgent(to)
|
||||
if (!hasAgent) {
|
||||
throw new Error(buildInjectRejectionMessage(to))
|
||||
throw injectRejectedError(to, 'no_agent_detected')
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@ import type { RpcContext } from '../../../core'
|
||||
import { createOrchestrationRpcHarness } from '../rpc-test-harness'
|
||||
import type { OrchestrationDb } from '../../../../orchestration/db'
|
||||
import type { OrcaRuntimeService } from '../../../../orca-runtime'
|
||||
import { buildInjectRejectionMessage } from '../messaging/inject-rejection-message'
|
||||
import { buildInjectRejectionMessage } from '../../../../../../shared/orchestration-dispatch-refusal-contract'
|
||||
import { createRootDispatch } from '../../../../orchestration/db/root-dispatch-test-fixture'
|
||||
|
||||
describe('orchestration RPC methods', () => {
|
||||
|
||||
@@ -152,9 +152,13 @@ export const CancelParams = z
|
||||
.object({
|
||||
envelope: MutationEnvelope,
|
||||
turnId: Identifier('Invalid turn id'),
|
||||
scope: z.literal('background-tasks').optional()
|
||||
scope: z.literal('background-tasks').optional(),
|
||||
taskId: Identifier('Invalid task id').optional()
|
||||
})
|
||||
.strict()
|
||||
.refine((value) => value.taskId === undefined || value.scope === 'background-tasks', {
|
||||
message: 'A task id requires background-task scope'
|
||||
})
|
||||
|
||||
export const RespondParams = z
|
||||
.object({
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
// `agentSession.subscribeStatus` — every structured session's projected status on one stream.
|
||||
//
|
||||
// Session lists read turn state from here instead of replaying transcripts: one stream per client
|
||||
// covers every session, and unlike a transcript subscription it retains none of them.
|
||||
|
||||
import { defineStreamingMethod, type RpcAnyMethod, type RpcContext } from '../core'
|
||||
import { requireStructuredHost as requireHost } from './structured-agent-session-gate'
|
||||
import { structuredAgentSessionStatusSubscriptionId } from './structured-agent-session-subscription-id'
|
||||
|
||||
/** Ties a stream to both ends that can close it — the runtime's subscription registry and the
|
||||
* transport abort — so either one runs `onClose` exactly once. */
|
||||
export function bindStructuredAgentSessionStream(
|
||||
ctx: RpcContext,
|
||||
subscriptionId: string,
|
||||
onClose: () => void
|
||||
): { isClosed: () => boolean } {
|
||||
let closed = false
|
||||
let releaseTransportSubscription = (): void => {}
|
||||
const onTransportAbort = (): void => releaseTransportSubscription()
|
||||
const cleanup = (): void => {
|
||||
closed = true
|
||||
ctx.signal?.removeEventListener('abort', onTransportAbort)
|
||||
onClose()
|
||||
}
|
||||
let registration: { releaseIfCurrent: () => void }
|
||||
if (typeof ctx.runtime.registerOwnedSubscriptionCleanup === 'function') {
|
||||
registration = ctx.runtime.registerOwnedSubscriptionCleanup(
|
||||
subscriptionId,
|
||||
cleanup,
|
||||
ctx.connectionId
|
||||
)
|
||||
} else {
|
||||
ctx.runtime.registerSubscriptionCleanup(subscriptionId, cleanup, ctx.connectionId)
|
||||
registration = { releaseIfCurrent: () => ctx.runtime.cleanupSubscription(subscriptionId) }
|
||||
}
|
||||
releaseTransportSubscription = registration.releaseIfCurrent
|
||||
ctx.signal?.addEventListener('abort', onTransportAbort, { once: true })
|
||||
if (ctx.signal?.aborted) {
|
||||
onTransportAbort()
|
||||
}
|
||||
return { isClosed: () => closed }
|
||||
}
|
||||
|
||||
export const STRUCTURED_AGENT_SESSION_STATUS_METHODS: RpcAnyMethod[] = [
|
||||
defineStreamingMethod({
|
||||
name: 'agentSession.subscribeStatus',
|
||||
params: null,
|
||||
handler: async (_params, ctx, emit) => {
|
||||
const host = requireHost(ctx)
|
||||
const subscriptionId = structuredAgentSessionStatusSubscriptionId(ctx)
|
||||
let dispose = (): void => {}
|
||||
const stream = bindStructuredAgentSessionStream(ctx, subscriptionId, () => dispose())
|
||||
if (stream.isClosed()) {
|
||||
return
|
||||
}
|
||||
dispose = host.subscribeStatus({ id: subscriptionId, emit })
|
||||
if (stream.isClosed()) {
|
||||
dispose()
|
||||
}
|
||||
}
|
||||
})
|
||||
]
|
||||
@@ -0,0 +1,29 @@
|
||||
// Subscription ids for the streaming `agentSession.*` methods.
|
||||
//
|
||||
// Shared control multiplexes several streams over one socket, so the frame id keeps one
|
||||
// subscriber from evicting another. It is appended only when present: collapsing a missing
|
||||
// frame id to a constant is the collision the rule exists to prevent.
|
||||
|
||||
import type { RpcContext } from '../core'
|
||||
|
||||
const SUBSCRIPTION_PREFIX = 'agentSession'
|
||||
|
||||
function withFrameId(ctx: RpcContext, base: string): string {
|
||||
return ctx.requestId ? `${base}:${ctx.requestId}` : base
|
||||
}
|
||||
|
||||
/** The id a session's streams share before the frame id. `unsubscribe` addresses this
|
||||
* directly and sweeps `${base}:` to reach every frame under it. */
|
||||
export function structuredAgentSessionSubscriptionBase(ctx: RpcContext, sessionId: string): string {
|
||||
return `${SUBSCRIPTION_PREFIX}:${ctx.connectionId ?? 'local'}:${sessionId}`
|
||||
}
|
||||
|
||||
/** One session's transcript stream. */
|
||||
export function structuredAgentSessionSubscriptionId(ctx: RpcContext, sessionId: string): string {
|
||||
return withFrameId(ctx, structuredAgentSessionSubscriptionBase(ctx, sessionId))
|
||||
}
|
||||
|
||||
/** The status feed, which is per connection rather than per session. */
|
||||
export function structuredAgentSessionStatusSubscriptionId(ctx: RpcContext): string {
|
||||
return withFrameId(ctx, `${SUBSCRIPTION_PREFIX}.status:${ctx.connectionId ?? 'local'}`)
|
||||
}
|
||||
@@ -2,8 +2,14 @@
|
||||
// accepts once they can.
|
||||
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
import type { AgentJournalRenderItem } from '../../../../shared/agent-session-journal-types'
|
||||
import type { AgentSessionJournal } from '../../../native-chat/agent-session-journal/journal-store'
|
||||
import type { StructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-host'
|
||||
import { setStructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-registry'
|
||||
import {
|
||||
StructuredAgentSessionStatusFeed,
|
||||
type StructuredAgentSessionStatusSubscriber
|
||||
} from '../../../native-chat/agent-session-wire/structured-agent-session-status-feed'
|
||||
import {
|
||||
RUNTIME_CAPABILITIES,
|
||||
RUNTIME_PROTOCOL_VERSION,
|
||||
@@ -64,6 +70,44 @@ function request(method: string, params: unknown): RpcRequest {
|
||||
let hostCalls: Record<string, ReturnType<typeof vi.fn>>
|
||||
let runtimeCalls: Record<string, ReturnType<typeof vi.fn>>
|
||||
|
||||
const STATUS_SESSION = 'session-status'
|
||||
const STATUS_ITEMS: AgentJournalRenderItem[] = [
|
||||
{
|
||||
itemId: 'user-1',
|
||||
sequence: 1,
|
||||
revision: 1,
|
||||
observedAt: 1,
|
||||
body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'write a poem' }] }
|
||||
},
|
||||
{
|
||||
itemId: 'turn-1',
|
||||
sequence: 2,
|
||||
revision: 1,
|
||||
observedAt: 2,
|
||||
body: { kind: 'status', text: 'Working', turnLifecycle: { turnId: 'turn-1', state: 'running' } }
|
||||
}
|
||||
]
|
||||
|
||||
/** One indexed session over a journal that reads back fixed items; the projection is real. */
|
||||
function statusFeed(): StructuredAgentSessionStatusFeed {
|
||||
return new StructuredAgentSessionStatusFeed({
|
||||
sessions: new Map([
|
||||
[
|
||||
STATUS_SESSION,
|
||||
{
|
||||
journal: {
|
||||
isReadOnly: false,
|
||||
snapshot: () => ({ items: STATUS_ITEMS })
|
||||
} as unknown as AgentSessionJournal,
|
||||
params: { location: { workspaceId: 'workspace-1' }, provider: 'codex' as const }
|
||||
}
|
||||
]
|
||||
]),
|
||||
getRecord: () => null,
|
||||
now: () => 1_000
|
||||
})
|
||||
}
|
||||
|
||||
function hostStub(): StructuredAgentSessionHost {
|
||||
hostCalls = {
|
||||
attach: vi.fn(async () => ({
|
||||
@@ -122,6 +166,11 @@ function hostStub(): StructuredAgentSessionHost {
|
||||
})),
|
||||
history: vi.fn(() => ({ ok: true, page: { items: [] } })),
|
||||
subscribe: vi.fn(() => () => undefined),
|
||||
// A real feed, so the snapshot this method hands back is a genuine projection rather
|
||||
// than a shape the stub restated.
|
||||
subscribeStatus: vi.fn((subscriber: StructuredAgentSessionStatusSubscriber) =>
|
||||
statusFeed().subscribe(subscriber)
|
||||
),
|
||||
unsubscribe: vi.fn()
|
||||
}
|
||||
return hostCalls as unknown as StructuredAgentSessionHost
|
||||
@@ -240,7 +289,7 @@ describe('capability gating', () => {
|
||||
}
|
||||
// Bump deliberately: the whole agentSession.* surface is behind the structured capability,
|
||||
// so an additive method is invisible to old clients and needs no protocol bump.
|
||||
expect(STRUCTURED_AGENT_SESSION_METHODS).toHaveLength(17)
|
||||
expect(STRUCTURED_AGENT_SESSION_METHODS).toHaveLength(18)
|
||||
})
|
||||
|
||||
it('hides the surface from a declared client that did not advertise it', async () => {
|
||||
@@ -480,6 +529,20 @@ describe('method routing', () => {
|
||||
])
|
||||
})
|
||||
|
||||
it('routes an optional background task id through cancellation', async () => {
|
||||
const params = {
|
||||
envelope: envelope(),
|
||||
turnId: 'background-tasks',
|
||||
scope: 'background-tasks' as const,
|
||||
taskId: 'task-2'
|
||||
}
|
||||
|
||||
const response = await call('agentSession.cancel', params, STRUCTURED_CLIENT)
|
||||
|
||||
expect(response).toMatchObject({ ok: true })
|
||||
expect(hostCalls.cancel).toHaveBeenCalledWith(expect.anything(), params)
|
||||
})
|
||||
|
||||
it('routes the structured handoff mutation through the host', async () => {
|
||||
const response = await call('agentSession.requestHandoff', {
|
||||
envelope: envelope(),
|
||||
@@ -510,6 +573,21 @@ describe('parameter validation', () => {
|
||||
})
|
||||
})
|
||||
|
||||
it('rejects invalid or unscoped background task ids', async () => {
|
||||
await rejects('agentSession.cancel', {
|
||||
envelope: envelope(),
|
||||
turnId: 'background-tasks',
|
||||
scope: 'background-tasks',
|
||||
taskId: ' task-2'
|
||||
})
|
||||
await rejects('agentSession.cancel', {
|
||||
envelope: envelope(),
|
||||
turnId: 'turn-1',
|
||||
taskId: 'task-2'
|
||||
})
|
||||
expect(hostCalls.cancel).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('refuses to let a client author anything but a user turn', async () => {
|
||||
await rejects(
|
||||
'agentSession.send',
|
||||
@@ -583,3 +661,31 @@ describe('parameter validation', () => {
|
||||
expect(response).toMatchObject({ ok: true })
|
||||
})
|
||||
})
|
||||
|
||||
describe('agentSession.subscribeStatus', () => {
|
||||
it('is invisible to a client without the structured capability', async () => {
|
||||
const reply = await call('agentSession.subscribeStatus', null, { clientKind: 'runtime' })
|
||||
expect(reply.ok).toBe(false)
|
||||
expect(hostCalls.subscribeStatus).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('opens the host status feed with a projected snapshot as its first reply', async () => {
|
||||
const reply = await call('agentSession.subscribeStatus', null, STRUCTURED_CLIENT)
|
||||
expect(reply).toMatchObject({
|
||||
ok: true,
|
||||
result: {
|
||||
type: 'snapshot',
|
||||
sessions: [
|
||||
{
|
||||
sessionId: STATUS_SESSION,
|
||||
workspaceId: 'workspace-1',
|
||||
agent: 'codex',
|
||||
status: 'working',
|
||||
latestPrompt: 'write a poem'
|
||||
}
|
||||
]
|
||||
}
|
||||
})
|
||||
expect(hostCalls.subscribeStatus).toHaveBeenCalledOnce()
|
||||
})
|
||||
})
|
||||
|
||||
@@ -21,6 +21,14 @@ import {
|
||||
import type { AgentSessionAttachParams } from '../../../native-chat/agent-session-wire/structured-agent-session-attach'
|
||||
import { STRUCTURED_AGENT_SESSION_HOLD_METHODS } from './structured-agent-session-hold'
|
||||
import { resolveUncommittedStructuredCreate } from './structured-agent-session-precommit-refusal'
|
||||
import {
|
||||
bindStructuredAgentSessionStream,
|
||||
STRUCTURED_AGENT_SESSION_STATUS_METHODS
|
||||
} from './structured-agent-session-status-stream'
|
||||
import {
|
||||
structuredAgentSessionSubscriptionBase as subscriptionBaseFor,
|
||||
structuredAgentSessionSubscriptionId as subscriptionIdFor
|
||||
} from './structured-agent-session-subscription-id'
|
||||
import {
|
||||
AttachParams,
|
||||
CancelParams,
|
||||
@@ -37,15 +45,6 @@ import {
|
||||
UnsubscribeParams
|
||||
} from './structured-agent-session-schemas'
|
||||
|
||||
const SUBSCRIPTION_PREFIX = 'agentSession'
|
||||
|
||||
function subscriptionIdFor(ctx: RpcContext, sessionId: string): string {
|
||||
const base = `${SUBSCRIPTION_PREFIX}:${ctx.connectionId ?? 'local'}:${sessionId}`
|
||||
// Shared control multiplexes several streams over one socket; the frame id
|
||||
// keeps one subscriber from evicting another on the same session.
|
||||
return ctx.requestId ? `${base}:${ctx.requestId}` : base
|
||||
}
|
||||
|
||||
/**
|
||||
* The attach-shaped entries take the location from the client instead of resolving it from a
|
||||
* worktree, so they never reach the worktree-resolving create-support check. Ask the executing
|
||||
@@ -245,33 +244,12 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [
|
||||
// Retain-only: reading history must never be what starts a provider process. Current clients
|
||||
// explicitly hold every open surface before subscribing.
|
||||
const streamHolder = `subscription:${subscriptionId}`
|
||||
let closed = false
|
||||
let dispose = (): void => {}
|
||||
let releaseTransportSubscription = (): void => {}
|
||||
const onTransportAbort = (): void => releaseTransportSubscription()
|
||||
const cleanup = () => {
|
||||
closed = true
|
||||
ctx.signal?.removeEventListener('abort', onTransportAbort)
|
||||
const stream = bindStructuredAgentSessionStream(ctx, subscriptionId, () => {
|
||||
dispose()
|
||||
host.release(params.sessionId, streamHolder)
|
||||
}
|
||||
let registration: { releaseIfCurrent: () => void }
|
||||
if (typeof ctx.runtime.registerOwnedSubscriptionCleanup === 'function') {
|
||||
registration = ctx.runtime.registerOwnedSubscriptionCleanup(
|
||||
subscriptionId,
|
||||
cleanup,
|
||||
ctx.connectionId
|
||||
)
|
||||
} else {
|
||||
ctx.runtime.registerSubscriptionCleanup(subscriptionId, cleanup, ctx.connectionId)
|
||||
registration = { releaseIfCurrent: () => ctx.runtime.cleanupSubscription(subscriptionId) }
|
||||
}
|
||||
releaseTransportSubscription = registration.releaseIfCurrent
|
||||
ctx.signal?.addEventListener('abort', onTransportAbort, { once: true })
|
||||
if (ctx.signal?.aborted) {
|
||||
onTransportAbort()
|
||||
}
|
||||
if (closed) {
|
||||
})
|
||||
if (stream.isClosed()) {
|
||||
return
|
||||
}
|
||||
// The host emits the opening snapshot (or the missed batch) synchronously
|
||||
@@ -282,7 +260,7 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [
|
||||
emit,
|
||||
...(params.cursor ? { cursor: params.cursor } : {})
|
||||
})
|
||||
if (closed) {
|
||||
if (stream.isClosed()) {
|
||||
dispose()
|
||||
} else {
|
||||
// Fire-and-forget, but never unhandled: a resume that refuses leaves the stream holding a
|
||||
@@ -300,8 +278,7 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [
|
||||
params: UnsubscribeParams,
|
||||
handler: async (params, ctx) => {
|
||||
requireHost(ctx)
|
||||
const connection = ctx.connectionId ?? 'local'
|
||||
const base = `${SUBSCRIPTION_PREFIX}:${connection}:${params.sessionId}`
|
||||
const base = subscriptionBaseFor(ctx, params.sessionId)
|
||||
if (params.subscriptionId) {
|
||||
ctx.runtime.cleanupSubscription(`${base}:${params.subscriptionId}`)
|
||||
return { unsubscribed: true }
|
||||
@@ -311,5 +288,6 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [
|
||||
return { unsubscribed: true }
|
||||
}
|
||||
}),
|
||||
...STRUCTURED_AGENT_SESSION_HOLD_METHODS
|
||||
...STRUCTURED_AGENT_SESSION_HOLD_METHODS,
|
||||
...STRUCTURED_AGENT_SESSION_STATUS_METHODS
|
||||
]
|
||||
|
||||
@@ -110,23 +110,31 @@ export async function resolveRuntimeLocalWorktreeCreateCandidate(args: {
|
||||
args.username,
|
||||
args.localWorktreeGitOptions
|
||||
)
|
||||
checkoutExistingBranch = await canCheckoutExistingLocalBranch(
|
||||
args.repo.path,
|
||||
branchName,
|
||||
args.baseBranch,
|
||||
...args.localWorktreeGitOptionArgs
|
||||
)
|
||||
if (checkoutExistingBranch && !selectedExistingLocalBranchName) {
|
||||
selectedExistingLocalBranchName = branchName
|
||||
const tryExistingBranch = async (): Promise<boolean> => {
|
||||
checkoutExistingBranch = await canCheckoutExistingLocalBranch(
|
||||
args.repo.path,
|
||||
branchName,
|
||||
args.baseBranch,
|
||||
...args.localWorktreeGitOptionArgs
|
||||
)
|
||||
return checkoutExistingBranch
|
||||
}
|
||||
const preferExistingBranch = Boolean(
|
||||
args.request.branchNameOverride || selectedExistingLocalBranchName
|
||||
)
|
||||
checkoutExistingBranch = preferExistingBranch && (await tryExistingBranch())
|
||||
branchConflictKind = checkoutExistingBranch
|
||||
? null
|
||||
: await getBranchConflictKind(
|
||||
args.repo.path,
|
||||
branchName,
|
||||
args.baseBranch,
|
||||
...args.localWorktreeGitOptionArgs
|
||||
args.localWorktreeGitOptions,
|
||||
preferExistingBranch ? undefined : tryExistingBranch
|
||||
)
|
||||
if (checkoutExistingBranch && !selectedExistingLocalBranchName) {
|
||||
selectedExistingLocalBranchName = branchName
|
||||
}
|
||||
const allowedPushTargetRemoteConflict =
|
||||
branchConflictKind &&
|
||||
isAllowedPushTargetRemoteConflict(branchConflictKind, branchName, args.request)
|
||||
|
||||
@@ -231,7 +231,7 @@ export class RuntimeRemoteFetchController {
|
||||
? baseBranch.slice(remoteRefPrefix.length)
|
||||
: baseBranch
|
||||
// A remote-tracking base needs both a configured remote and a branch component.
|
||||
if (!shortBaseBranch.includes('/')) {
|
||||
if (shortBaseBranch.indexOf('/') <= 0 || shortBaseBranch.endsWith('/')) {
|
||||
return null
|
||||
}
|
||||
let remotes: string[]
|
||||
|
||||
@@ -273,6 +273,22 @@ describe('worktree scan admin-fingerprint gate', () => {
|
||||
}
|
||||
})
|
||||
|
||||
it('resolves a just-created id after invalidation even within both cache TTLs', async () => {
|
||||
const { runtime, list } = makeRuntime()
|
||||
listWorktreesStrictMock.mockResolvedValueOnce([
|
||||
{ path: REPO_PATH, head: 'abc', branch: 'main', isBare: false, isMainWorktree: true }
|
||||
])
|
||||
await list()
|
||||
await expect(runtime.showManagedWorktree(`id:${WORKTREE_ID}`)).rejects.toThrow(
|
||||
'selector_not_found'
|
||||
)
|
||||
runtime.invalidateWorktreeCatalog(REPO_ID)
|
||||
await expect(runtime.showManagedWorktree(`id:${WORKTREE_ID}`)).resolves.toMatchObject({
|
||||
id: WORKTREE_ID
|
||||
})
|
||||
expect(scanCount()).toBe(2)
|
||||
})
|
||||
|
||||
it('scans when the probe cannot describe the repo', async () => {
|
||||
vi.useFakeTimers()
|
||||
try {
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
import { EventEmitter } from 'node:events'
|
||||
import { expect, it, vi } from 'vitest'
|
||||
|
||||
vi.mock('electron', () => ({
|
||||
app: {
|
||||
isPackaged: false,
|
||||
getAppPath: () => '/host/app'
|
||||
}
|
||||
}))
|
||||
vi.mock('../persistence', () => ({
|
||||
getCanonicalUserDataPath: () => '/host/user-data'
|
||||
}))
|
||||
|
||||
import { OrcaRuntimeService } from '../runtime/orca-runtime'
|
||||
import { runRemoteOrcaCli } from './ssh-remote-orca-cli'
|
||||
|
||||
// Why: the SSH bridge captures the host CLI child's stdout and exit code without reparsing; this
|
||||
// pins that a typed refusal envelope and its nonzero exit reach the remote agent unchanged.
|
||||
it('relays typed dispatch refusal codes from the host CLI unchanged', async () => {
|
||||
const child = new EventEmitter() as EventEmitter & {
|
||||
stdout: EventEmitter
|
||||
stderr: EventEmitter
|
||||
stdin: { end: ReturnType<typeof vi.fn>; on: ReturnType<typeof vi.fn> }
|
||||
kill: ReturnType<typeof vi.fn>
|
||||
}
|
||||
child.stdout = new EventEmitter()
|
||||
child.stderr = new EventEmitter()
|
||||
child.stdin = { end: vi.fn(), on: vi.fn() }
|
||||
child.kill = vi.fn()
|
||||
const spawn = vi.fn(() => child)
|
||||
const refusal = {
|
||||
id: 'rpc_1',
|
||||
ok: false,
|
||||
error: {
|
||||
code: 'task_not_startable',
|
||||
message: 'Task task_1 is pending; only ready tasks can be dispatched',
|
||||
data: { taskId: 'task_1', status: 'pending', unmetDependencies: ['task_0'] }
|
||||
},
|
||||
_meta: { runtimeId: 'runtime_1' }
|
||||
}
|
||||
|
||||
const resultPromise = runRemoteOrcaCli(
|
||||
new OrcaRuntimeService(),
|
||||
{
|
||||
argv: ['orchestration', 'dispatch', '--task', 'task_1', '--to', 'term_w', '--json'],
|
||||
cwd: '/home/alice/repo',
|
||||
env: { ORCA_TERMINAL_HANDLE: 'term_ssh' }
|
||||
},
|
||||
{
|
||||
execPath: '/host/electron',
|
||||
cliEntryPath: '/host/app/out/cli/index.js',
|
||||
userDataPath: '/host/user-data',
|
||||
entryExists: () => true,
|
||||
spawn: spawn as never
|
||||
}
|
||||
)
|
||||
|
||||
const stdout = `${JSON.stringify(refusal, null, 2)}\n`
|
||||
await Promise.resolve()
|
||||
child.stdout.emit('data', Buffer.from(stdout))
|
||||
child.emit('close', 1)
|
||||
|
||||
expect(await resultPromise).toEqual({ stdout, stderr: '', exitCode: 1 })
|
||||
})
|
||||
@@ -78,6 +78,34 @@ describe('createMainWindow', () => {
|
||||
}
|
||||
}
|
||||
|
||||
it.each(['darwin', 'linux', 'win32'] as const)(
|
||||
'keeps explicit background startup hidden through ready/load/fallback on %s',
|
||||
(platform) => {
|
||||
vi.useFakeTimers()
|
||||
vi.stubEnv('ORCA_BACKGROUND_LAUNCH', '1')
|
||||
const { browserWindowInstance, windowHandlers } = createStartupRevealWindowFixture()
|
||||
const showInactive = vi.fn()
|
||||
Object.assign(browserWindowInstance, { showInactive })
|
||||
try {
|
||||
withPlatform(platform, () => {
|
||||
createMainWindow(createStartupRevealStore(true) as never, { revealOnDidFinishLoad: true })
|
||||
const revealAfterLoad = browserWindowInstance.webContents.on.mock.calls.find(
|
||||
([event]) => event === 'did-finish-load'
|
||||
)?.[1]
|
||||
expect(revealAfterLoad).toBeTypeOf('function')
|
||||
revealAfterLoad?.()
|
||||
windowHandlers['ready-to-show']()
|
||||
vi.advanceTimersByTime(10_000)
|
||||
expect(browserWindowInstance.show).not.toHaveBeenCalled()
|
||||
expect(showInactive).not.toHaveBeenCalled()
|
||||
expect(browserWindowInstance.maximize).not.toHaveBeenCalled()
|
||||
})
|
||||
} finally {
|
||||
vi.unstubAllEnvs()
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
it('ignores duplicate ready-to-show events after startup maximize has already run', () => {
|
||||
const { browserWindowInstance, windowHandlers } = createStartupRevealWindowFixture()
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import type { App, BrowserWindow } from 'electron'
|
||||
import { describe, expect, it, vi } from 'vitest'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import { focusExistingMainWindow } from './focus-existing-window'
|
||||
|
||||
type FakeWindowOptions = {
|
||||
@@ -78,7 +78,32 @@ function makeTimer(): {
|
||||
}
|
||||
}
|
||||
|
||||
afterEach(() => vi.unstubAllEnvs())
|
||||
|
||||
describe('focusExistingMainWindow', () => {
|
||||
it.each(['darwin', 'linux', 'win32'] as const)(
|
||||
'never restores or activates a background window on %s',
|
||||
(platform) => {
|
||||
vi.stubEnv('ORCA_BACKGROUND_LAUNCH', '1')
|
||||
vi.stubEnv('ORCA_E2E_FOREGROUND', '1')
|
||||
const app = makeFakeApp()
|
||||
const window = makeFakeWindow({ minimized: true })
|
||||
const timer = makeTimer()
|
||||
focusExistingMainWindow({
|
||||
app,
|
||||
getWindow: () => window,
|
||||
openWindow: vi.fn(),
|
||||
platform,
|
||||
setTimeout: timer.setTimeout
|
||||
})
|
||||
expect(app.focus).not.toHaveBeenCalled()
|
||||
for (const call of Object.values(window.calls)) {
|
||||
expect(call).not.toHaveBeenCalled()
|
||||
}
|
||||
expect(timer.scheduledMs()).toEqual([])
|
||||
}
|
||||
)
|
||||
|
||||
it('aggressively foregrounds an existing Windows window on second launch', () => {
|
||||
const app = makeFakeApp()
|
||||
const window = makeFakeWindow()
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
import type { App, BrowserWindow } from 'electron'
|
||||
import { isBackgroundLaunch, showWindowWithoutStealingFocus } from './foreground-activation-policy'
|
||||
import {
|
||||
isBackgroundLaunch,
|
||||
isWindowlessLaunch,
|
||||
showWindowWithoutStealingFocus
|
||||
} from './foreground-activation-policy'
|
||||
|
||||
type FocusTimer = (callback: () => void, ms: number) => unknown
|
||||
|
||||
@@ -34,7 +38,7 @@ function safelyFocusApp(app: Pick<App, 'focus'>): void {
|
||||
}
|
||||
|
||||
export function safelyRevealWindow(window: BrowserWindow): void {
|
||||
if (window.isDestroyed()) {
|
||||
if (window.isDestroyed() || isWindowlessLaunch()) {
|
||||
return
|
||||
}
|
||||
if (window.isMinimized()) {
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user