merge main into worktree technical improvements

This commit is contained in:
Neil
2026-09-05 05:06:49 -07:00
669 changed files with 49347 additions and 4041 deletions
@@ -91,7 +91,11 @@ jobs:
test -n "${CAPACITY_SERVICE_ACCOUNT}"
test -n "${DIRECTOR_RUNTIME_SERVICE_ACCOUNT}"
# Full history: the monitor evidence this job verifies is sealed at an ancestor commit,
# and the provenance check fails closed on a commit a shallow clone left out.
- uses: actions/checkout@v4
with:
fetch-depth: 0
- uses: pnpm/action-setup@v4
with: { package_json_file: cloud/package.json }
@@ -177,11 +181,12 @@ jobs:
env:
ORCA_RELAY_ADMIN_ID_TOKEN: ${{ steps.deploy-auth.outputs.id_token }}
run: |
RETRY_ARGS=()
if test "${WAVE_INDEX}" != 0; then RETRY_ARGS=(--retry-freshness); fi
# Freshness-only failures are publish lag, not health, on every wave
# including the first; the CLI still caps the retry at the wave's
# evidence-age budget, so this cannot mutate on aged evidence.
pnpm incident:relay-preflight -- \
--state-file "${OUTPUT_DIRECTORY}/relay-${MONITOR_RUN_ID}-dry-run.state.json" \
--wave-index "${WAVE_INDEX}" "${RETRY_ARGS[@]}"
--wave-index "${WAVE_INDEX}" --retry-freshness
- name: Require durable rehome disabled and exact selector
env:
@@ -271,10 +276,22 @@ jobs:
env:
ORCA_RELAY_ADMIN_ID_TOKEN: ${{ steps.deploy-auth.outputs.id_token }}
run: |
CURRENT_RUNTIME="$(curl --fail-with-body --max-time 30 \
--request POST "${CELL_ORIGIN}/v1/admin/runtime-status" \
--header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \
--header 'Content-Type: application/json' --data '{"v":1}')"
# A single transient 5xx (LB warm-up behind a fresh instance) must not
# fail a canary; 4xx (auth, generation mismatch) still fails fast.
admin_post() {
local out="${RUNNER_TEMP}/$1.json"
if ! curl --fail-with-body --max-time 30 \
--retry 3 --retry-delay 2 --retry-connrefused --output "${out}" \
--request POST "$2" \
--header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \
--header 'Content-Type: application/json' --data "$3"; then
cat "${out}" >&2
return 1
fi
cat "${out}"
}
CURRENT_RUNTIME="$(admin_post current-runtime \
"${CELL_ORIGIN}/v1/admin/runtime-status" '{"v":1}')"
# A rollback that failed between template apply and admission restore
# leaves the cell already on the rollback image; resume from that
# state instead of demanding the pre-rollback predecessor.
@@ -370,11 +387,9 @@ jobs:
if .regionalRehomeProtocol == null then "regionalRehomeProtocol" else empty end
] | if length > 0 then "runtime predecessor normalized legacy fields=" + join(",") else empty end' \
<<< "${CURRENT_RUNTIME}"
CURRENT_DIRECTOR_STATUS="$(curl --fail-with-body --max-time 30 \
--request POST "${DIRECTOR_ORIGIN}/v1/admin/cell-status" \
--header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \
--header 'Content-Type: application/json' \
--data "$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")"
CURRENT_DIRECTOR_STATUS="$(admin_post current-cell-status \
"${DIRECTOR_ORIGIN}/v1/admin/cell-status" \
"$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")"
SOURCE_INCARNATION="$(jq -er '.status.runtime.cellIncarnation' \
<<< "${CURRENT_DIRECTOR_STATUS}")"
if test "${ROLLBACK_RESUME}" = true && ! jq -e \
@@ -418,13 +433,13 @@ jobs:
# result's generation is authoritative either way.
ISOLATE_RESULT="$(node dev/scripts/prepare-relay-production-capacity-canary.mjs \
--director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \
--cell-id "${TARGET_CELL_ID}" --mode isolate)"
--cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode isolate)"
echo "${ISOLATE_RESULT}"
ISOLATE_GENERATION="$(jq -er '.generation' <<< "${ISOLATE_RESULT}")"
echo "SELECTOR_GENERATION_AFTER_ISOLATE=${ISOLATE_GENERATION}" >> "${GITHUB_ENV}"
node dev/scripts/prepare-relay-production-capacity-canary.mjs \
--director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \
--cell-id "${TARGET_CELL_ID}" --mode drain
--cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode drain
node dev/scripts/verify-relay-capacity-transition.mjs \
--director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \
--cell-id "${TARGET_CELL_ID}" --hard-cap "${EXPECTED_HARD_CAP}" \
@@ -486,6 +501,7 @@ jobs:
--rollback-image "${DESIRED_IMAGE}" \
--rehome-director-service-account "${DIRECTOR_RUNTIME_SERVICE_ACCOUNT}" \
--rehome-audience https://relay.onorca.dev/v1/admin/host-drain \
--regional-rehome-protocol "${DESIRED_REHOME_PROTOCOL}" \
| jq -e '.changes == 2' >/dev/null
fi
gcloud compute instance-groups managed wait-until "${MIG_NAME}" --stable \
@@ -511,7 +527,8 @@ jobs:
--unobserved-bound "${EXPECTED_UNOBSERVED_BOUND}" --image "${DESIRED_IMAGE}" \
--rollback-image "${IMAGE_REPOSITORY}@${CURRENT_IMAGE_DIGEST}" \
--rehome-director-service-account "${DIRECTOR_RUNTIME_SERVICE_ACCOUNT}" \
--rehome-audience https://relay.onorca.dev/v1/admin/host-drain
--rehome-audience https://relay.onorca.dev/v1/admin/host-drain \
--regional-rehome-protocol "${DESIRED_REHOME_PROTOCOL}"
terraform -chdir=infra/terraform apply -auto-approve \
"${RUNNER_TEMP}/relay-same-cap.tfplan"
gcloud compute instance-groups managed wait-until "${MIG_NAME}" --stable \
@@ -532,6 +549,20 @@ jobs:
env:
ORCA_RELAY_ADMIN_ID_TOKEN: ${{ steps.post-auth.outputs.id_token }}
run: |
# A single transient 5xx (LB warm-up behind a fresh instance) must not
# fail a canary; 4xx (auth, generation mismatch) still fails fast.
admin_post() {
local out="${RUNNER_TEMP}/$1.json"
if ! curl --fail-with-body --max-time 30 \
--retry 3 --retry-delay 2 --retry-connrefused --output "${out}" \
--request POST "$2" \
--header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \
--header 'Content-Type: application/json' --data "$3"; then
cat "${out}" >&2
return 1
fi
cat "${out}"
}
node dev/scripts/verify-relay-capacity-transition.mjs \
--director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \
--cell-id "${TARGET_CELL_ID}" --hard-cap "${EXPECTED_HARD_CAP}" \
@@ -539,19 +570,15 @@ jobs:
--heartbeat fresh --admission migration-only --draining forbidden \
--activity allowed --expected-image-digests "${DESIRED_IMAGE_DIGEST}" \
--regional-rehome-protocol "${DESIRED_REHOME_PROTOCOL}" --timeout-ms 900000
TARGET_RUNTIME="$(curl --fail-with-body --max-time 30 \
--request POST "${CELL_ORIGIN}/v1/admin/runtime-status" \
--header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \
--header 'Content-Type: application/json' --data '{"v":1}')"
TARGET_RUNTIME="$(admin_post target-runtime \
"${CELL_ORIGIN}/v1/admin/runtime-status" '{"v":1}')"
jq -e --arg digest "${DESIRED_IMAGE_DIGEST}" \
--argjson protocol "${DESIRED_REHOME_PROTOCOL}" \
'.imageDigest == $digest and (.regionalRehomeProtocol // 0) == $protocol' \
<<< "${TARGET_RUNTIME}" >/dev/null
TARGET_DIRECTOR_STATUS="$(curl --fail-with-body --max-time 30 \
--request POST "${DIRECTOR_ORIGIN}/v1/admin/cell-status" \
--header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \
--header 'Content-Type: application/json' \
--data "$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")"
TARGET_DIRECTOR_STATUS="$(admin_post target-cell-status \
"${DIRECTOR_ORIGIN}/v1/admin/cell-status" \
"$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")"
TARGET_INCARNATION="$(jq -er '.status.runtime.cellIncarnation' \
<<< "${TARGET_DIRECTOR_STATUS}")"
if test "${ROLLBACK_RESUME}" = true; then
@@ -588,7 +615,7 @@ jobs:
echo "MUTATION_STARTED=true" >> "${GITHUB_ENV}"
ACTIVATE_RESULT="$(node dev/scripts/prepare-relay-production-capacity-canary.mjs \
--director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \
--cell-id "${TARGET_CELL_ID}" --mode activate)"
--cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode activate)"
echo "${ACTIVATE_RESULT}"
SELECTOR_GENERATION_AFTER_ACTIVATE="$(jq -er '.generation' \
<<< "${ACTIVATE_RESULT}")"
@@ -627,7 +654,7 @@ jobs:
test "${MUTATION_STARTED:-false}" = true || exit 0
ISOLATE_RESULT="$(node dev/scripts/prepare-relay-production-capacity-canary.mjs \
--director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \
--cell-id "${TARGET_CELL_ID}" --mode isolate)"
--cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode isolate)"
echo "${ISOLATE_RESULT}"
# The isolate result carries the authoritative post-isolate generation;
# fixed offsets are wrong whenever an earlier isolate was a no-op.
@@ -87,13 +87,18 @@ jobs:
gate:
if: ${{ vars.ORCA_CLOUD_OPERATIONS_ENABLED == 'true' && (github.ref == 'refs/heads/main') }}
runs-on: blacksmith-2vcpu-ubuntu-2204
timeout-minutes: 10
# Headroom for the full-history checkout the canary provenance check needs.
timeout-minutes: 15
environment: production
outputs:
cells: ${{ steps.wave.outputs.cells }}
job-mode: ${{ steps.wave.outputs.job-mode }}
steps:
# Full history: the canary authority a batch verifies is sealed at an ancestor commit, and
# the provenance check fails closed on a commit a shallow clone left out.
- uses: actions/checkout@v4
with:
fetch-depth: 0
- uses: actions/setup-node@v4
with: { node-version: 24 }
@@ -95,7 +95,11 @@ jobs:
;;
esac
# Full history: the monitor evidence this job verifies is sealed at an ancestor commit,
# and the provenance check fails closed on a commit a shallow clone left out.
- uses: actions/checkout@v4
with:
fetch-depth: 0
- uses: actions/setup-node@v4
with:
+1
View File
@@ -807,6 +807,7 @@ jobs:
src/main/agent-hooks/windows-hook-payload-delivery.test.ts
src/main/windows/windows-pty-job.win32.test.ts
src/main/windows/windows-host-job.win32.test.ts
src/main/windows-live-tree-kill.win32.test.ts
src/main/wsl/wsl-runner.test.ts
src/main/wsl/wsl-guest-environment.test.ts
src/main/wsl/wsl-invocation-boundary.test.ts
@@ -6,7 +6,10 @@ import {
livePreflightGcloud,
runIncidentLivePreflight
} from './incident-live-preflight-cli.js'
import type { IncidentSample } from './incident-monitor.js'
import {
INCIDENT_MONITOR_THRESHOLDS,
type IncidentSample
} from './incident-monitor.js'
import type { AdmissionSelector } from './incident-selector.js'
const directories: string[] = []
@@ -313,7 +316,7 @@ describe('relay incident live preflight', () => {
it('retries freshness-only failures when explicitly requested', async () => {
const stale = sample()
stale.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt =
new Date(now - 180_001).toISOString()
new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString()
const missing = sample()
delete missing.sources['relay-logs']
const collect = vi.fn()
@@ -331,11 +334,44 @@ describe('relay incident live preflight', () => {
expect(wait).toHaveBeenNthCalledWith(2, 15_000)
})
it('retries a first-wave stale sample and passes on the fresh one', async () => {
const stale = sample()
stale.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt =
new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString()
const collect = vi.fn().mockResolvedValueOnce(stale).mockResolvedValueOnce(sample())
const wait = vi.fn(async () => undefined)
await expect(runIncidentLivePreflight(
['--state-file', stateFile(), '--wave-index', '0', '--retry-freshness'],
{ now: () => now, collect, wait }
)).resolves.toBeUndefined()
expect(collect).toHaveBeenCalledTimes(2)
expect(wait).toHaveBeenCalledOnce()
})
it('stops retrying when the next wait would exceed the evidence-age bound', async () => {
const completedAt = now - 290_000
const stale = sample()
stale.sources['cloud-monitoring']!.observedAt = new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString()
const collect = vi.fn(async () => stale)
const wait = vi.fn(async () => undefined)
await expect(runIncidentLivePreflight(
['--state-file', stateFile('strict', {
startedAt: new Date(completedAt - 17 * 60_000).toISOString(),
windowStartedAt: new Date(completedAt - 16 * 60_000).toISOString(),
lastSampleAt: new Date(completedAt - 30_000).toISOString(),
completedAt: new Date(completedAt).toISOString()
}), '--retry-freshness'],
{ now: () => now, collect, wait }
)).rejects.toThrow('cloud-monitoring/source_stale')
expect(collect).toHaveBeenCalledOnce()
expect(wait).not.toHaveBeenCalled()
})
it('does not retry a threshold failure', async () => {
const unhealthy = sample()
unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.value = 0.9
unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt =
new Date(now - 180_001).toISOString()
new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString()
const collect = vi.fn(async () => unhealthy)
const wait = vi.fn(async () => undefined)
await expect(runIncidentLivePreflight(
@@ -348,7 +384,7 @@ describe('relay incident live preflight', () => {
it('fails closed after the bounded freshness retry window', async () => {
const stale = sample()
stale.sources['cloud-monitoring']!.observedAt = new Date(now - 180_001).toISOString()
stale.sources['cloud-monitoring']!.observedAt = new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString()
const collect = vi.fn(async () => stale)
const wait = vi.fn(async () => undefined)
await expect(runIncidentLivePreflight(
@@ -7,6 +7,7 @@ import { suppliedIdentityToken } from './incident-monitor-cli.js'
import { AdmissionSelectorSchema, type AdmissionSelector } from './incident-selector.js'
import {
evaluateIncidentSample,
FRESHNESS_FAILURE_CODES,
preDrainDryRunPassed,
type IncidentSample
} from './incident-monitor.js'
@@ -18,12 +19,6 @@ const MONITOR_EVIDENCE_MAX_AGE_MS = 5 * 60_000
// Matches the same-cap cell job timeout-minutes; bounds each predecessor wave.
const WAVE_PREDECESSOR_TIMEOUT_MS = 75 * 60_000
const WAVE_INDEX_PATTERN = /^[0-3]$/
const FRESHNESS_FAILURE_CODES = new Set([
'signal_missing',
'signal_stale',
'source_missing',
'source_stale'
])
export function livePreflightGcloud(
gcloud: ReturnType<typeof createGcloudClient>,
@@ -173,7 +168,11 @@ export async function runIncidentLivePreflight(
const freshnessOnly = evaluation.failures.every((failure) =>
FRESHNESS_FAILURE_CODES.has(failure.code)
)
if (!freshnessOnly || attempt === attempts) {
// Waiting must never carry the mutation past the same evidence-age bound
// the entry check enforces, so the wave budget also caps the retry window.
const budgetExhausted =
now() + FRESHNESS_RETRY_INTERVAL_MS - completedAt > maxEvidenceAgeMs
if (!freshnessOnly || attempt === attempts || budgetExhausted) {
throw new Error(
`relay live preflight failed: ${evaluation.failures
.map((failure) => `${failure.source}/${failure.code}`)
@@ -50,6 +50,8 @@ const StateSchema = z.object({
continuityEvents: z.array(z.object({
recordedAt: z.string(),
windowSequence: z.number().int().nonnegative(),
// Pre-2026-09-05 state files predate tolerated freshness gaps.
tolerated: z.boolean().default(false),
failures: z.array(z.object({
code: z.string(),
source: z.enum(['active-probe', 'cloud-monitoring', 'relay-logs', 'director-admin']),
@@ -93,7 +93,7 @@ describe('incident monitor sources', () => {
})
it('zero-fills an expired sparse lock-wait point', async () => {
let pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs
let pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs
const fetchImpl: typeof fetch = async () => Response.json({
timeSeries: [{
points: [{
@@ -141,7 +141,7 @@ describe('incident monitor sources', () => {
it('freshens a sparse zero without masking a recent nonzero lock wait', async () => {
let value = 0
const pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs
const pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs
const readAt = now + 11_879
const fetchImpl: typeof fetch = async () => Response.json({
timeSeries: [{
@@ -95,7 +95,7 @@ export const GOOGLE_METRICS: GoogleMetricDefinition[] = [
'resource.type="cloudsql_database" AND metric.label."wait_event_type"="Lock"',
aggregation: 'latest-max',
emptyIsZero: true,
zeroAfterMs: INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs
zeroAfterMs: INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs
},
{
signal: 'cloud_sql.deadlocks',
+246 -12
View File
@@ -2,6 +2,7 @@ import { describe, expect, it } from 'vitest'
import {
evaluateIncidentSample,
INCIDENT_CHECKPOINT_MINUTES,
INCIDENT_FRESHNESS_TOLERANCE_SAMPLES,
INCIDENT_MONITOR_THRESHOLDS,
INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS,
initialIncidentMonitorState,
@@ -182,12 +183,49 @@ describe('incident monitor evaluator', () => {
code: 'source_missing',
source: 'relay-logs'
})
const stale = healthySample(startedAt - 180_001)
const stale = healthySample(
startedAt - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1
)
const failures = evaluateIncidentSample(stale, startedAt).failures
expect(failures.some((failure) => failure.source === 'cloud-monitoring')).toBe(true)
expect(failures.some((failure) => failure.source === 'active-probe')).toBe(true)
})
// Why: production run 33944873727 at 2026-09-05T04:46:09Z read
// cloud_sql.lock_waits 189 286 ms old and restarted a 15-minute window on
// Google's publish lag. Cloud SQL documents 60 s sampling plus up to 165 s of
// invisibility, so that age is Google's clock, not our fleet.
it('reads a 189-second cloud signal as fresh and holds the other sources at 180 s', () => {
const lagged = healthySample()
lagged.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] =
signal(0, startedAt - 189_286)
expect(evaluateIncidentSample(lagged, startedAt)).toMatchObject({
status: 'green',
failures: []
})
const laggedDirector = healthySample()
laggedDirector.sources['director-admin']!.observedAt =
new Date(startedAt - 189_286).toISOString()
expect(evaluateIncidentSample(laggedDirector, startedAt).failures).toContainEqual(
expect.objectContaining({ code: 'source_stale', source: 'director-admin' })
)
})
it('still fails a cloud signal past the documented publish lag', () => {
const dark = healthySample()
dark.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = signal(
0,
startedAt - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1
)
expect(evaluateIncidentSample(dark, startedAt).failures).toContainEqual(
expect.objectContaining({
code: 'signal_stale',
source: 'cloud-monitoring',
signal: 'cloud_sql.lock_waits'
})
)
})
it('freezes on SQL, director, relay pool, heartbeat, and migration breaches', () => {
const sample = healthySample()
sample.sources['cloud-monitoring']!.signals['cloud_sql.cpu'] = signal(0.81)
@@ -592,7 +630,7 @@ describe('incident monitor lifecycle', () => {
'restarts a %i-minute continuous window after stale telemetry',
async (durationMinutes) => {
let now = startedAt
let staleInjected = false
let staleSamples = INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1
const checkpoints: Array<[number, number]> = []
const state = initialIncidentMonitorState({
incidentId: 'incident-1',
@@ -612,9 +650,11 @@ describe('incident monitor lifecycle', () => {
now += ms
},
collect: async () => {
if (!staleInjected && now === startedAt + 5 * 60_000) {
staleInjected = true
return healthySample(now - 180_001)
if (staleSamples > 0 && now >= startedAt + 5 * 60_000) {
staleSamples--
return healthySample(
now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1
)
}
return healthySample(now)
},
@@ -623,16 +663,20 @@ describe('incident monitor lifecycle', () => {
checkpoints.push([summary.windowSequence, summary.checkpointMinute])
}
})
const restartMinute = 5 + INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1
expect(result.windowSequence).toBe(1)
expect(result.windowStartedAt).toBe(
new Date(startedAt + 6 * 60_000).toISOString()
new Date(startedAt + restartMinute * 60_000).toISOString()
)
expect(result.completedAt).toBe(
new Date(startedAt + (durationMinutes + 6) * 60_000).toISOString()
new Date(startedAt + (durationMinutes + restartMinute) * 60_000).toISOString()
)
expect(result.sampleCount).toBe(durationMinutes + 1)
expect(result.continuityEvents).toHaveLength(1)
expect(result.continuityEvents[0]!.failures).toEqual(
expect(result.continuityEvents.map((event) => event.tolerated)).toEqual([
...Array<boolean>(INCIDENT_FRESHNESS_TOLERANCE_SAMPLES).fill(true),
false
])
expect(result.continuityEvents.at(-1)!.failures).toEqual(
expect.arrayContaining([
expect.objectContaining({ code: 'source_stale' })
])
@@ -642,6 +686,188 @@ describe('incident monitor lifecycle', () => {
}
)
// Why: run 33944873727 on 2026-09-05 restarted at 04:46:09Z on a single
// 189-second cloud reading and then blew the 25-minute lineage cap, so a
// green fleet produced no verdict at all. One unread sample now continues the
// window; the sample is still checked against every threshold it can read.
it('carries a 15-minute window through a single stale cloud sample', async () => {
let now = startedAt
const state = initialIncidentMonitorState({
incidentId: 'incident-1',
environment: 'production',
expectedSelector: selector,
preDrainDryRun: true,
migrationPolicy: 'strict',
recoverySourceCellId: null,
capacityCellId: null,
startedAt: new Date(startedAt).toISOString(),
durationMinutes: 15,
intervalMs: 60_000
})
const result = await runIncidentMonitor(state, {
now: () => now,
wait: async (ms) => {
now += ms
},
collect: async () => {
const sample = healthySample(now)
if (now === startedAt + 10 * 60_000) {
sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] =
signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1)
}
return sample
},
persist: async () => {},
checkpoint: async () => {}
})
expect(result.windowSequence).toBe(0)
expect(result.windowStartedAt).toBe(new Date(startedAt).toISOString())
expect(result.completedAt).toBe(new Date(startedAt + 15 * 60_000).toISOString())
expect(result.sampleCount).toBe(16)
expect(result.frozenAt).toBeNull()
expect(result.continuityEvents).toEqual([{
recordedAt: new Date(startedAt + 10 * 60_000).toISOString(),
windowSequence: 0,
tolerated: true,
failures: [expect.objectContaining({
code: 'signal_stale',
source: 'cloud-monitoring',
signal: 'cloud_sql.lock_waits'
})]
}])
expect(preDrainDryRunPassed(result)).toBe(true)
})
it('gives a signal a fresh budget only after it reads fresh again', async () => {
let now = startedAt
const staleMinutes = new Set([3, 5, 6, 9, 10])
const state = initialIncidentMonitorState({
incidentId: 'incident-1',
environment: 'production',
expectedSelector: selector,
preDrainDryRun: true,
migrationPolicy: 'strict',
recoverySourceCellId: null,
capacityCellId: null,
startedAt: new Date(startedAt).toISOString(),
durationMinutes: 15,
intervalMs: 60_000
})
const result = await runIncidentMonitor(state, {
now: () => now,
wait: async (ms) => {
now += ms
},
collect: async () => {
const sample = healthySample(now)
if (staleMinutes.has((now - startedAt) / 60_000)) {
sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] =
signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1)
}
return sample
},
persist: async () => {},
checkpoint: async () => {}
})
expect(result.windowSequence).toBe(0)
expect(result.continuityEvents).toHaveLength(staleMinutes.size)
expect(result.continuityEvents.every((event) => event.tolerated)).toBe(true)
expect(preDrainDryRunPassed(result)).toBe(true)
})
it('does not hand a resumed monitor a fresh tolerance budget', async () => {
let now = startedAt + 3 * 60_000
const resumed = {
...initialIncidentMonitorState({
incidentId: 'incident-1',
environment: 'production',
expectedSelector: selector,
preDrainDryRun: true,
migrationPolicy: 'strict',
recoverySourceCellId: null,
capacityCellId: null,
startedAt: new Date(startedAt).toISOString(),
durationMinutes: 15,
intervalMs: 60_000
}),
windowStartedAt: new Date(startedAt).toISOString(),
lastSampleAt: new Date(startedAt + 2 * 60_000).toISOString(),
sampleCount: 3,
totalSampleCount: 3,
continuityEvents: Array.from(
{ length: INCIDENT_FRESHNESS_TOLERANCE_SAMPLES },
(_, index) => ({
recordedAt: new Date(startedAt + (index + 1) * 60_000).toISOString(),
windowSequence: 0,
tolerated: true,
failures: [{
code: 'signal_stale',
source: 'cloud-monitoring' as const,
signal: 'cloud_sql.lock_waits'
}]
})
)
}
const stop = new Error('stop after the resumed sample')
await expect(runIncidentMonitor(resumed, {
now: () => now,
wait: async () => {
throw stop
},
collect: async () => {
const sample = healthySample(now)
sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] =
signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1)
return sample
},
persist: async (state) => {
expect(state.windowSequence).toBe(1)
expect(state.windowStartedAt).toBeNull()
expect(state.continuityEvents.at(-1)!.tolerated).toBe(false)
},
checkpoint: async () => {}
})).rejects.toThrow(stop)
})
it('freezes on a threshold breach that arrives with a tolerated stale signal', async () => {
let now = startedAt
const state = initialIncidentMonitorState({
incidentId: 'incident-1',
environment: 'production',
expectedSelector: selector,
preDrainDryRun: true,
migrationPolicy: 'strict',
recoverySourceCellId: null,
capacityCellId: null,
startedAt: new Date(startedAt).toISOString(),
durationMinutes: 15,
intervalMs: 60_000
})
const result = await runIncidentMonitor(state, {
now: () => now,
wait: async (ms) => {
now += ms
},
collect: async () => {
const sample = healthySample(now)
if (now === startedAt + 2 * 60_000) {
sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] =
signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1)
sample.sources['cloud-monitoring']!.signals['cloud_sql.cpu'] = signal(0.81, now)
}
return sample
},
persist: async () => {},
checkpoint: async () => {}
})
expect(result.frozenAt).toBe(new Date(startedAt + 2 * 60_000).toISOString())
expect(result.failures).toContainEqual(expect.objectContaining({
code: 'threshold_max',
signal: 'cloud_sql.cpu'
}))
expect(preDrainDryRunPassed(result)).toBe(false)
})
it('resets at the next fresh sample after a runner gap', async () => {
let now = startedAt + 10 * 60_000
const state = {
@@ -690,13 +916,21 @@ describe('incident monitor lifecycle', () => {
durationMinutes: 15,
intervalMs: 60_000
})
let staleSamples = INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1
const result = await runIncidentMonitor(state, {
now: () => now,
wait: async (ms) => {
now += ms
},
collect: async () =>
healthySample(now === startedAt + 10 * 60_000 ? now - 180_001 : now),
collect: async () => {
if (staleSamples > 0 && now >= startedAt + 10 * 60_000) {
staleSamples--
return healthySample(
now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1
)
}
return healthySample(now)
},
persist: async () => {},
checkpoint: async () => {}
})
@@ -706,7 +940,7 @@ describe('incident monitor lifecycle', () => {
)
expect(result.frozenAt).not.toBeNull()
expect(result.windowSequence).toBe(1)
expect(result.sampleCount).toBe(15)
expect(result.sampleCount).toBe(13)
expect(result.failures).toContainEqual({
code: 'continuity_deadline_exceeded',
source: 'active-probe',
+99 -6
View File
@@ -6,7 +6,27 @@ import {
export const INCIDENT_MONITOR_THRESHOLDS = {
activeProbeMaxAgeMs: 60_000,
cloudDataMaxAgeMs: 180_000,
// Why: Cloud Monitoring publishes on Google's clock, not ours. Per the metric
// list read 2026-09-05, Cloud Run instance_count / cpu / memory /
// max_request_concurrencies / request_count are "Sampled every 60 seconds.
// After sampling, data is not visible for up to 120 seconds" (60+120=180 s),
// and Cloud SQL cpu / memory / num_backends / backends_in_wait /
// deadlock_count say "up to 165 seconds" (60+165=225 s). Window-sum signals
// age differently: observedAt is the newest point in the 5-minute query
// window, so a label series that stops emitting reads as 300 s old while its
// summed value is still complete. 330 s clears the worst of the three (the
// 300 s query window) plus ~30 s of collect-to-evaluate latency. The old
// 180 s bar restarted healthy 15-minute windows at 181 s, 189 s and 255 s on
// 2026-09-04/05, once burning the whole 25-minute lineage with no verdict.
cloudDataMaxAgeMs: 330_000,
// Why: the director admin API answers live on our own request, so hold its
// freshness bar where it sat while it shared cloudDataMaxAgeMs.
directorAdminMaxAgeMs: 180_000,
// Why: how long a nonzero backends-in-wait point is carried before it reads as
// zero. Held at the pre-2026-09-05 cloud bar: carrying it for the full
// cloudDataMaxAgeMs would hand the evaluator a point older than its own
// freshness bar as soon as collection latency is added.
cloudLockWaitCarryMs: 180_000,
relayLogMaxAgeMs: 180_000,
heartbeatMaxAgeMs: 45_000,
endpointLatencyMs: 2_000,
@@ -175,6 +195,7 @@ export type IncidentMonitorState = {
continuityEvents: {
recordedAt: string
windowSequence: number
tolerated: boolean
failures: IncidentFailure[]
}[]
frozenAt: string | null
@@ -307,7 +328,7 @@ const SOURCE_MAX_AGE: Record<IncidentSourceName, number> = {
'active-probe': INCIDENT_MONITOR_THRESHOLDS.activeProbeMaxAgeMs,
'cloud-monitoring': INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs,
'relay-logs': INCIDENT_MONITOR_THRESHOLDS.relayLogMaxAgeMs,
'director-admin': INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs
'director-admin': INCIDENT_MONITOR_THRESHOLDS.directorAdminMaxAgeMs
}
function ageMs(timestamp: string, nowMs: number): number {
@@ -608,14 +629,59 @@ function checkpointMinutes(durationMinutes: number): number[] {
return INCIDENT_CHECKPOINT_MINUTES.filter((minute) => minute <= durationMinutes)
}
const CONTINUITY_FAILURE_CODES = new Set([
'collector_failed',
'monitor_gap',
// Freshness-only failures: we could not read a signal this sample. Distinct from
// collector_failed / monitor_gap, where the whole sample is absent.
export const FRESHNESS_FAILURE_CODES = new Set([
'signal_missing',
'signal_stale',
'source_missing',
'source_stale'
])
const CONTINUITY_FAILURE_CODES = new Set([
'collector_failed',
'monitor_gap',
...FRESHNESS_FAILURE_CODES
])
// Why: Cloud Monitoring overshoots its own publish bar, and one unread sample is
// not evidence of an unhealthy fleet. Under the 25-minute lineage cap a restart
// past minute 10 costs the entire verdict, so a healthy fleet produced none on
// 2026-09-05. A signal may miss this many consecutive samples before the window
// restarts; the sample is still evaluated against every threshold it can read,
// and a threshold breach still freezes the run outright.
export const INCIDENT_FRESHNESS_TOLERANCE_SAMPLES = 2
function freshnessKey(failure: IncidentFailure): string {
return `${failure.source}/${failure.signal ?? '*'}`
}
// Rebuild the per-signal tolerated streak from the trailing continuity events so a
// resumed monitor cannot hand a signal a fresh budget.
function resumeFreshnessStreaks(
state: IncidentMonitorState
): Map<string, number> {
const events = state.continuityEvents
const streaks = new Map<string, number>()
const last = events[events.length - 1]
if (!last?.tolerated) return streaks
for (const key of new Set(last.failures.map(freshnessKey))) {
let streak = 0
let laterAt: number | null = null
for (let index = events.length - 1; index >= 0; index--) {
const event = events[index]!
const recordedAt = Date.parse(event.recordedAt)
if (!event.tolerated) break
if (laterAt !== null && laterAt - recordedAt > state.intervalMs * 1.5) break
if (!event.failures.some((failure) => freshnessKey(failure) === key)) break
streak++
laterAt = recordedAt
}
streaks.set(key, streak)
}
return streaks
}
function resetContinuousWindow(
state: IncidentMonitorState,
recordedAt: string,
@@ -631,6 +697,7 @@ function resetContinuousWindow(
state.continuityEvents.push({
recordedAt,
windowSequence: state.windowSequence,
tolerated: false,
failures
})
}
@@ -681,6 +748,7 @@ export async function runIncidentMonitor(
await dependencies.persist(state)
return state
}
const freshnessStreaks = resumeFreshnessStreaks(state)
while (state.completedAt === null) {
if (dependencies.now() > lineageDeadlineMs) {
completeContinuityDeadline(state, dependencies.now(), lineageStartMs)
@@ -715,9 +783,34 @@ export async function runIncidentMonitor(
const thresholdFailures = evaluation.failures.filter((failure) =>
!CONTINUITY_FAILURE_CODES.has(failure.code)
)
if (continuityFailures.length > 0) {
const toleratedKeys = new Set(
state.windowStartedAt !== null &&
continuityFailures.length > 0 &&
continuityFailures.every((failure) => FRESHNESS_FAILURE_CODES.has(failure.code))
? continuityFailures.map(freshnessKey)
: []
)
for (const key of [...freshnessStreaks.keys()]) {
if (!toleratedKeys.has(key)) freshnessStreaks.delete(key)
}
let tolerated = toleratedKeys.size > 0
for (const key of toleratedKeys) {
const streak = (freshnessStreaks.get(key) ?? 0) + 1
freshnessStreaks.set(key, streak)
if (streak > INCIDENT_FRESHNESS_TOLERANCE_SAMPLES) tolerated = false
}
if (continuityFailures.length > 0 && !tolerated) {
freshnessStreaks.clear()
resetContinuousWindow(state, evaluation.evaluatedAt, continuityFailures)
} else {
if (tolerated) {
state.continuityEvents.push({
recordedAt: evaluation.evaluatedAt,
windowSequence: state.windowSequence,
tolerated: true,
failures: continuityFailures
})
}
if (state.windowStartedAt === null) {
state.windowStartedAt = evaluation.evaluatedAt
}
@@ -13,6 +13,41 @@ const runService = {
latestReadyRevision: 'projects/project/revisions/revision-one'
}
const sleepingStagingGcloud: GcloudClient = { accessToken: async () => 'a'.repeat(40) }
// Staging's Cloud SQL is stopped, so this inventory reads REST only and probes no endpoint.
type MigOutcome = 'ok' | 'throw' | 'missing'
const sleepingStagingFetch = (migOutcome: (migName: string) => MigOutcome): typeof fetch =>
async (input) => {
const url = new URL(String(input))
if (url.hostname === 'run.googleapis.com') return Response.json(runService)
if (url.hostname === 'sqladmin.googleapis.com') return Response.json({
state: 'STOPPED',
databaseVersion: 'POSTGRES_17',
settings: { activationPolicy: 'NEVER', availabilityType: 'ZONAL', tier: 'db-custom-1-3840' }
})
if (url.hostname === 'certificatemanager.googleapis.com') return Response.json({
managed: { domains: ['*.relay-staging.onorca.dev'], state: 'ACTIVE' }
})
if (url.pathname.includes('/instanceGroupManagers/')) {
const name = url.pathname.split('/').at(-1)!
const outcome = migOutcome(name)
if (outcome === 'throw') throw new TypeError('fetch failed')
if (outcome === 'missing') return new Response(null, { status: 404 })
return Response.json({
name,
targetSize: 0,
size: '0',
instanceGroup: `projects/project/zones/zone/instanceGroups/${name}`,
instanceTemplate: `projects/project/global/instanceTemplates/template-${name}`,
status: { isStable: true }
})
}
if (url.pathname.includes('/instanceTemplates/')) return Response.json({ properties: {} })
if (url.pathname.endsWith('/getHealth')) return Response.json([])
throw new Error(`Unexpected request to ${url.hostname}${url.pathname}`)
}
describe('readResourceInventory', () => {
it('does not delay a healthy endpoint sample', async () => {
let calls = 0
@@ -249,6 +284,74 @@ describe('readResourceInventory', () => {
expect(JSON.stringify(result)).not.toContain('SECRET_TEXT')
})
it('re-asks a MIG read that failed once before calling a cell powered-unknown', async () => {
const parkedCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]!
const waits: number[] = []
let parkedMigCalls = 0
const result = await readResourceInventory(
RELAY_OPS_ENVIRONMENTS.staging,
sleepingStagingGcloud,
sleepingStagingFetch((migName) => {
if (!migName.endsWith(parkedCell.hostname)) return 'ok'
parkedMigCalls += 1
return parkedMigCalls === 1 ? 'throw' : 'ok'
}),
{ wait: async (ms) => { waits.push(ms) } }
)
const parked = result.cells.find((cell) => cell.cellId === parkedCell.cellId)!
// The MIG was fine and parked at zero; one transient read must not erase that reading.
expect(parked.targetSize).toBe(0)
expect(parkedMigCalls).toBe(2)
expect(waits).toEqual([1_000])
expect(result.warnings).toEqual([])
})
it('reports a MIG unavailable only when the retry fails too', async () => {
const parkedCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]!
const waits: number[] = []
let parkedMigCalls = 0
const result = await readResourceInventory(
RELAY_OPS_ENVIRONMENTS.staging,
sleepingStagingGcloud,
sleepingStagingFetch((migName) => {
if (!migName.endsWith(parkedCell.hostname)) return 'ok'
parkedMigCalls += 1
return 'throw'
}),
{ wait: async (ms) => { waits.push(ms) } }
)
const parked = result.cells.find((cell) => cell.cellId === parkedCell.cellId)!
expect(parked.targetSize).toBeNull()
expect(parked.backendHealth).toBe('unknown')
expect(parkedMigCalls).toBe(2)
expect(waits).toEqual([1_000])
expect(result.warnings).toEqual([
`${parkedCell.hostname.toUpperCase()} MIG inventory is unavailable.`
])
})
it('does not re-ask a MIG read the API answered with 404', async () => {
const missingCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]!
const waits: number[] = []
let missingMigCalls = 0
const result = await readResourceInventory(
RELAY_OPS_ENVIRONMENTS.staging,
sleepingStagingGcloud,
sleepingStagingFetch((migName) => {
if (!migName.endsWith(missingCell.hostname)) return 'ok'
missingMigCalls += 1
return 'missing'
}),
{ wait: async (ms) => { waits.push(ms) } }
)
expect(result.cells.find((cell) => cell.cellId === missingCell.cellId)!.targetSize).toBeNull()
expect(missingMigCalls).toBe(1)
expect(waits).toEqual([])
})
it('represents missing credentials as unknown inventory, never sleeping', async () => {
const gcloud: GcloudClient = {
accessToken: async () => { throw new Error('sensitive context') }
+35 -5
View File
@@ -103,6 +103,8 @@ export type ResourceInventory = {
const unavailableEndpoint = (): EndpointHealth => ({ health: null, ready: null, latencyMs: null })
const independentEndpointRetryDelayMs = 11_000
const transientProbeRetryDelayMs = 1_000
const sleep = async (ms: number): Promise<void> =>
await new Promise((resolvePromise) => setTimeout(resolvePromise, ms))
function finalSegment(value: string): string {
return value.split('/').at(-1) ?? value
@@ -121,6 +123,12 @@ function parseService(value: unknown): ServiceInventory {
}
}
class GoogleApiError extends Error {
constructor(readonly status: number) {
super(`Google API returned ${status}`)
}
}
async function googleRequest(
fetchImpl: typeof fetch,
token: string,
@@ -135,10 +143,24 @@ async function googleRequest(
},
signal: AbortSignal.timeout(30_000)
})
if (!response.ok) throw new Error(`Google API returned ${response.status}`)
if (!response.ok) throw new GoogleApiError(response.status)
return await response.json()
}
// A 404 is the API's answer about the resource; anything else is the absence of a reading, so re-ask.
async function readOnceMore(
read: () => Promise<unknown>,
wait: (ms: number) => Promise<void>
): Promise<unknown> {
try {
return await read()
} catch (error) {
if (error instanceof GoogleApiError && error.status === 404) throw error
await wait(transientProbeRetryDelayMs)
return await read()
}
}
// A reading the endpoint actually produced: ok is its answer, latencyMs is that answer's round trip.
type PathReading = { ok: boolean; latencyMs: number | null }
@@ -201,8 +223,7 @@ export async function probeEndpointHealth(
options: EndpointProbeOptions = {}
): Promise<EndpointHealth> {
const requiresReady = options.requiresReady ?? true
const wait = options.wait ??
(async (ms: number) => await new Promise((resolvePromise) => setTimeout(resolvePromise, ms)))
const wait = options.wait ?? sleep
const accepted = (probe: EndpointHealth): boolean =>
probe.health === true &&
(!requiresReady || probe.ready === true) &&
@@ -325,11 +346,17 @@ function unavailableInventory(environment: RelayOpsEnvironment, warning: string)
}
}
export type ResourceInventoryOptions = {
wait?: (ms: number) => Promise<void>
}
export async function readResourceInventory(
environment: RelayOpsEnvironment,
gcloud: GcloudClient,
fetchImpl: typeof fetch = fetch
fetchImpl: typeof fetch = fetch,
options: ResourceInventoryOptions = {}
): Promise<ResourceInventory> {
const wait = options.wait ?? sleep
let token: string
try {
token = await gcloud.accessToken()
@@ -356,7 +383,10 @@ export async function readResourceInventory(
token,
`https://certificatemanager.googleapis.com/v1/projects/${environment.project}/locations/global/certificates/${environment.certificateName}`
),
...environment.cells.map((cell) => googleRequest(fetchImpl, token, migUrl(cell)))
// One transient Compute read must never become a verdict on a cell's power state.
...environment.cells.map((cell) =>
readOnceMore(async () => await googleRequest(fetchImpl, token, migUrl(cell)), wait)
)
])
const warnings: string[] = []
const directorValue = parsed(settled[0]!, RunServiceSchema, 'Director service inventory is unavailable.', warnings)
@@ -1,4 +1,5 @@
import { pathToFileURL } from 'node:url'
import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs'
import { inspectAdmissionSelector } from './relay-admission-selector.mjs'
const DIRECTOR_ORIGIN = 'https://relay.onorca.dev'
@@ -229,15 +230,20 @@ export async function recoverRegionalRehomeEnable(config, post) {
export async function operateRegionalRehome(config, dependencies = {}) {
const fetchImpl = dependencies.fetch ?? fetch
const post = dependencies.post ?? (async (path, body) => await responseJson(
await fetchImpl(`${config.directorOrigin}${path}`, {
method: 'POST',
headers: {
authorization: `Bearer ${config.token}`,
'content-type': 'application/json'
// Generation-guarded writes make a retry a no-op or an explicit mismatch, never a double apply.
await fetchAdminOnceMore(
fetchImpl,
`${config.directorOrigin}${path}`,
{
method: 'POST',
headers: {
authorization: `Bearer ${config.token}`,
'content-type': 'application/json'
},
body: JSON.stringify(body)
},
body: JSON.stringify(body),
signal: AbortSignal.timeout(30_000)
}),
{ wait: dependencies.wait }
),
path
))
if (config.mode === 'recover-enable') {
@@ -263,3 +263,55 @@ test('main executes recovery mode and emits verified disabled control', async ()
control: control(6, false)
})
})
test('retries a transient 503 on the director control endpoint', async () => {
const config = parseRegionalRehomeArguments(
argumentsFor('inspect'),
{ ORCA_RELAY_ADMIN_ID_TOKEN: 'token' }
)
const paths = []
let selectorCalls = 0
const result = await operateRegionalRehome(config, {
wait: async () => {},
fetch: async (url) => {
const path = new URL(url).pathname
paths.push(path)
if (path === '/v1/admin/admission-selector/status') {
selectorCalls += 1
// The first read of each admin path 503s the way a warming instance does.
if (selectorCalls === 1) return new Response('warming up', { status: 503 })
return Response.json({ selector: { generation: 11, membership } })
}
if (paths.filter((value) => value === path).length === 1) {
return new Response('warming up', { status: 503 })
}
return Response.json({ v: 1, control: control(4, false) })
}
})
assert.equal(result.control.generation, 4)
assert.deepEqual(paths, [
'/v1/admin/admission-selector/status',
'/v1/admin/admission-selector/status',
'/v1/admin/regional-rehome-control',
'/v1/admin/regional-rehome-control'
])
})
test('fails when both attempts at the director control endpoint return 503', async () => {
const config = parseRegionalRehomeArguments(
argumentsFor('inspect'),
{ ORCA_RELAY_ADMIN_ID_TOKEN: 'token' }
)
let calls = 0
await assert.rejects(
operateRegionalRehome(config, {
wait: async () => {},
fetch: async () => {
calls += 1
return new Response('warming up', { status: 503 })
}
}),
/returned 503/
)
assert.equal(calls, 2)
})
@@ -1,10 +1,12 @@
import { pathToFileURL } from 'node:url'
import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs'
import {
applyExactAdmissionSelector,
inspectAdmissionSelector,
membershipWithStates,
selectorCellState
} from './relay-admission-selector.mjs'
import { SAME_CAP_CELLS } from './relay-production-same-cap-wave.mjs'
const DIRECTOR_ORIGIN = 'https://relay.onorca.dev'
export const PRODUCTION_CAPACITY_CELL_IDS = [
@@ -30,6 +32,9 @@ function cellOrigin(cellId) {
return `https://${cellId.slice('production-gce-'.length)}.relay.onorca.dev`
}
// The same-cap roll covers the Asia cells the US-only capacity rollout never touches.
const APPROVED_CELL_LISTS = { 'same-cap': SAME_CAP_CELLS }
export function parseProductionCapacityCellArguments(argv) {
const values = {}
for (let index = 0; index < argv.length; index += 2) {
@@ -41,8 +46,15 @@ export function parseProductionCapacityCellArguments(argv) {
if (!['isolate', 'drain', 'activate'].includes(values.mode)) {
throw new Error('--mode must be isolate, drain, or activate')
}
const approvedList = values['approved-cells']
if (approvedList !== undefined && !APPROVED_CELL_LISTS[approvedList]) {
throw new Error('--approved-cells is not a known allowlist')
}
const approvedCellIds = approvedList === undefined
? PRODUCTION_CAPACITY_CELL_IDS
: APPROVED_CELL_LISTS[approvedList]
const cellId = values['cell-id']
if (!PRODUCTION_CAPACITY_CELL_IDS.includes(cellId)) {
if (!approvedCellIds.includes(cellId)) {
throw new Error('production capacity target is not approved')
}
const expectedCellOrigin = cellOrigin(cellId)
@@ -72,12 +84,16 @@ export async function prepareProductionCapacityCell(config, overrides = {}) {
if (!token || token.length > 8_192) throw new Error('admin identity token is unavailable')
const postAt = async (origin, path, body) =>
await responseJson(
await fetchImpl(`${origin}${path}`, {
method: 'POST',
headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' },
body: JSON.stringify(body),
signal: AbortSignal.timeout(30_000)
}),
await fetchAdminOnceMore(
fetchImpl,
`${origin}${path}`,
{
method: 'POST',
headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' },
body: JSON.stringify(body)
},
{ wait: overrides.wait }
),
path
)
const post = async (path, body) => await postAt(config.directorOrigin, path, body)
@@ -104,6 +104,47 @@ describe('production Relay capacity cell admission', () => {
'--cell-id', 'production-gce-c7',
'--mode', 'isolate'
]), /origin is not exact/)
assert.throws(() => parseProductionCapacityCellArguments([
'--director-origin', 'https://relay.onorca.dev',
'--cell-origin', 'https://c27.relay.onorca.dev',
'--cell-id', 'production-gce-c27',
'--mode', 'isolate'
]), /not approved/)
})
it('admits the same-cap Asia cells only under the same-cap allowlist', () => {
for (const cellId of ['production-gce-c27', 'production-gce-c28', 'production-gce-c29']) {
const hostname = cellId.slice('production-gce-'.length)
assert.deepEqual(parseProductionCapacityCellArguments([
'--director-origin', 'https://relay.onorca.dev',
'--cell-origin', `https://${hostname}.relay.onorca.dev`,
'--cell-id', cellId,
'--approved-cells', 'same-cap',
'--mode', 'isolate'
]), {
directorOrigin: 'https://relay.onorca.dev',
cellOrigin: `https://${hostname}.relay.onorca.dev`,
cellId,
mode: 'isolate'
})
}
for (const cellId of ['production-gce-c17', 'production-gce-c18', 'production-gce-c30']) {
const hostname = cellId.slice('production-gce-'.length)
assert.throws(() => parseProductionCapacityCellArguments([
'--director-origin', 'https://relay.onorca.dev',
'--cell-origin', `https://${hostname}.relay.onorca.dev`,
'--cell-id', cellId,
'--approved-cells', 'same-cap',
'--mode', 'isolate'
]), /not approved/)
}
assert.throws(() => parseProductionCapacityCellArguments([
'--director-origin', 'https://relay.onorca.dev',
'--cell-origin', 'https://c27.relay.onorca.dev',
'--cell-id', 'production-gce-c27',
'--approved-cells', 'every-cell',
'--mode', 'isolate'
]), /not a known allowlist/)
})
it('isolates only the selected cell without depending on its runtime', async () => {
@@ -170,4 +211,42 @@ describe('production Relay capacity cell admission', () => {
/irreversible/
)
})
it('retries a transient 503 on the cell drain endpoint', async () => {
let calls = 0
const result = await prepareProductionCapacityCell(
{ ...config, mode: 'drain' },
{
token: 'token',
wait: async () => {},
fetch: async (url) => {
assert.equal(new URL(url).pathname, '/v1/admin/drain')
calls += 1
if (calls === 1) return response({ error: 'warming up' }, 503)
return response({ v: 1, draining: true })
}
}
)
assert.equal(calls, 2)
assert.deepEqual(result, { changed: false, drained: true })
})
it('fails when both drain attempts return a transient 503', async () => {
let calls = 0
await assert.rejects(
prepareProductionCapacityCell(
{ ...config, mode: 'drain' },
{
token: 'token',
wait: async () => {},
fetch: async () => {
calls += 1
return response({ error: 'warming up' }, 503)
}
}
),
/returned 503/
)
assert.equal(calls, 2)
})
})
@@ -1,4 +1,5 @@
import { pathToFileURL } from 'node:url'
import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs'
const PRODUCTION_CELL = /^production-gce-c(?:7|8|9|10|13|14|15|16|19|20|21|22|23|24|25|26)$/
const DIRECTOR_ORIGIN = 'https://relay.onorca.dev'
@@ -35,7 +36,8 @@ export function parseRehomeTrustProbeArguments(argv, environment = process.env)
export async function probeRehomeTrust(config, dependencies = {}) {
const fetchImpl = dependencies.fetch ?? fetch
const response = await fetchImpl(
const response = await fetchAdminOnceMore(
fetchImpl,
`${config.directorOrigin}/v1/admin/regional-rehome-trust-probe`,
{
method: 'POST',
@@ -47,9 +49,9 @@ export async function probeRehomeTrust(config, dependencies = {}) {
v: 1,
sourceCellId: config.cellId,
sourceCellIncarnation: config.cellIncarnation
}),
signal: AbortSignal.timeout(30_000)
}
})
},
{ wait: dependencies.wait }
)
const body = await response.json().catch(() => ({}))
if (!response.ok) {
@@ -68,3 +68,46 @@ test('rejects partial or mismatched proof', async () => {
/incomplete/
)
})
const provenProbe = {
v: 1,
dedicatedIdentity: {
firstOutcome: 'host-not-connected',
secondOutcome: 'host-not-connected',
accepted: true,
idempotent: true
},
sharedRuntimeIdentityRejected: true,
proven: true
}
test('retries a transient 503 on the trust probe and proves on the second answer', async () => {
const config = parseRehomeTrustProbeArguments(argv, environment)
let calls = 0
const result = await probeRehomeTrust(config, {
wait: async () => {},
fetch: async () => {
calls += 1
if (calls === 1) return new Response('warming up', { status: 503 })
return Response.json(provenProbe)
}
})
assert.equal(calls, 2)
assert.equal(result.proven, true)
})
test('fails when both trust-probe attempts return a transient 503', async () => {
const config = parseRehomeTrustProbeArguments(argv, environment)
let calls = 0
await assert.rejects(
probeRehomeTrust(config, {
wait: async () => {},
fetch: async () => {
calls += 1
return new Response('warming up', { status: 503 })
}
}),
/returned 503/
)
assert.equal(calls, 2)
})
@@ -0,0 +1,43 @@
import assert from 'node:assert/strict'
import { readFileSync } from 'node:fs'
import { test } from 'node:test'
import { fileURLToPath } from 'node:url'
import { relayWorkflowUrl } from './relay-repository.mjs'
const WORKFLOWS = [
'deploy-relay-production-same-cap-job.yml',
'operate-relay-production-rehome-job.yml'
]
function workflow(name) {
return readFileSync(fileURLToPath(relayWorkflowUrl(name)), 'utf8')
}
// A single transient 5xx from a warming instance behind the global load balancer
// must not fail a canary, so no admin endpoint may be read by a bare curl.
test('no admin endpoint is reached by a curl without a bounded retry', () => {
for (const name of WORKFLOWS) {
for (const invocation of workflow(name).split(/\bcurl\b/).slice(1)) {
const flags = invocation.split('\n }')[0]
assert.match(flags, /--retry 3 --retry-delay 2 --retry-connrefused/, name)
assert.match(flags, /--max-time 30/, name)
// --retry-all-errors would also retry 401, 403, and 409, which are final.
assert.doesNotMatch(flags, /--retry-all-errors/, name)
}
}
})
test('every retried admin request captures only the final attempt body', () => {
const job = workflow('deploy-relay-production-same-cap-job.yml')
// --fail-with-body writes every failed attempt to stdout, so a retried
// request must land in a file curl truncates per attempt.
assert.match(job, /--output "\$\{out\}"/)
assert.equal(job.split('admin_post() {').length - 1, 2)
for (const call of [
/CURRENT_RUNTIME="\$\(admin_post current-runtime/,
/CURRENT_DIRECTOR_STATUS="\$\(admin_post current-cell-status/,
/TARGET_RUNTIME="\$\(admin_post target-runtime/,
/TARGET_DIRECTOR_STATUS="\$\(admin_post target-cell-status/
]) assert.match(job, call)
assert.doesNotMatch(job, /\$\(curl /)
})
@@ -0,0 +1,29 @@
// A single transient 5xx (load-balancer warm-up behind a fresh instance) must not fail a
// deploy step. 4xx is never retried: auth and generation-mismatch answers are final.
const TRANSIENT_STATUSES = [500, 502, 503, 504]
const RETRY_DELAY_MS = 2_000
const REQUEST_TIMEOUT_MS = 30_000
export function isTransientAdminStatus(status) {
return TRANSIENT_STATUSES.includes(status)
}
// Each attempt gets its own timeout budget, so a reused signal cannot abort the retry.
export async function fetchAdminOnceMore(fetchImpl, url, init, overrides = {}) {
const wait = overrides.wait ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms)))
const timeoutMs = overrides.timeoutMs ?? REQUEST_TIMEOUT_MS
const retryDelayMs = overrides.retryDelayMs ?? RETRY_DELAY_MS
const attempt = async () =>
await fetchImpl(url, { ...init, signal: AbortSignal.timeout(timeoutMs) })
let response
try {
response = await attempt()
} catch {
await wait(retryDelayMs)
return await attempt()
}
if (!isTransientAdminStatus(response.status)) return response
await response.arrayBuffer?.().catch(() => undefined)
await wait(retryDelayMs)
return await attempt()
}
@@ -0,0 +1,130 @@
import assert from 'node:assert/strict'
import { test } from 'node:test'
import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs'
const url = 'https://relay.onorca.dev/v1/admin/cell-status'
const init = { method: 'POST', body: '{"v":1}' }
function recordingWait(waits) {
return async (ms) => { waits.push(ms) }
}
test('a single transient 5xx is retried and the second answer is returned', async () => {
const waits = []
const statuses = [503, 200]
let calls = 0
const response = await fetchAdminOnceMore(
async () => {
calls += 1
const status = statuses.shift()
return new Response(JSON.stringify({ ok: status === 200 }), { status })
},
url,
init,
{ wait: recordingWait(waits) }
)
assert.equal(calls, 2)
assert.equal(response.status, 200)
assert.deepEqual(waits, [2_000])
assert.deepEqual(await response.json(), { ok: true })
})
test('a connection failure is retried and the second answer is returned', async () => {
const waits = []
let calls = 0
const response = await fetchAdminOnceMore(
async () => {
calls += 1
if (calls === 1) throw new TypeError('fetch failed')
return Response.json({ ok: true })
},
url,
init,
{ wait: recordingWait(waits) }
)
assert.equal(calls, 2)
assert.equal(response.status, 200)
assert.deepEqual(waits, [2_000])
})
test('two transient failures surface the second answer without a third attempt', async () => {
let calls = 0
const response = await fetchAdminOnceMore(
async () => {
calls += 1
return new Response('down', { status: 503 })
},
url,
init,
{ wait: async () => {} }
)
assert.equal(calls, 2)
assert.equal(response.status, 503)
})
test('two connection failures rethrow the second error', async () => {
let calls = 0
await assert.rejects(
fetchAdminOnceMore(
async () => {
calls += 1
throw new TypeError(`fetch failed ${calls}`)
},
url,
init,
{ wait: async () => {} }
),
/fetch failed 2/
)
assert.equal(calls, 2)
})
test('4xx is final: auth and generation-mismatch answers are never retried', async () => {
for (const status of [400, 401, 403, 404, 409, 429]) {
let calls = 0
const response = await fetchAdminOnceMore(
async () => {
calls += 1
return new Response('no', { status })
},
url,
init,
{ wait: async () => { throw new Error('must not wait') } }
)
assert.equal(calls, 1, `status ${status} must not be retried`)
assert.equal(response.status, status)
}
})
test('each attempt carries its own unexpired timeout signal', async () => {
const signals = []
await fetchAdminOnceMore(
async (_url, attemptInit) => {
signals.push(attemptInit.signal)
return new Response('down', { status: 502 })
},
url,
init,
{ wait: async () => {}, timeoutMs: 30_000 }
)
assert.equal(signals.length, 2)
assert.notEqual(signals[0], signals[1])
assert.equal(signals[1].aborted, false)
})
test('the caller init is forwarded unchanged apart from the signal', async () => {
let seen
await fetchAdminOnceMore(
async (seenUrl, attemptInit) => {
seen = { seenUrl, attemptInit }
return Response.json({})
},
url,
{ method: 'POST', headers: { authorization: 'Bearer t' }, body: '{"v":1}' },
{ wait: async () => {} }
)
assert.equal(seen.seenUrl, url)
assert.equal(seen.attemptInit.method, 'POST')
assert.deepEqual(seen.attemptInit.headers, { authorization: 'Bearer t' })
assert.equal(seen.attemptInit.body, '{"v":1}')
})
@@ -0,0 +1,94 @@
import { spawnSync } from 'node:child_process'
import { fileURLToPath } from 'node:url'
import {
RELAY_REPOSITORY_ROOT,
relayTreePath,
relayWorkflowPath
} from './relay-repository.mjs'
const SHA = /^[a-f0-9]{40}$/
// Every file that decides how relay evidence is produced, sealed, verified, and then spent against
// production; identical content across two commits is what makes the older commit's verdict binding.
export const TRUSTED_EVIDENCE_CODE_PATHS = [
// Produces and seals the 15-minute dry-run evidence.
relayWorkflowPath('monitor-relay-production.yml'),
relayWorkflowPath('monitor-relay-production-job.yml'),
// Download it, verify its authority, and mutate production on it.
relayWorkflowPath('deploy-relay-production-same-cap.yml'),
relayWorkflowPath('deploy-relay-production-same-cap-job.yml'),
relayWorkflowPath('operate-relay-production-rehome.yml'),
relayWorkflowPath('operate-relay-production-rehome-job.yml'),
// Sealing, verification, the wave/canary authority, and the path constants below.
relayTreePath('dev/scripts/relay-evidence-code-provenance.mjs'),
relayTreePath('dev/scripts/relay-monitor-evidence.mjs'),
relayTreePath('dev/scripts/relay-production-same-cap-wave.mjs'),
relayTreePath('dev/scripts/relay-repository.mjs'),
// Every other script those jobs run against live production.
relayTreePath('dev/scripts/infra.mjs'),
relayTreePath('dev/scripts/operate-relay-regional-rehome.mjs'),
relayTreePath('dev/scripts/prepare-relay-production-capacity-canary.mjs'),
relayTreePath('dev/scripts/probe-relay-rehome-trust.mjs'),
relayTreePath('dev/scripts/validate-relay-capacity-plan.mjs'),
relayTreePath('dev/scripts/verify-relay-capacity-transition.mjs'),
// The monitor itself and the live preflight recheck, plus anything that changes their behaviour.
relayTreePath('apps/relay-ops'),
relayTreePath('package.json'),
relayTreePath('pnpm-lock.yaml'),
relayTreePath('pnpm-workspace.yaml'),
// The Cloud SQL rollout lease every mutation job takes and releases.
'.github/actions/cloud-sql-rollout-lease'
]
function git(root, args) {
const result = spawnSync('git', ['-C', root, ...args], { encoding: 'utf8' })
if (result.error) throw new Error('relay evidence provenance cannot run git')
return result
}
/**
* Accepts evidence sealed at a different commit only when the current commit descends from it and
* every trusted path is byte-identical, so the verdict provably came from this exact code. Anything
* git cannot answer (no checkout, unknown commit, shallow clone) fails closed.
*/
export function requireSameEvidenceCode({
sealedSha,
currentSha,
label,
repositoryRoot = fileURLToPath(RELAY_REPOSITORY_ROOT)
}) {
if (!SHA.test(sealedSha ?? '') || !SHA.test(currentSha ?? '')) {
throw new Error(`${label} commit is invalid`)
}
if (sealedSha === currentSha) return
if (git(repositoryRoot, ['rev-parse', '--git-dir']).status !== 0) {
throw new Error(`${label} commit cannot be compared without a git checkout`)
}
for (const sha of [sealedSha, currentSha]) {
if (git(repositoryRoot, ['rev-parse', '--verify', '--quiet', `${sha}^{commit}`]).status !== 0) {
throw new Error(
`${label} commit ${sha} is unknown to this checkout; check out with fetch-depth: 0`
)
}
}
const ancestry = git(repositoryRoot, ['merge-base', '--is-ancestor', sealedSha, currentSha])
if (ancestry.status === 1) {
throw new Error(`${label} commit ${sealedSha} is not an ancestor of ${currentSha}`)
}
if (ancestry.status !== 0) {
throw new Error(`${label} commit ancestry could not be determined`)
}
const diff = git(repositoryRoot, [
'diff',
'--name-only',
sealedSha,
currentSha,
'--',
...TRUSTED_EVIDENCE_CODE_PATHS
])
if (diff.status !== 0) throw new Error(`${label} commit comparison failed`)
const changed = diff.stdout.split('\n').filter(Boolean)
if (changed.length > 0) {
throw new Error(`${label} code changed after it was sealed: ${changed.join(',')}`)
}
}
+18 -5
View File
@@ -2,6 +2,7 @@ import { createHash } from 'node:crypto'
import { chmod, readFile, readdir, stat, writeFile } from 'node:fs/promises'
import { basename, join, resolve } from 'node:path'
import { pathToFileURL } from 'node:url'
import { requireSameEvidenceCode } from './relay-evidence-code-provenance.mjs'
const SAFE_ID = /^[A-Za-z0-9][A-Za-z0-9._-]{1,127}$/
const SHA = /^[a-f0-9]{40}$/
@@ -102,7 +103,7 @@ export async function createEvidenceManifest(argv) {
return manifest
}
async function readAndVerifyManifest(directory, expected) {
async function readAndVerifyManifest(directory, expected, sameCodeCommit) {
const manifest = JSON.parse(
await readFile(join(directory, 'evidence-manifest.json'), 'utf8')
)
@@ -111,11 +112,23 @@ async function readAndVerifyManifest(directory, expected) {
manifest.incidentId !== expected.incidentId ||
manifest.runId !== expected.runId ||
manifest.runAttempt !== expected.runAttempt ||
manifest.commitSha !== expected.commitSha ||
manifest.mode !== expected.mode
!SHA.test(manifest.commitSha ?? '') ||
manifest.mode !== expected.mode ||
(!sameCodeCommit && manifest.commitSha !== expected.commitSha)
) {
throw new Error('relay monitor evidence provenance does not match')
}
// Unrelated merges land on main every few minutes, so the deployer resolves a newer commit than
// the monitor it must trust; identical monitor and mutation code is the property the SHA stood in
// for. Restore and mutation keep the exact-SHA bind: both run at the commit that sealed them.
if (sameCodeCommit) {
requireSameEvidenceCode({
sealedSha: manifest.commitSha,
currentSha: expected.commitSha,
label: 'relay monitor evidence',
...sameCodeCommit
})
}
const names = Object.keys(manifest.files ?? {})
if (!names.includes(`${expected.incidentId}.state.json`)) {
throw new Error('relay monitor evidence has no durable state')
@@ -209,12 +222,12 @@ function validCompletedDryRunState(state, expected, nowMs, maxAgeMs) {
)
}
export async function verifyDryRunAuthority(argv, now = Date.now) {
export async function verifyDryRunAuthority(argv, now = Date.now, repositoryRoot) {
const values = argumentsByName(argv)
const directory = resolve(values.directory ?? '')
const expected = provenance(values)
if (expected.mode !== 'dry-run') throw new Error('relay mutation requires dry-run evidence')
const manifest = await readAndVerifyManifest(directory, expected)
const manifest = await readAndVerifyManifest(directory, expected, { repositoryRoot })
const state = JSON.parse(
await readFile(join(directory, `${expected.incidentId}.state.json`), 'utf8')
)
@@ -1,9 +1,15 @@
import assert from 'node:assert/strict'
import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises'
import { execFileSync } from 'node:child_process'
import { mkdir, mkdtemp, readFile, rm, stat, writeFile } from 'node:fs/promises'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { dirname, join } from 'node:path'
import test from 'node:test'
import { relayWorkflowPath, relayWorkflowUrl } from './relay-repository.mjs'
import { TRUSTED_EVIDENCE_CODE_PATHS } from './relay-evidence-code-provenance.mjs'
import {
RELAY_REPOSITORY_ROOT,
relayWorkflowPath,
relayWorkflowUrl
} from './relay-repository.mjs'
import {
createEvidenceManifest,
verifyDryRunAuthority,
@@ -12,7 +18,7 @@ import {
} from './relay-monitor-evidence.mjs'
const now = Date.parse('2026-07-28T12:00:00.000Z')
const provenance = [
const provenanceFor = (commitSha) => [
'--incident-id',
'relay-123',
'--run-id',
@@ -20,10 +26,11 @@ const provenance = [
'--run-attempt',
'1',
'--commit-sha',
'a'.repeat(40),
commitSha,
'--mode',
'dry-run'
]
const provenance = provenanceFor('a'.repeat(40))
const selector = {
generation: 2,
membership: {
@@ -513,3 +520,157 @@ test('monitor uses a reusable job so exact job_workflow_ref is present', async (
assert.match(job, /workflow_call:/)
assert.match(job, /environment: production/)
})
function gitIn(root, ...args) {
return execFileSync('git', ['-C', root, ...args], { encoding: 'utf8' }).trim()
}
// A real repository shaped like main under unrelated merge traffic: one sealed commit, a
// descendant that only touched untrusted files, a descendant that touched the monitor, and a
// sibling that never descended from the seal.
async function trustedCodeRepository() {
const root = await mkdtemp(join(tmpdir(), 'relay-evidence-repository-'))
gitIn(root, 'init', '--quiet')
gitIn(root, 'config', 'user.email', 'relay@example.test')
gitIn(root, 'config', 'user.name', 'Relay Evidence Test')
gitIn(root, 'config', 'commit.gpgsign', 'false')
const commit = async (path, body, message) => {
await mkdir(dirname(join(root, path)), { recursive: true })
await writeFile(join(root, path), body)
gitIn(root, 'add', '--all')
gitIn(root, 'commit', '--quiet', '--no-verify', '--message', message)
return gitIn(root, 'rev-parse', 'HEAD')
}
const base = await commit(
'cloud/apps/relay-ops/src/incident-monitor.ts',
'export const v = 1\n',
'monitor'
)
const sealed = await commit('README.md', 'base\n', 'base')
const sameCode = await commit('README.md', 'an unrelated merge\n', 'unrelated')
const changedCode = await commit(
'cloud/apps/relay-ops/src/incident-monitor.ts',
'export const v = 2\n',
'monitor change'
)
// Branches before the seal, so the seal is not in its history even though its code matches.
gitIn(root, 'checkout', '--quiet', '--detach', base)
const sibling = await commit('README.md', 'a divergent line\n', 'divergent')
return { root, sealed, sameCode, changedCode, sibling }
}
const authorityAt = (directory, commitSha, repositoryRoot) => verifyDryRunAuthority(
[
'--directory',
directory,
...provenanceFor(commitSha),
'--required-migration-policy',
'strict'
],
() => now,
repositoryRoot
)
test('accepts dry-run evidence sealed by identical code at an ancestor commit', async () => {
const repository = await trustedCodeRepository()
const directory = await evidenceDirectory()
try {
await createEvidenceManifest([
'--directory',
directory,
...provenanceFor(repository.sealed)
])
// An exact match never consults git: a root with no checkout at all still verifies.
await assert.doesNotReject(authorityAt(directory, repository.sealed, directory))
await assert.doesNotReject(authorityAt(directory, repository.sameCode, repository.root))
} finally {
await rm(repository.root, { recursive: true, force: true })
await rm(directory, { recursive: true, force: true })
}
})
test('rejects dry-run evidence whose monitor code or lineage differs', async () => {
const repository = await trustedCodeRepository()
const directory = await evidenceDirectory()
try {
await createEvidenceManifest([
'--directory',
directory,
...provenanceFor(repository.sealed)
])
await assert.rejects(
authorityAt(directory, repository.changedCode, repository.root),
/code changed after it was sealed: cloud\/apps\/relay-ops\/src\/incident-monitor\.ts/
)
await assert.rejects(
authorityAt(directory, repository.sibling, repository.root),
/is not an ancestor of/
)
// Fails closed: a shallow clone that never fetched the sealed commit proves nothing.
await assert.rejects(
authorityAt(directory, 'f'.repeat(40), repository.root),
/unknown to this checkout/
)
// Fails closed: no checkout to compare against.
await assert.rejects(
authorityAt(directory, repository.sameCode, directory),
/cannot be compared without a git checkout/
)
} finally {
await rm(repository.root, { recursive: true, force: true })
await rm(directory, { recursive: true, force: true })
}
})
test('keeps restore and mutation bound to the exact sealing commit', async () => {
const repository = await trustedCodeRepository()
const directory = await evidenceDirectory()
try {
await createEvidenceManifest([
'--directory',
directory,
...provenanceFor(repository.sealed)
])
await assert.rejects(
verifyRestoredEvidence([
'--directory',
directory,
...provenanceFor(repository.sameCode)
]),
/provenance does not match/
)
await assert.rejects(
verifyMutationEvidence(
[
'--directory',
directory,
...provenanceFor(repository.sameCode),
'--mutation-mode',
'execute',
'--source-cell-id',
'c1',
'--director-origin',
'https://relay.example'
],
{ ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' },
async () => Response.json({ selector }),
() => now
),
/provenance does not match/
)
} finally {
await rm(repository.root, { recursive: true, force: true })
await rm(directory, { recursive: true, force: true })
}
})
// A trusted path that no longer exists silently stops being compared, so the same-code rule would
// pass over code it was written to pin.
test('every trusted provenance path exists in this checkout', async () => {
for (const path of TRUSTED_EVIDENCE_CODE_PATHS) {
await assert.doesNotReject(
stat(new URL(path, RELAY_REPOSITORY_ROOT)),
`${path} is missing`
)
}
})
@@ -1,5 +1,6 @@
import { readFileSync } from 'node:fs'
import { pathToFileURL } from 'node:url'
import { requireSameEvidenceCode } from './relay-evidence-code-provenance.mjs'
export const SAME_CAP_CELLS = [
'production-gce-c7', 'production-gce-c8', 'production-gce-c9', 'production-gce-c10',
@@ -85,10 +86,10 @@ export function canaryAuthority(input) {
}
}
export function verifyCanaryAuthority(authority, expected) {
export function verifyCanaryAuthority(authority, expected, repositoryRoot) {
if (
authority?.v !== 1 ||
authority.commitSha !== expected.commitSha ||
!/^[0-9a-f]{40}$/.test(authority.commitSha ?? '') ||
authority.runId !== expected.runId ||
authority.targetDigest !== expected.targetDigest ||
authority.rollbackDigest !== expected.rollbackDigest ||
@@ -96,6 +97,14 @@ export function verifyCanaryAuthority(authority, expected) {
authority.rehomeGeneration !== Number(expected.rehomeGeneration) ||
!SAME_CAP_CELLS.includes(authority.cellId)
) throw new Error('canary authority does not match this batch')
// The batch dispatch resolves main after the canary sealed, so bind to the same code, not the
// same SHA; every field above still pins this batch to that exact canary.
requireSameEvidenceCode({
sealedSha: authority.commitSha,
currentSha: expected.commitSha,
label: 'relay same-cap canary authority',
repositoryRoot
})
return authority
}
@@ -1,4 +1,8 @@
import assert from 'node:assert/strict'
import { execFileSync } from 'node:child_process'
import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'
import { tmpdir } from 'node:os'
import { dirname, join } from 'node:path'
import { test } from 'node:test'
import {
canaryAuthority,
@@ -104,3 +108,66 @@ test('seals and verifies canary authority for later batches', () => {
rehomeGeneration: '4'
}), /does not match/)
})
function gitIn(root, ...args) {
return execFileSync('git', ['-C', root, ...args], { encoding: 'utf8' }).trim()
}
async function canaryRepository() {
const root = await mkdtemp(join(tmpdir(), 'relay-same-cap-canary-'))
gitIn(root, 'init', '--quiet')
gitIn(root, 'config', 'user.email', 'relay@example.test')
gitIn(root, 'config', 'user.name', 'Relay Wave Test')
gitIn(root, 'config', 'commit.gpgsign', 'false')
const commit = async (path, body, message) => {
await mkdir(dirname(join(root, path)), { recursive: true })
await writeFile(join(root, path), body)
gitIn(root, 'add', '--all')
gitIn(root, 'commit', '--quiet', '--no-verify', '--message', message)
return gitIn(root, 'rev-parse', 'HEAD')
}
const sealed = await commit(
'cloud/dev/scripts/relay-production-same-cap-wave.mjs',
'export const v = 1\n',
'wave'
)
const sameCode = await commit('README.md', 'an unrelated merge\n', 'unrelated')
const changedCode = await commit(
'cloud/dev/scripts/relay-production-same-cap-wave.mjs',
'export const v = 2\n',
'wave change'
)
return { root, sealed, sameCode, changedCode }
}
test('a batch trusts a canary sealed by identical code at an ancestor commit', async () => {
const repository = await canaryRepository()
try {
const authority = canaryAuthority({
cellIds: 'production-gce-c7',
targetDigest,
rollbackDigest,
confirmation: `ROLL_RELAY_SAME_CAP ${targetDigest} production-gce-c7`,
commitSha: repository.sealed,
runId: '42',
selectorGeneration: '11',
rehomeGeneration: '4'
})
const verifyAt = (commitSha, repositoryRoot) => verifyCanaryAuthority(authority, {
commitSha,
runId: '42',
targetDigest,
rollbackDigest,
selectorGeneration: '13',
rehomeGeneration: '4'
}, repositoryRoot)
assert.equal(verifyAt(repository.sameCode, repository.root).cellId, 'production-gce-c7')
assert.throws(
() => verifyAt(repository.changedCode, repository.root),
/code changed after it was sealed/
)
assert.throws(() => verifyAt('f'.repeat(40), repository.root), /unknown to this checkout/)
} finally {
await rm(repository.root, { recursive: true, force: true })
}
})
@@ -73,7 +73,10 @@ test('same-cap wrapper is reusable, canary-bound, and sequential', () => {
job,
/--rollback-image "\$\{DESIRED_IMAGE\}" \\\n {16}--rehome-director-service-account "\$\{DIRECTOR_RUNTIME_SERVICE_ACCOUNT\}"/
)
assert.match(job, /host-drain \\\n {14}\| jq -e '\.changes == 2' >\/dev\/null/)
assert.match(
job,
/host-drain \\\n {16}--regional-rehome-protocol "\$\{DESIRED_REHOME_PROTOCOL\}" \\\n {14}\| jq -e '\.changes == 2' >\/dev\/null/
)
assert.match(job, /resume requires the isolated migration-only cell/)
assert.match(job, /test "\$\{TARGET_INCARNATION\}" = "\$\{SOURCE_INCARNATION\}"/)
assert.match(job, /\(.regionalRehomeProtocol \/\/ 0\) == \$protocol/)
@@ -92,7 +95,11 @@ test('same-cap wrapper is reusable, canary-bound, and sequential', () => {
// age checks must scale by wave or cell_2+ can never pass; the bound's
// per-wave step is the cell job timeout, so the two must move together.
assert.match(job, /--required-migration-policy strict \\\n --wave-index "\$\{WAVE_INDEX\}"/)
assert.match(job, /dry-run\.state\.json" \\\n --wave-index "\$\{WAVE_INDEX\}" "\$\{RETRY_ARGS\[@\]\}"/)
// Wave 0 must retry freshness-only failures too: one Cloud Monitoring publish
// lag at the sample instant is not health evidence, and single-shot wave 0
// failed a whole batch on a series that was fresh again a minute later.
assert.match(job, /dry-run\.state\.json" \\\n --wave-index "\$\{WAVE_INDEX\}" --retry-freshness/)
assert.doesNotMatch(job, /RETRY_ARGS/)
assert.match(job, /timeout-minutes: 75/)
// Both age gates step by the cell job timeout above; the constant is
// duplicated across the two languages, so pin each copy to it.
+15
View File
@@ -1,4 +1,6 @@
import { readFileSync } from 'node:fs'
import { relative } from 'node:path'
import { fileURLToPath } from 'node:url'
// Single place naming the repository the Relay workflows live in and where their files sit. The
// public-repo copy moves this tree under cloud/, prefixes every workflow filename, and changes the
@@ -11,6 +13,19 @@ export const RELAY_WORKFLOW_FILE_PREFIX = 'cloud-'
// this tree moves under cloud/, so the depth changes at the copy even though the layout does not.
export const RELAY_WORKFLOW_DIRECTORY = new URL('../../../.github/workflows/', import.meta.url)
// Repository root, derived from the one directory above that already tracks the copy's depth.
export const RELAY_REPOSITORY_ROOT = new URL('../../', RELAY_WORKFLOW_DIRECTORY)
// Repository-relative path for a file in this tree. The prefix is 'cloud/' here and empty where
// the tree is the repository root, so callers naming git paths never restate the layout.
export function relayTreePath(suffix) {
const prefix = relative(
fileURLToPath(RELAY_REPOSITORY_ROOT),
fileURLToPath(new URL('../../', import.meta.url))
).split(/[\\/]/).filter(Boolean)
return [...prefix, suffix].join('/')
}
export function relayWorkflowFile(name) {
return `${RELAY_WORKFLOW_FILE_PREFIX}${name}`
}
@@ -0,0 +1,217 @@
import assert from 'node:assert/strict'
import { spawnSync } from 'node:child_process'
import { readFileSync } from 'node:fs'
import { describe, it } from 'node:test'
import { parseProductionCapacityCellArguments } from './prepare-relay-production-capacity-canary.mjs'
import { SAME_CAP_CELLS } from './relay-production-same-cap-wave.mjs'
import { readRelayWorkflow } from './relay-repository.mjs'
import { validateCapacityPlan } from './validate-relay-capacity-plan.mjs'
const workflow = readRelayWorkflow('deploy-relay-production-same-cap-job.yml')
const capacityWorkflow = readRelayWorkflow('deploy-relay-production-capacity-job.yml')
const production = readFileSync(
new URL('../../infra/terraform/environments/production.tfvars', import.meta.url),
'utf8'
)
const REHOME_SOURCE_CELLS = rehomeSourceCells()
const DIRECTOR_IDENTITY = 'relay-director@onorca-cloud.iam.gserviceaccount.com'
const AUDIENCE = 'https://relay.onorca.dev/v1/admin/host-drain'
const ROLLBACK_IMAGE = `us-central1-docker.pkg.dev/p/orca-cloud/relay@sha256:${'d'.repeat(64)}`
const TARGET_IMAGE = `us-central1-docker.pkg.dev/p/orca-cloud/relay@sha256:${'e'.repeat(64)}`
// The startup template emits rehome trust only for cells in this list, so it is what decides
// whether a cell's plan may carry those lines at all.
function rehomeSourceCells() {
const start = production.indexOf('relay_region_rehome_source_cell_ids = [')
assert.notEqual(start, -1, 'production.tfvars has no rehome source cell list')
const end = production.indexOf(']', start)
assert.notEqual(end, -1, 'the rehome source cell list is unterminated')
return new Set(
[...production.slice(start, end).matchAll(/"([^"]+)"/g)].map(([, cell]) => cell)
)
}
function startupScript({ cap, image, trusted }) {
return [
` printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '${cap}'`,
` printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '60'`,
...(trusted ? [
` printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '${DIRECTOR_IDENTITY}'`,
` printf 'ORCA_RELAY_REHOME_AUDIENCE=%s\\n' '${AUDIENCE}'`
] : []),
`printf 'ORCA_RELAY_IMAGE_DIGEST=%s\\n' '${image.split('@')[1]}'`,
`docker pull '${image}'`,
'docker run --detach \\',
' --name orca-relay \\',
` '${image}'`
].join('\n')
}
// The exact shape the apply step's plan has: template replaced, MIG rebound to it.
function rollPlan({ cellId, cap, protocol }) {
return {
configuration: {
root_module: {
resources: [{
address: 'google_compute_instance_group_manager.relay_gce_cell',
expressions: {
version: [{
instance_template: {
references: [
'google_compute_instance_template.relay_gce_cell',
'each.key'
]
},
name: { constant_value: 'primary' }
}]
}
}]
}
},
resource_changes: [
{
address: `google_compute_instance_template.relay_gce_cell[${JSON.stringify(cellId)}]`,
change: {
actions: ['create', 'delete'],
before: {
metadata_startup_script: startupScript({
cap,
image: ROLLBACK_IMAGE,
trusted: protocol === 1
})
},
after: {
metadata_startup_script: startupScript({
cap,
image: TARGET_IMAGE,
trusted: protocol === 1
}),
self_link: null
},
after_unknown: { self_link: true }
}
},
{
address: `google_compute_instance_group_manager.relay_gce_cell[${JSON.stringify(cellId)}]`,
change: {
actions: ['update'],
before: { target_size: 1, version: [{ instance_template: 'old' }] },
after: { target_size: 1, version: [{ instance_template: null }] },
after_unknown: { version: [{ instance_template: true }] }
}
}
]
}
}
function hostname(cellId) {
return cellId.slice('production-gce-'.length)
}
// The job resolves cap and region from the cell id before any admin call; run that block alone.
function resolveCellShape(cellId) {
const start = workflow.indexOf(' TARGET_HOSTNAME="${TARGET_CELL_ID#production-gce-}"')
assert.notEqual(start, -1, 'the same-cap cell shape block is missing')
const end = workflow.indexOf('\n esac\n', start)
assert.notEqual(end, -1, 'the same-cap cell shape block has no esac')
const script = workflow.slice(start, end + '\n esac'.length).replace(/^ {10}/gm, '')
return spawnSync('bash', [
'-euo',
'pipefail',
'-c',
`${script}\necho "\${EXPECTED_REGION} \${EXPECTED_HARD_CAP}"`
], { env: { ...process.env, TARGET_CELL_ID: cellId }, encoding: 'utf8' })
}
describe('same-cap roll scripts accept every same-cap cell', () => {
it('parses every wave cell through the same-cap canary allowlist', () => {
for (const cellId of SAME_CAP_CELLS) {
for (const mode of ['isolate', 'drain', 'activate']) {
assert.deepEqual(parseProductionCapacityCellArguments([
'--director-origin', 'https://relay.onorca.dev',
'--cell-origin', `https://${hostname(cellId)}.relay.onorca.dev`,
'--cell-id', cellId,
'--approved-cells', 'same-cap',
'--mode', mode
]), {
directorOrigin: 'https://relay.onorca.dev',
cellOrigin: `https://${hostname(cellId)}.relay.onorca.dev`,
cellId,
mode
})
}
}
})
it('resolves a cap and region for every wave cell and refuses anything else', () => {
for (const cellId of SAME_CAP_CELLS) {
const resolved = resolveCellShape(cellId)
assert.equal(resolved.status, 0, `${cellId}: ${resolved.stderr}`)
assert.match(resolved.stdout.trim(), /^(us-central1 1000|asia-east2 3000)$/)
}
assert.equal(resolveCellShape('production-gce-c17').status, 1)
assert.equal(resolveCellShape('production-gce-c30').status, 1)
})
it('passes the same-cap allowlist on every canary invocation the job runs', () => {
const invocations = workflow.split('prepare-relay-production-capacity-canary.mjs').slice(1)
assert.equal(invocations.length, 4)
for (const invocation of invocations) {
const lines = invocation.split('\n')
const end = lines.findIndex((line) => !line.endsWith('\\'))
const call = lines.slice(0, end + 1).join(' ')
assert.match(call, /--approved-cells same-cap/)
assert.match(call, /--mode (isolate|drain|activate)/)
}
})
it('passes this cell\'s rehome protocol on every plan validation the job runs', () => {
const invocations = workflow.split('validate-relay-capacity-plan.mjs').slice(1)
assert.equal(invocations.length, 2)
for (const invocation of invocations) {
const lines = invocation.split('\n')
const end = lines.findIndex((line) => !line.trimEnd().endsWith('\\'))
const call = lines.slice(0, end + 1).join(' ')
assert.match(call, /--mode same-cap-cell/)
assert.match(call, /--regional-rehome-protocol "\$\{DESIRED_REHOME_PROTOCOL\}"/)
}
})
it('validates a correct plan for every wave cell at that cell\'s rehome protocol', () => {
for (const cellId of SAME_CAP_CELLS) {
const [region, cap] = resolveCellShape(cellId).stdout.trim().split(' ')
const protocol = REHOME_SOURCE_CELLS.has(cellId) ? 1 : 0
assert.equal(protocol, region === 'us-central1' ? 1 : 0, cellId)
const config = {
mode: 'same-cap-cell',
cellId,
hardCap: Number(cap),
unobservedBound: 60,
image: TARGET_IMAGE,
rollbackImage: ROLLBACK_IMAGE,
rehomeDirectorServiceAccount: DIRECTOR_IDENTITY,
rehomeAudience: AUDIENCE,
regionalRehomeProtocol: String(protocol)
}
const plan = rollPlan({ cellId, cap, protocol })
assert.deepEqual(
validateCapacityPlan(plan, config),
{ mode: 'same-cap-cell', changes: 2 },
cellId
)
// The other protocol must reject the same plan, or the flag decides nothing.
assert.throws(
() => validateCapacityPlan(plan, {
...config,
regionalRehomeProtocol: String(1 - protocol)
}),
/reviewed image and capacity/,
cellId
)
}
})
it('leaves the US-only capacity job on the default allowlist', () => {
assert.doesNotMatch(capacityWorkflow, /--approved-cells/)
})
})
@@ -4,7 +4,18 @@ import { pathToFileURL } from 'node:url'
const SERVICE_ACCOUNT_EMAIL =
/^[a-z][a-z0-9-]{4,28}[a-z0-9]@[a-z0-9-]+\.iam\.gserviceaccount\.com$/
function parseArguments(argv) {
const REHOME_CONFIG =
/^ printf 'ORCA_RELAY_REHOME_(?:DIRECTOR_SERVICE_ACCOUNT|AUDIENCE)=%s\\n' '[^'\n]+'$/
// Only cells listed as regional rehome sources get rehome trust lines in their startup script.
function rehomeProtocol({ regionalRehomeProtocol }) {
if (![0, 1, '0', '1'].includes(regionalRehomeProtocol)) {
throw new Error('same-cap Terraform plan has an invalid regional rehome protocol')
}
return Number(regionalRehomeProtocol)
}
export function parseCapacityPlanArguments(argv) {
const values = {}
for (let index = 0; index < argv.length; index += 2) {
const key = argv[index]
@@ -31,8 +42,12 @@ function parseArguments(argv) {
values.mode === 'same-cap-cell' &&
(!values['rollback-image'] ||
!values['rehome-director-service-account'] ||
!values['rehome-audience'])
!values['rehome-audience'] ||
!['0', '1'].includes(values['regional-rehome-protocol']))
) throw new Error('same-cap validation requires rollback image and rehome trust config')
if (values.mode !== 'same-cap-cell' && values['regional-rehome-protocol'] !== undefined) {
throw new Error('--regional-rehome-protocol applies only to same-cap-cell validation')
}
if (values.mode === 'same-cap-image' && !values['rollback-image']) {
throw new Error('same-cap image validation requires a rollback image')
}
@@ -51,7 +66,8 @@ function parseArguments(argv) {
capacityServiceAccount: values['capacity-service-account'],
rollbackImage: values['rollback-image'],
rehomeDirectorServiceAccount: values['rehome-director-service-account'],
rehomeAudience: values['rehome-audience']
rehomeAudience: values['rehome-audience'],
regionalRehomeProtocol: values['regional-rehome-protocol']
}
}
@@ -175,15 +191,13 @@ function normalizedStartupScript(
/^ printf 'ORCA_RELAY_CELL_CONNECTION_(?:HARD_CAP|UNOBSERVED_BOUND)=%s\\n' '[0-9]+'$/
const capacityIdentity =
/^ printf 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT=%s\\n' '[a-z][a-z0-9-]{4,28}[a-z0-9]@[a-z0-9-]+\.iam\.gserviceaccount\.com'$/
const rehomeConfig =
/^ printf 'ORCA_RELAY_REHOME_(?:DIRECTOR_SERVICE_ACCOUNT|AUDIENCE)=%s\\n' '[^'\n]+'$/
return script
.split('\n')
.filter(
(line) =>
(preserveCapacity || !capacityAssignment.test(line)) &&
(!stripCapacityIdentity || !capacityIdentity.test(line)) &&
(!stripRehomeConfig || !rehomeConfig.test(line))
(!stripRehomeConfig || !REHOME_CONFIG.test(line))
)
.join('\n')
.replaceAll(image, '<relay-image>')
@@ -213,7 +227,8 @@ function requireDesiredStartupScript(script, config) {
` printf 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT=%s\\n' '${config.capacityServiceAccount}'`
])
}
if (config.mode === 'same-cap-cell') {
const rehomeTrusted = config.mode === 'same-cap-cell' && rehomeProtocol(config) === 1
if (rehomeTrusted) {
expected.push(
[
/^ printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '[^'\n]+'$/,
@@ -225,9 +240,15 @@ function requireDesiredStartupScript(script, config) {
]
)
}
// A protocol-0 cell is not a rehome source, so gaining any rehome trust line is real drift.
const unexpectedRehome =
config.mode === 'same-cap-cell' &&
!rehomeTrusted &&
lines.some((line) => REHOME_CONFIG.test(line))
if (
typeof script !== 'string' ||
relayImage(script) !== config.image ||
unexpectedRehome ||
expected.some(([pattern, line]) => !hasExactSingleAssignment(lines, pattern, line))
) {
throw new Error('cell plan does not contain the reviewed image and capacity')
@@ -450,6 +471,9 @@ export function validateCapacityPlan(plan, config) {
) {
throw new Error('capacity Terraform plan has an invalid service account')
}
if (config.mode === 'same-cap-cell') {
rehomeProtocol(config)
}
if (
config.mode === 'same-cap-cell' &&
(!SERVICE_ACCOUNT_EMAIL.test(config.rehomeDirectorServiceAccount ?? '') ||
@@ -504,7 +528,7 @@ export function validateCapacityPlan(plan, config) {
}
export function main(argv = process.argv.slice(2)) {
const config = parseArguments(argv)
const config = parseCapacityPlanArguments(argv)
const plan = JSON.parse(readFileSync(0, 'utf8'))
process.stdout.write(`${JSON.stringify({ event: 'relay_capacity_plan_verified', ...validateCapacityPlan(plan, config) })}\n`)
}
@@ -1,6 +1,9 @@
import assert from 'node:assert/strict'
import { test } from 'node:test'
import { validateCapacityPlan as validateCapacityPlanRaw } from './validate-relay-capacity-plan.mjs'
import {
parseCapacityPlanArguments,
validateCapacityPlan as validateCapacityPlanRaw
} from './validate-relay-capacity-plan.mjs'
const config = {
cellId: 'staging-gce-c3',
@@ -466,7 +469,8 @@ test('same-cap mode preserves 1000/60 while adding only the reviewed trust confi
image,
rollbackImage,
rehomeDirectorServiceAccount: directorIdentity,
rehomeAudience: audience
rehomeAudience: audience,
regionalRehomeProtocol: '1'
}
assert.deepEqual(
validateCapacityPlan({ resource_changes: [template, manager] }, sameCapConfig),
@@ -644,3 +648,134 @@ test('same-cap mode preserves 1000/60 while adding only the reviewed trust confi
{ mode: 'same-cap-image', changes: 1, changeKind: 'manager-convergence' }
)
})
test('protocol-0 same-cap cells roll without rehome trust lines', () => {
const rollbackImage = `us-docker.pkg.dev/project/relay/image@sha256:${'d'.repeat(64)}`
const image = `us-docker.pkg.dev/project/relay/image@sha256:${'e'.repeat(64)}`
const directorIdentity = 'relay-director@project.iam.gserviceaccount.com'
const audience = 'https://relay.example.com/v1/admin/host-drain'
const startup = ({ selectedImage, trust = false }) => [
` printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '3000'`,
` printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '60'`,
` printf 'ORCA_RELAY_CELL_REGION=%s\\n' 'asia-east2'`,
...(trust ? [
` printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '${directorIdentity}'`,
` printf 'ORCA_RELAY_REHOME_AUDIENCE=%s\\n' '${audience}'`
] : []),
`printf 'ORCA_RELAY_IMAGE_DIGEST=%s\\n' '${selectedImage.split('@')[1]}'`,
`docker pull '${selectedImage}'`,
'docker run --detach \\',
' --name orca-relay \\',
` '${selectedImage}'`
].join('\n')
const template = {
address: 'google_compute_instance_template.relay_gce_cell["production-gce-c27"]',
change: {
actions: ['create', 'delete'],
before: { metadata_startup_script: startup({ selectedImage: rollbackImage }) },
after: { metadata_startup_script: startup({ selectedImage: image }), self_link: null },
after_unknown: { self_link: true }
}
}
const manager = {
address: 'google_compute_instance_group_manager.relay_gce_cell["production-gce-c27"]',
change: {
actions: ['update'],
before: { target_size: 1, version: [{ instance_template: 'old' }] },
after: { target_size: 1, version: [{ instance_template: null }] },
after_unknown: { version: [{ instance_template: true }] }
}
}
const asiaConfig = {
cellId: 'production-gce-c27',
hardCap: 3_000,
unobservedBound: 60,
mode: 'same-cap-cell',
image,
rollbackImage,
rehomeDirectorServiceAccount: directorIdentity,
rehomeAudience: audience,
regionalRehomeProtocol: '0'
}
assert.deepEqual(
validateCapacityPlan({ resource_changes: [template, manager] }, asiaConfig),
{ mode: 'same-cap-cell', changes: 2 }
)
const gainsTrust = structuredClone(template)
gainsTrust.change.after.metadata_startup_script = startup({
selectedImage: image,
trust: true
})
assert.throws(
() => validateCapacityPlan({ resource_changes: [gainsTrust, manager] }, asiaConfig),
/reviewed image and capacity/
)
// Under protocol 1 that same script is the reviewed roll: trust is added, not drift.
assert.deepEqual(
validateCapacityPlan(
{ resource_changes: [gainsTrust, manager] },
{ ...asiaConfig, regionalRehomeProtocol: '1' }
),
{ mode: 'same-cap-cell', changes: 2 }
)
// A protocol-1 cell whose script has no rehome lines is the pre-existing failure, unchanged.
assert.throws(
() => validateCapacityPlan(
{ resource_changes: [template, manager] },
{ ...asiaConfig, regionalRehomeProtocol: '1' }
),
/reviewed image and capacity/
)
for (const protocol of [undefined, '', '2', 'yes']) {
assert.throws(
() => validateCapacityPlan(
{ resource_changes: [template, manager] },
{ ...asiaConfig, regionalRehomeProtocol: protocol }
),
/invalid regional rehome protocol/
)
}
})
test('the rehome protocol argument is required by same-cap-cell mode alone', () => {
const image = `us-docker.pkg.dev/project/relay/image@sha256:${'e'.repeat(64)}`
const rollbackImage = `us-docker.pkg.dev/project/relay/image@sha256:${'d'.repeat(64)}`
const sameCapArguments = (...extra) => [
'--mode', 'same-cap-cell',
'--cell-id', 'production-gce-c27',
'--hard-cap', '3000',
'--unobserved-bound', '60',
'--image', image,
'--rollback-image', rollbackImage,
'--rehome-director-service-account', 'relay-director@project.iam.gserviceaccount.com',
'--rehome-audience', 'https://relay.onorca.dev/v1/admin/host-drain',
...extra
]
assert.equal(
parseCapacityPlanArguments(sameCapArguments('--regional-rehome-protocol', '0'))
.regionalRehomeProtocol,
'0'
)
assert.throws(
() => parseCapacityPlanArguments(sameCapArguments()),
/requires rollback image and rehome trust config/
)
for (const protocol of ['', '2', 'true']) {
assert.throws(
() => parseCapacityPlanArguments(sameCapArguments('--regional-rehome-protocol', protocol)),
/requires rollback image and rehome trust config/
)
}
assert.throws(
() => parseCapacityPlanArguments([
'--mode', 'bootstrap-cell',
'--cell-id', 'staging-gce-c3',
'--hard-cap', '1000',
'--unobserved-bound', '60',
'--image', image,
'--capacity-service-account', 'orca-cap@onorca-cloud.iam.gserviceaccount.com',
'--regional-rehome-protocol', '0'
]),
/applies only to same-cap-cell validation/
)
})
@@ -1,4 +1,5 @@
import { pathToFileURL } from 'node:url'
import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs'
const CAPACITY_PROTOCOL = 2
@@ -378,9 +379,12 @@ export async function verifyCapacityTransition(config, overrides = {}) {
const token = overrides.token ?? process.env.ORCA_RELAY_ADMIN_ID_TOKEN
if (!token || token.length > 8_192) throw new Error('admin identity token is unavailable')
const health = await responseJson(
await fetchImpl(`${config.directorOrigin}/health`, {
signal: AbortSignal.timeout(15_000)
}),
await fetchAdminOnceMore(
fetchImpl,
`${config.directorOrigin}/health`,
{},
{ wait, timeoutMs: 15_000 }
),
'director health'
)
if (health.ok !== true || health.connectionCapacityProtocol !== CAPACITY_PROTOCOL) {
@@ -394,12 +398,16 @@ export async function verifyCapacityTransition(config, overrides = {}) {
lastObservation = { runtimeAvailable: runtime !== null }
if ((runtime === null) === (config.runtime === 'unavailable')) {
const result = await responseJson(
await fetchImpl(`${config.directorOrigin}/v1/admin/cell-status`, {
method: 'POST',
headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' },
body: JSON.stringify({ v: 1, cellId: config.cellId }),
signal: AbortSignal.timeout(30_000)
}),
await fetchAdminOnceMore(
fetchImpl,
`${config.directorOrigin}/v1/admin/cell-status`,
{
method: 'POST',
headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' },
body: JSON.stringify({ v: 1, cellId: config.cellId })
},
{ wait }
),
'cell status'
)
const status = result.status
@@ -1094,3 +1094,77 @@ test('does not retry a rejected cell admin token', async () => {
)
assert.equal(waits, 0)
})
test('retries a transient 503 on the director cell-status read', async () => {
const base = harness()
const statusCalls = []
const result = await verifyCapacityTransition(config, {
token: 'masked-token',
wait: async () => {},
fetch: async (url, options) => {
const path = new URL(url).pathname
if (path !== '/v1/admin/cell-status') return await base(url, options)
statusCalls.push(path)
if (statusCalls.length === 1) return new Response('warming up', { status: 503 })
return await base(url, options)
}
})
assert.equal(statusCalls.length, 2)
assert.equal(result.cellId, config.cellId)
})
test('fails when both director cell-status attempts return a transient 503', async () => {
const base = harness()
let statusCalls = 0
await assert.rejects(
verifyCapacityTransition(config, {
token: 'masked-token',
wait: async () => {},
fetch: async (url, options) => {
const path = new URL(url).pathname
if (path !== '/v1/admin/cell-status') return await base(url, options)
statusCalls += 1
return new Response('warming up', { status: 503 })
}
}),
/cell status returned 503/
)
assert.equal(statusCalls, 2)
})
test('retries a transient 503 on the director health preflight', async () => {
const base = harness()
let healthCalls = 0
const result = await verifyCapacityTransition(config, {
token: 'masked-token',
wait: async () => {},
fetch: async (url, options) => {
const path = new URL(url).pathname
if (path !== '/health') return await base(url, options)
healthCalls += 1
if (healthCalls === 1) return new Response('warming up', { status: 503 })
return await base(url, options)
}
})
assert.equal(healthCalls, 2)
assert.equal(result.cellId, config.cellId)
})
test('fails when both director health attempts return a transient 503', async () => {
const base = harness()
let healthCalls = 0
await assert.rejects(
verifyCapacityTransition(config, {
token: 'masked-token',
wait: async () => {},
fetch: async (url, options) => {
const path = new URL(url).pathname
if (path !== '/health') return await base(url, options)
healthCalls += 1
return new Response('warming up', { status: 503 })
}
}),
/director health returned 503/
)
assert.equal(healthCalls, 2)
})
+28 -1
View File
@@ -73,6 +73,13 @@ for a committed forward-recovery gate. Durable files default to
gap resets the active window at the next fresh sample and preserves the prior
window evidence. A threshold freeze never clears automatically.
A signal that reads missing or stale may miss up to two consecutive samples
without restarting the window. The sample still counts and is still checked
against every threshold it can read, and each tolerated gap is recorded in
`continuityEvents` with `tolerated: true`. A third consecutive miss of the same
signal, a failed collector, a runner gap, or any threshold breach restarts or
freezes as before.
A production candidate or multi-target mutation must download the exact
dry-run artifact by workflow run ID and attempt. It verifies the artifact
hashes and provenance, requires a green completed 15-minute state no older
@@ -89,7 +96,8 @@ durably marked consumed before mutation and cannot authorize another run.
| Signal | Freeze condition |
| --- | ---: |
| Active probe age | over 60 seconds |
| Cloud/log data age | over 180 seconds |
| Cloud Monitoring data age | over 330 seconds |
| Relay log and director admin data age | over 180 seconds |
| Cell heartbeat age | over 45 seconds |
| Endpoint latency | over 2,000 ms |
| Cloud SQL CPU | over 80% |
@@ -155,6 +163,25 @@ heartbeats, and matching live admission.
separate it from today's baseline; the exhausted-retry bar (incident peak
467 vs bar 300), director concurrency, and the pool bars carry that role.
Re-tighten after the fleet is on the 500 ms lock wait.
- Raised the Cloud Monitoring freshness bar from 180 s to 330 s and let a
freshness-only failure miss up to two consecutive samples without restarting
the window (2026-09-05). Basis: Google's metric list documents Cloud Run
`request_count`, `container/instance_count`, `container/cpu/utilizations`,
`container/memory/utilizations` and `container/max_request_concurrencies` as
"Sampled every 60 seconds. After sampling, data is not visible for up to 120
seconds", and Cloud SQL `database/cpu/utilization`,
`database/memory/utilization`, `database/postgresql/num_backends`,
`database/postgresql/backends_in_wait` and `database/postgresql/deadlock_count`
as "up to 165 seconds", so the newest visible point is up to 180 s and 225 s
old respectively. Window-sum signals age further: `observedAt` is the newest
point in the 5-minute query window, so a label series that stops emitting
reads as 300 s old while its summed value is complete. The old bar sat under
all three. Production on 2026-09-04/05 restarted healthy 15-minute windows at
181 s and 255 s (`auth.errors`, run 33928912676) and at 189 s
(`cloud_sql.lock_waits`, run 33944873727), and the last of those then blew the
25-minute lineage cap at 1 500 004 ms, so a green fleet produced no verdict.
The director admin bar stays at 180 s and the nonzero lock-wait carry window
stays at 180 s; both publish on our own cadence.
- Recalibrated the exhausted-PostgreSQL-retry freeze from 0 to 300 per five
minutes (2026-09-04). Basis: #18521 cut the request-path cell-inventory
lock wait from the 1 s pool `lock_timeout` to 500 ms, so contended waiters
+2 -7
View File
@@ -211,7 +211,8 @@ resource "google_logging_metric" "relay_snapshot" {
label_extractors = {
role = "EXTRACT(jsonPayload.role)"
cell_id = "EXTRACT(jsonPayload.cellId)"
region = "EXTRACT(jsonPayload.region)"
# No region label: adding one replaces all 21 live metrics (label change = delete+create),
# which resets history and blanks the relay alert policies during the swap.
}
metric_descriptor {
@@ -230,12 +231,6 @@ resource "google_logging_metric" "relay_snapshot" {
value_type = "STRING"
description = "Durable relay cell identifier."
}
labels {
key = "region"
value_type = "STRING"
description = "Coarse Relay region."
}
}
bucket_options {
+1 -1
View File
@@ -21,7 +21,7 @@
"load:relay:recovery-gate": "node dev/scripts/run-relay-recovery-wave-gate.mjs",
"ops:relay": "pnpm --filter @orca-cloud/relay-ops dev",
"pretest": "node --test dev/scripts/capture-terraform-plan-baseline.test.mjs dev/scripts/operate-relay-asia-admission.test.mjs dev/scripts/prepare-relay-asia-director-cells.test.mjs dev/scripts/prepare-relay-asia-topology-input.test.mjs dev/scripts/production-cloud-sql-rollout-lock.test.mjs dev/scripts/read-relay-serving-regional-placement-version.test.mjs dev/scripts/relay-asia-admission-workflow.test.mjs dev/scripts/relay-asia-rollout-evidence.test.mjs dev/scripts/relay-asia-topology-workflow.test.mjs dev/scripts/relay-cloud-sql-connection-budget.test.mjs dev/scripts/relay-load-reader-evidence.test.mjs dev/scripts/relay-staging-deploy-identity.test.mjs dev/scripts/sanitize-relay-asia-admission-result.test.mjs dev/scripts/terraform-root-partition.test.mjs dev/scripts/validate-relay-asia-topology-plan.test.mjs ../.github/actions/cloud-sql-rollout-lease/action-contract.test.mjs ../.github/actions/cloud-sql-rollout-lease/storage-lease.test.mjs",
"test": "pnpm -r test && node --test dev/scripts/classify-relay-production-capacity-director.test.mjs dev/scripts/classify-relay-staging-bootstrap.test.mjs dev/scripts/deploy-relay-blue-green.test.mjs dev/scripts/deploy-relay-gce-candidate.test.mjs dev/scripts/deploy-relay-gce-multi-target.test.mjs dev/scripts/github-smoke-token.test.mjs dev/scripts/infra.test.mjs dev/scripts/operate-relay-regional-rehome.test.mjs dev/scripts/power-staging-relay.test.mjs dev/scripts/prepare-relay-capacity-canary.test.mjs dev/scripts/prepare-relay-production-capacity-canary.test.mjs dev/scripts/probe-relay-legacy-admission.test.mjs dev/scripts/probe-relay-rehome-trust.test.mjs dev/scripts/production-cell-image-digest-consistency.test.mjs dev/scripts/read-relay-production-capacity-identity.test.mjs dev/scripts/relay-admission-selector.test.mjs dev/scripts/relay-gce-terraform-fence.test.mjs dev/scripts/relay-load-connection-failure.test.mjs dev/scripts/relay-load-control-peer.test.mjs dev/scripts/relay-load-director-capacity-gate.test.mjs dev/scripts/relay-load-model.test.mjs dev/scripts/relay-load-phase-barrier.test.mjs dev/scripts/relay-load-placement-boundary.test.mjs dev/scripts/relay-load-profile.test.mjs dev/scripts/relay-load-rebind-boundary.test.mjs dev/scripts/relay-load-region-behavior.test.mjs dev/scripts/relay-load-request-unit-boundary.test.mjs dev/scripts/relay-load-run-lifecycle.test.mjs dev/scripts/relay-monitor-evidence.test.mjs dev/scripts/relay-production-capacity-wave.test.mjs dev/scripts/relay-production-capacity-workflow.test.mjs dev/scripts/relay-production-identity-boundaries.test.mjs dev/scripts/relay-production-same-cap-wave.test.mjs dev/scripts/relay-public-workflow-contract.test.mjs dev/scripts/relay-recovery-wave-gate.test.mjs dev/scripts/relay-region-observation-evidence.test.mjs dev/scripts/relay-regional-rehome-workflow.test.mjs dev/scripts/relay-rehome-aggregate-evidence.test.mjs dev/scripts/relay-repository.test.mjs dev/scripts/relay-staging-c4-refresh-workflow.test.mjs dev/scripts/relay-staging-capacity-identity.test.mjs dev/scripts/staging-relay-apply-guard.test.mjs dev/scripts/validate-relay-capacity-plan.test.mjs dev/scripts/verify-relay-capacity-transition.test.mjs dev/scripts/verify-relay-legacy-bootstrap.test.mjs dev/scripts/workload-identity-attribute-conditions.test.mjs",
"test": "pnpm -r test && node --test dev/scripts/classify-relay-production-capacity-director.test.mjs dev/scripts/classify-relay-staging-bootstrap.test.mjs dev/scripts/deploy-relay-blue-green.test.mjs dev/scripts/deploy-relay-gce-candidate.test.mjs dev/scripts/deploy-relay-gce-multi-target.test.mjs dev/scripts/github-smoke-token.test.mjs dev/scripts/infra.test.mjs dev/scripts/operate-relay-regional-rehome.test.mjs dev/scripts/power-staging-relay.test.mjs dev/scripts/prepare-relay-capacity-canary.test.mjs dev/scripts/prepare-relay-production-capacity-canary.test.mjs dev/scripts/probe-relay-legacy-admission.test.mjs dev/scripts/probe-relay-rehome-trust.test.mjs dev/scripts/production-cell-image-digest-consistency.test.mjs dev/scripts/read-relay-production-capacity-identity.test.mjs dev/scripts/relay-admin-endpoint-retry-workflow.test.mjs dev/scripts/relay-admin-transient-retry.test.mjs dev/scripts/relay-admission-selector.test.mjs dev/scripts/relay-gce-terraform-fence.test.mjs dev/scripts/relay-load-connection-failure.test.mjs dev/scripts/relay-load-control-peer.test.mjs dev/scripts/relay-load-director-capacity-gate.test.mjs dev/scripts/relay-load-model.test.mjs dev/scripts/relay-load-phase-barrier.test.mjs dev/scripts/relay-load-placement-boundary.test.mjs dev/scripts/relay-load-profile.test.mjs dev/scripts/relay-load-rebind-boundary.test.mjs dev/scripts/relay-load-region-behavior.test.mjs dev/scripts/relay-load-request-unit-boundary.test.mjs dev/scripts/relay-load-run-lifecycle.test.mjs dev/scripts/relay-monitor-evidence.test.mjs dev/scripts/relay-production-capacity-wave.test.mjs dev/scripts/relay-production-capacity-workflow.test.mjs dev/scripts/relay-production-identity-boundaries.test.mjs dev/scripts/relay-production-same-cap-wave.test.mjs dev/scripts/relay-public-workflow-contract.test.mjs dev/scripts/relay-recovery-wave-gate.test.mjs dev/scripts/relay-region-observation-evidence.test.mjs dev/scripts/relay-regional-rehome-workflow.test.mjs dev/scripts/relay-rehome-aggregate-evidence.test.mjs dev/scripts/relay-repository.test.mjs dev/scripts/relay-same-cap-script-census.test.mjs dev/scripts/relay-staging-c4-refresh-workflow.test.mjs dev/scripts/relay-staging-capacity-identity.test.mjs dev/scripts/staging-relay-apply-guard.test.mjs dev/scripts/validate-relay-capacity-plan.test.mjs dev/scripts/verify-relay-capacity-transition.test.mjs dev/scripts/verify-relay-legacy-bootstrap.test.mjs dev/scripts/workload-identity-attribute-conditions.test.mjs",
"typecheck": "pnpm -r typecheck"
},
"devDependencies": {
+7 -2
View File
@@ -6,8 +6,13 @@ ARG LIBASOUND_PACKAGE=libasound2t64
ENV DEBIAN_FRONTEND=noninteractive
# Install Electron's link-time libraries without adding a display server or FUSE.
RUN apt-get update \
&& apt-get install -y --no-install-recommends \
# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts.
RUN for attempt in 1 2 3 4 5; do \
apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \
if [ "$attempt" = 5 ]; then exit 100; fi; \
rm -rf /var/lib/apt/lists/*; sleep 20; \
done \
&& apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \
bash \
ca-certificates \
coreutils \
+7 -2
View File
@@ -5,8 +5,13 @@ ARG LIBASOUND_PACKAGE=libasound2t64
ENV DEBIAN_FRONTEND=noninteractive
RUN apt-get update \
&& apt-get install -y --no-install-recommends \
# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts.
RUN for attempt in 1 2 3 4 5; do \
apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \
if [ "$attempt" = 5 ]; then exit 100; fi; \
rm -rf /var/lib/apt/lists/*; sleep 20; \
done \
&& apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \
bash \
ca-certificates \
dbus-x11 \
@@ -2,8 +2,13 @@ FROM ubuntu@sha256:678c6550cc43645e08669028bc177f50be4e7c5b8cca677067b1914d4afc7
ENV DEBIAN_FRONTEND=noninteractive
RUN apt-get update \
&& apt-get install -y --no-install-recommends \
# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts.
RUN for attempt in 1 2 3 4 5; do \
apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \
if [ "$attempt" = 5 ]; then exit 100; fi; \
rm -rf /var/lib/apt/lists/*; sleep 20; \
done \
&& apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \
bash \
ca-certificates \
dbus-x11 \
@@ -12,14 +12,15 @@ const { join, resolve } = require('node:path')
* `_inSocket`, and it wraps a real Windows named-pipe handle from `fs.openSync(term.conin, 'w')`.
* Every terminal leaks one File handle for the life of the host process.
*
* The obvious fix -- and the one the desktop patch ships -- releases it at the TOP of the branch,
* before `_getConsoleProcessList()` forks and before the native kill. That is measurably worse than
* leaving the leak alone: teardown aborts partway, the forked console-list agent is never reaped,
* and both pipe handles stay alive instead of one. This asset releases it at the END of the branch
* instead, after the fork and the kill have already happened.
* The obvious fix -- and the placement `config/patches/node-pty@1.1.0.patch` uses -- releases it at
* the TOP of the branch, before `_getConsoleProcessList()` forks and before the native kill. That is
* measurably worse than leaving the leak alone: teardown aborts partway, the forked console-list
* agent is never reaped, and both pipe handles stay alive instead of one. This asset releases it at
* the END of the branch instead, after the fork and the kill have already happened.
*
* Measured on a Windows SSH host, 20 spawn/kill cycles, handles bucketed by NT object type
* (identical numbers standalone and through a real relay):
* (identical numbers standalone and through a real relay). Every row is the NON-DLL branch, which
* is the branch a relay runs -- see the divergence note below for why that matters:
*
* published node-pty File +1/terminal, Process flat
* desktop patch placement File +2/terminal, Process +1/terminal <-- 3x WORSE
@@ -34,18 +35,64 @@ const { join, resolve } = require('node:path')
* Why this ships as a relay asset rather than only in config/patches/node-pty@1.1.0.patch: pnpm
* patches do not cross the SSH boundary -- a relay host runs the tree `npm install` put there.
*
* DELIBERATE DIVERGENCE FROM THE DESKTOP: the desktop patch has the early placement and therefore
* the +2 File / +1 Process regression, measured against its exact installed tree. Correcting it
* there is a separate change with its own verification, so the two trees differ on this one hunk on
* purpose, and the test pins that so a future "sync the patches" does not copy the bug back.
* DELIBERATE DIVERGENCE FROM THE DESKTOP, AND WHY IT IS NOT A DESKTOP-TERMINAL BUG: the two hosts
* do not run the same branch of `kill()`. node-pty defaults `_useConptyDll` to false
* (`windowsPtyAgent.js`). Every desktop site that opens a terminal pane sets it true --
* `local-pty-utils.ts` (two) and `native-pty-spawn.ts` -- as does the `windows-conpty-warmup.ts`
* warm-up, so all of those take the `else` branch, where UPSTREAM ALREADY destroys the input
* socket. The relay passes no such option (`src/relay/pty-handler.ts`), so it takes the
* `!useConptyDll` branch -- the one this asset and the desktop patch both edit.
*
* NOT ADDRESSED, AND A SEPARATE DEFECT THAT IS STILL OPEN: a terminal that exits on its own is
* still torn down through `kill()` -- both hosts call `destroy()` on natural exit and
* `WindowsTerminal.destroy()` is `kill()` -- but the shell is already gone by then, and the
* ordering this patch relies on does not hold. Measured over 20 self-exit cycles with that
* `destroy()` issued: published +3 File/+1 Process per terminal, desktop-patched +2/+1, this tree
* +2/+1. So this patch does not close it and the desktop patch does not either. It is reachable
* for every Windows user, local and relay, on every terminal closed by typing `exit`.
* THE DESKTOP IS NOT ENTIRELY OFF THAT BRANCH. Two desktop sites omit the option and so run it
* too: the hidden rate-limit probes in `src/main/rate-limits/claude-pty.ts` and
* `codex-pty-rate-limit-probe.ts`. Both recur -- their fetchers poll -- and both tear down through
* `kill()`, so this hunk is live on the desktop, just never for a pane a user can see. Do not
* restate this as "the desktop never executes that branch": that sentence stood here for two
* revisions and is false.
*
* What the numbers above therefore do NOT cover: they were measured on relay-style spawn/kill
* cycles. Whether the early placement costs the same +2 File / +1 Process across a probe's
* lifecycle is UNMEASURED -- plausible, not established, and worth measuring before anyone quotes
* a desktop figure. What IS settled is the claim this comment replaced: that the desktop patch made
* every Windows user worse off ON EVERY TERMINAL. Terminals take the DLL branch, and the harness
* that produced that claim defaulted into the branch it was not trying to measure.
*
* The divergence is therefore about which branch each host runs for the workload that matters, not
* about a regression in the terminals users open. The test still pins it, because a future "sync
* the patches" would put the early placement onto the relay's branch, where it does cost +2 File
* and +1 Process per terminal.
*
* If you extend this enumeration, grep for `node-pty` rather than for a static import: those two
* probes were missed three times because they use `await import('node-pty')`.
*
* THE SELF-EXIT LEAK: FIXED FOR THE DESKTOP BY #18635, STILL LIVE ON A RELAY. A terminal that exits
* on its own is also torn down through `kill()` -- both hosts call `destroy()` on natural exit and
* `WindowsTerminal.destroy()` is `kill()` -- but the shell is already gone by then, so the ordering
* this asset relies on does not hold. Measured over 20 self-exit cycles on the NON-DLL branch:
* published +3 File/+1 Process per terminal, desktop patch placement +2/+1, this tree +2/+1. This
* asset does not close it.
*
* #18635 does, in `config/patches/node-pty@1.1.0.patch`: the baton outlives the shell so `PtyKill`
* still reaches `ClosePseudoConsole`, plus an unconditional conout dispose on the DLL branch. That
* fix does not reach a Windows relay, and no hunk in THIS file can carry it, because it is mostly
* NATIVE (`src/win/conpty.cc`) and this asset only rewrites `lib/*.js`. Three delivery paths exist
* and none currently covers Windows:
*
* - the pnpm patch does not cross the SSH boundary -- the remote `npm install` yields upstream's
* unpatched node-pty;
* - the orcad prebuild matrix has no win32 entry (`MATRIX_SLOTS`,
* `config/scripts/build-orcad-prebuilds.mjs`), so no Windows binary is ever compiled from
* patched source to ship;
* - a relay asset CAN patch native source and rebuild on the host -- that is exactly what
* `node-pty-1.1.0-master-cloexec-patch.cjs` does -- but it returns
* `skipped:unsupported-platform` for anything but linux/darwin. Extending it to win32 means
* requiring an MSVC toolchain on the relay host, a far heavier precondition than on Linux,
* where node-gyp already runs at install time.
*
* So a Windows SSH relay still leaks a pseudoconsole per self-exiting terminal, and closing it is a
* DELIVERY problem, not another hunk here. Do not read #18635's flat self-exit relay numbers as
* covering deployed relays: they were measured against a locally rebuilt binary, so they describe
* the relay CODE PATH on a patched tree, not the tree a relay host actually installs.
*/
const EXPECTED_NODE_PTY_VERSION = '1.1.0'
File diff suppressed because one or more lines are too long
@@ -0,0 +1,139 @@
#!/usr/bin/env node
// Counts how many whole-host process-table captures the agent-completion cadence costs.
//
// Local panes all resolve out of one TTL-deduped snapshot, and the inspection queue collapses
// every shared-observation task enqueued in the same tick onto a single capture. So the capture
// count is the number of DISTINCT wake instants across panes, not the number of pane wakes.
//
// This drives the production interval picker (`nextCadenceInspectionDelayMs`) against a baseline
// that reproduces the pre-change ±10% jitter, over a simulated wall-clock window.
import { spawnSync } from 'node:child_process'
import fs from 'node:fs'
import nodeModule from 'node:module'
import path from 'node:path'
import process from 'node:process'
import { fileURLToPath } from 'node:url'
if (!process.execArgv.includes('--experimental-transform-types')) {
const result = spawnSync(
process.execPath,
['--experimental-transform-types', '--no-warnings', import.meta.filename],
{ stdio: 'inherit' }
)
process.exit(result.status ?? 1)
}
nodeModule.registerHooks({
resolve(specifier, context, nextResolve) {
if (specifier.startsWith('.') && !/\.[cm]?[jt]s$/.test(specifier) && context.parentURL) {
const candidate = new URL(`${specifier}.ts`, context.parentURL)
if (fs.existsSync(fileURLToPath(candidate))) {
return { url: candidate.href, shortCircuit: true }
}
}
return nextResolve(specifier, context)
}
})
const ROOT = path.resolve(import.meta.dirname, '../..')
const WINDOW_MS = Number(process.env.ORCA_INSPECTION_BENCH_WINDOW_MS ?? '60000')
const PANE_COUNTS = (process.env.ORCA_INSPECTION_BENCH_PANES ?? '1,2,4,8')
.split(',')
.map((value) => Number(value.trim()))
if (!Number.isSafeInteger(WINDOW_MS) || WINDOW_MS <= 0) {
throw new Error(`ORCA_INSPECTION_BENCH_WINDOW_MS must be a positive integer, got ${WINDOW_MS}`)
}
for (const paneCount of PANE_COUNTS) {
if (!Number.isSafeInteger(paneCount) || paneCount <= 0) {
throw new Error(`ORCA_INSPECTION_BENCH_PANES entries must be positive, got ${paneCount}`)
}
}
const { nextCadenceInspectionDelayMs } = await import(
path.join(ROOT, 'src/renderer/src/components/terminal-pane/agent-completion-poll-interval.ts')
)
const { POLL_TIER_INTERVAL_MS } = await import(
path.join(ROOT, 'src/renderer/src/components/terminal-pane/agent-completion-poll-cadence.ts')
)
const { PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS } = await import(
path.join(ROOT, 'src/shared/process-table-snapshot-reader.ts')
)
// Pre-change: independent ±10% jitter per pane, re-rolled on every reschedule.
function baselineDelayMs(baseMs) {
return Math.round(baseMs * (1 + (Math.random() * 0.2 - 0.1)))
}
function simulate(paneCount, baseMs, pickDelay) {
const startedAt = 1_700_000_000_000
const wakes = []
for (let pane = 0; pane < paneCount; pane += 1) {
// Panes mount at arbitrary moments, which is what spreads them apart in the first place.
let clock = startedAt + Math.floor(Math.random() * baseMs)
while ((clock += pickDelay(baseMs, clock)) < startedAt + WINDOW_MS) {
wakes.push(clock)
}
}
// A wake is served from the snapshot the previous capture produced until that snapshot's TTL
// lapses, so the TTL window starts at the capture, not on an epoch grid.
let captures = 0
let snapshotExpiresAt = -Infinity
for (const wakeAt of wakes.sort((left, right) => left - right)) {
if (wakeAt >= snapshotExpiresAt) {
captures += 1
snapshotExpiresAt = wakeAt + PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS
}
}
return captures
}
function medianOf(rounds, run) {
const samples = Array.from({ length: rounds }, run).sort((left, right) => left - right)
return samples[Math.floor(samples.length / 2)]
}
const baseMs = POLL_TIER_INTERVAL_MS.idle
console.log(
`Agent-completion cadence — whole-host \`ps\` captures over ${WINDOW_MS / 1000}s at the idle tier (${baseMs}ms)\n`
)
console.log('| visible panes | before | after | reduction |')
console.log('| --- | --- | --- | --- |')
for (const paneCount of PANE_COUNTS) {
const before = medianOf(21, () => simulate(paneCount, baseMs, baselineDelayMs))
const after = medianOf(21, () =>
simulate(paneCount, baseMs, (base, now) =>
nextCadenceInspectionDelayMs({
baseMs: base,
hasConsecutiveErrors: false,
alignToSharedGrid: true,
now
})
)
)
// A window shorter than one cadence tier can leave the baseline at zero; reporting a
// percentage off that divides by zero and prints a meaningless reduction.
const reduction = before > 0 ? `${(((before - after) / before) * 100).toFixed(0)}%` : 'n/a'
console.log(`| ${paneCount} | ${before} | ${after} | ${reduction} |`)
}
// Detection latency must not regress: the grid deadline is always within one interval.
let worstDelay = 0
for (let sample = 0; sample < 100_000; sample += 1) {
const now = 1_700_000_000_000 + sample * 7
worstDelay = Math.max(
worstDelay,
nextCadenceInspectionDelayMs({
baseMs,
hasConsecutiveErrors: false,
alignToSharedGrid: true,
now
})
)
}
if (worstDelay > baseMs) {
throw new Error(`grid alignment delayed a poll to ${worstDelay}ms, above the ${baseMs}ms tier`)
}
console.log(
`\nWorst observed wait: ${worstDelay}ms (tier interval ${baseMs}ms) — no inspection is ever delayed.`
)
@@ -1,11 +1,18 @@
// The relay's copy of the ConPTY teardown release, and the guard that keeps it in lockstep with the
// desktop's own node-pty patch. pnpm patches do not cross the SSH boundary, so a relay runs the tree
// `npm install` put there; the desktop had this fix and the relay did not, and every terminal on a
// Windows SSH host leaked one File handle for the life of the relay process.
// The relay's copy of the ConPTY teardown release, and the guard that keeps it in lockstep with
// `config/patches/node-pty@1.1.0.patch`. pnpm patches do not cross the SSH boundary, so a relay runs
// the tree `npm install` put there, and every terminal on a Windows SSH host leaked one File handle
// for the life of the relay process.
//
// The ORDER of the conin release is the fix. Releasing it at the top of the branch -- what the
// desktop patch does -- was measured at 3x WORSE than shipping nothing (File +2/terminal and a new
// Process +1/terminal); releasing it after the console-list fork and the native kill is flat.
// The ORDER of the conin release is the fix. Releasing it at the top of the branch -- the placement
// the desktop patch uses -- was measured at 3x WORSE than shipping nothing (File +2/terminal and a
// new Process +1/terminal); releasing it after the console-list fork and the native kill is flat.
//
// Those numbers are the `!useConptyDll` branch, which is the branch a RELAY runs. Every desktop
// site that opens a terminal pane sets `useConptyDll: true` and takes the other branch, where
// upstream already destroys the input socket. Two hidden rate-limit probes
// (`src/main/rate-limits/claude-pty.ts`, `codex-pty-rate-limit-probe.ts`) do omit the option and so
// do run this hunk, but no user-visible pane does. The divergence pinned below is about which
// branch each host runs for terminals -- not about a regression in the panes users open.
import { createRequire } from 'node:module'
import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
import { join, resolve } from 'node:path'
@@ -90,9 +97,11 @@ describe('Windows SSH relay node-pty ConPTY teardown patch', () => {
)
})
// The one hunk that must NOT match the desktop, and the reason is measured, not stylistic:
// releasing conin before `_getConsoleProcessList()` forks aborts teardown partway.
it('releases conin after the console-list fork, not before it like the desktop patch', () => {
// The one hunk that must NOT match the desktop patch, and the reason is measured, not stylistic:
// on the branch a relay runs, releasing conin before `_getConsoleProcessList()` forks aborts
// teardown partway. Desktop terminal panes take the other branch, so no pane is affected either
// way; what this guards is a patch sync putting the early placement onto the relay's branch.
it('releases conin after the console-list fork, unlike the desktop patch placement', () => {
const fixture = writeNodePtyFixture('1.1.0')
patchNodePtyWindowsTeardown(fixture.root)
const patched = readFileSync(join(fixture.libDir, 'windowsPtyAgent.js'), 'utf8')
@@ -108,7 +117,8 @@ describe('Windows SSH relay node-pty ConPTY teardown patch', () => {
expect(branch.indexOf('this._inSocket.destroy();')).toBeGreaterThan(
branch.indexOf('this._getConsoleProcessList()')
)
// Pinned so a future "sync the relay asset to config/patches" cannot copy the regression back.
// Pinned so a future "sync the relay asset to config/patches" cannot copy the early placement
// onto the relay's branch, where it costs +2 File and +1 Process per terminal.
expect(patched).not.toBe(readFileSync(desktopPath('windowsPtyAgent.js'), 'utf8'))
})
+1
View File
@@ -219,6 +219,7 @@ const WINDOWS_PACKAGE_TESTS = [
'src/main/agent-hooks/windows-hook-payload-delivery.test.ts',
'src/main/windows/windows-pty-job.win32.test.ts',
'src/main/windows/windows-host-job.win32.test.ts',
'src/main/windows-live-tree-kill.win32.test.ts',
'src/main/wsl/wsl-runner.test.ts',
'src/main/wsl/wsl-guest-environment.test.ts',
'src/main/wsl/wsl-invocation-boundary.test.ts',
@@ -0,0 +1,364 @@
#!/usr/bin/env node
// Benchmarks four renderer projections that scaled worse than linearly with user data, each on a
// path that reruns per keystroke or per store write.
//
// Scenarios 1, 3 and 4 time the production export against a hand-written reproduction of the
// pre-change shape and assert both agree first. Scenario 2 is MODELLED on both sides: the
// projection lives inside the `useTabGroupItemProjections` React hook and cannot be imported
// without a renderer, so it reproduces the before/after loops rather than driving production.
import { spawnSync } from 'node:child_process'
import { transformSync } from 'esbuild'
import { performance } from 'node:perf_hooks'
import fs from 'node:fs'
import nodeModule from 'node:module'
import path from 'node:path'
import process from 'node:process'
import { fileURLToPath, pathToFileURL } from 'node:url'
if (!process.execArgv.includes('--experimental-transform-types')) {
const result = spawnSync(
process.execPath,
['--experimental-transform-types', '--no-warnings', import.meta.filename],
{ stdio: 'inherit' }
)
process.exit(result.status ?? 1)
}
const ROOT = path.resolve(import.meta.dirname, '../..')
const RENDERER = path.join(ROOT, 'src/renderer/src')
nodeModule.registerHooks({
resolve(specifier, context, nextResolve) {
if (!context.parentURL) {
return nextResolve(specifier, context)
}
const candidates = specifier.startsWith('@/')
? ['.ts', '.tsx', '/index.ts', '/index.tsx', ''].map(
(suffix) => path.join(RENDERER, specifier.slice(2)) + suffix
)
: specifier.startsWith('.') && !/\.[cm]?[jt]sx?$/.test(specifier)
? ['.ts', '.tsx'].map((suffix) =>
fileURLToPath(new URL(specifier + suffix, context.parentURL))
)
: []
const resolved = candidates.find((file) => fs.existsSync(file) && fs.statSync(file).isFile())
return resolved
? { url: pathToFileURL(resolved).href, shortCircuit: true }
: nextResolve(specifier, context)
},
// Node strips types from .ts but not .tsx; the sidebar row model transitively imports icons.
load(url, context, nextLoad) {
if (url.endsWith('.tsx')) {
const source = fs.readFileSync(fileURLToPath(url), 'utf8')
const { code } = transformSync(source, { loader: 'tsx', format: 'esm', jsx: 'automatic' })
return { format: 'module', source: code, shortCircuit: true }
}
if (url.endsWith('.json') && !url.includes('/node_modules/')) {
const source = fs.readFileSync(fileURLToPath(url), 'utf8')
return { format: 'module', source: `export default ${source}`, shortCircuit: true }
}
return nextLoad(url, context)
}
})
const importRenderer = (relativePath) =>
import(pathToFileURL(path.join(RENDERER, relativePath)).href)
function envInt(name, fallback) {
const value = Number(process.env[name] ?? fallback)
if (!Number.isSafeInteger(value) || value <= 0) {
throw new Error(`${name} must be a positive integer, got ${value}`)
}
return value
}
const KEYSTROKES = envInt('ORCA_QUADRATIC_BENCH_KEYSTROKES', 12)
const WORKTREES = envInt('ORCA_QUADRATIC_BENCH_WORKTREES', 300)
const TABS = envInt('ORCA_QUADRATIC_BENCH_TABS', 60)
const OPEN_FILES = envInt('ORCA_QUADRATIC_BENCH_OPEN_FILES', 120)
const CHANGED_FILES = envInt('ORCA_QUADRATIC_BENCH_CHANGED_FILES', 5000)
const SIDEBAR_ROWS = envInt('ORCA_QUADRATIC_BENCH_SIDEBAR_ROWS', 600)
const SIDEBAR_REPOS = envInt('ORCA_QUADRATIC_BENCH_SIDEBAR_REPOS', 80)
if (SIDEBAR_REPOS > SIDEBAR_ROWS) {
throw new Error(
'ORCA_QUADRATIC_BENCH_SIDEBAR_REPOS must not exceed ORCA_QUADRATIC_BENCH_SIDEBAR_ROWS'
)
}
function timeRounds(run, rounds = 7) {
run()
const samples = Array.from({ length: rounds }, () => {
const start = performance.now()
run()
return performance.now() - start
}).sort((left, right) => left - right)
return samples[Math.floor(rounds / 2)]
}
function repeat(times, run) {
return () => {
let last
for (let round = 0; round < times; round += 1) {
last = run()
}
return last
}
}
const results = []
function compare({ label, scale, drives, before, after }) {
if (JSON.stringify(before()) !== JSON.stringify(after())) {
throw new Error(`${label}: baseline disagreed with the indexed shape`)
}
results.push({ label, scale, drives, beforeMs: timeRounds(before), afterMs: timeRounds(after) })
}
// ------------------------------------------------- 1. workspace board search index
const { buildWorkspaceBoardPaletteDocuments, matchWorkspaceBoardWorktrees } = await importRenderer(
'components/sidebar/workspace-kanban-search.ts'
)
const repoMap = new Map([
['repo-1', { id: 'repo-1', name: 'orca', path: '/tmp/orca', branch: 'main' }]
])
const boardWorktrees = Array.from({ length: WORKTREES }, (_, index) => ({
id: `repo-1::/tmp/worktree-${index}`,
repoId: 'repo-1',
path: `/tmp/worktree-${index}`,
branch: `feature/search-target-${index}`,
title: `Workspace ${index} search target`,
isMain: false
}))
const queries = Array.from({ length: KEYSTROKES }, (_, index) => 'search'.slice(0, (index % 6) + 1))
const matchAll = (documents) =>
queries.map((query) => [
...matchWorkspaceBoardWorktrees({ worktrees: boardWorktrees, query, repoMap, documents })
])
compare({
label: 'workspace board filter (per keystroke burst)',
scale: `${WORKTREES} worktrees x ${KEYSTROKES} keystrokes`,
drives: 'production',
// Omitting `documents` is the pre-change shape: the index is rebuilt inside every match.
before: () => matchAll(undefined),
// The hook memoizes the index on [worktrees, repoMap]; only the match reruns per keystroke.
after: () => matchAll(buildWorkspaceBoardPaletteDocuments({ worktrees: boardWorktrees, repoMap }))
})
// ------------------------------------------------- 2. tab-group projections (modelled)
const groupTabs = Array.from({ length: TABS }, (_, index) => ({
id: `tab-${index}`,
entityId: `entity-${index}`,
contentType: index % 3 === 0 ? 'editor' : 'terminal'
}))
const openFiles = Array.from({ length: OPEN_FILES }, (_, index) => ({
id: `entity-${index}`,
path: `/tmp/file-${index}.ts`
}))
const tabOrder = groupTabs.map((tab) => tab.id)
// Production memoizes each index on its own source list, so a unified-tab write reuses it.
const openFileById = new Map(openFiles.map((item) => [item.id, item]))
const groupTabById = new Map(groupTabs.map((item) => [item.id, item]))
function tabProjections(findOpenFile, findGroupTab) {
const editorItems = groupTabs
.filter((item) => item.contentType === 'editor')
.map((item) => findOpenFile(item.entityId))
.filter((file) => file !== undefined)
const order = tabOrder.map((itemId) => findGroupTab(itemId)?.entityId ?? itemId)
return [editorItems, order]
}
compare({
label: 'tab-group projections (per unified-tab write)',
scale: `${TABS} tabs x ${OPEN_FILES} open files`,
drives: 'modelled',
before: repeat(200, () =>
tabProjections(
(id) => openFiles.find((candidate) => candidate.id === id),
(id) => groupTabs.find((candidate) => candidate.id === id)
)
),
after: repeat(200, () =>
tabProjections(
(id) => openFileById.get(id),
(id) => groupTabById.get(id)
)
)
})
// ------------------------------------------------- 3. source-control tree build
const { buildSourceControlTree } = await importRenderer(
'components/right-sidebar/source-control-tree.ts'
)
const { normalizeRelativePath } = await importRenderer('lib/path.ts')
const { splitPathSegments } = await importRenderer('components/right-sidebar/path-tree.ts')
const { compareFileNames } = await import(
pathToFileURL(path.join(ROOT, 'src/shared/file-name-sort.ts')).href
)
const changedEntries = Array.from({ length: CHANGED_FILES }, (_, index) => ({
path: `src/area-${index % 20}/module-${index % 60}/nested/deep/part-${index % 7}/file-${index}.ts`
}))
// Pre-change `buildSourceControlTree`: identical except each ancestor path is re-joined.
function buildSourceControlTreeBefore(area, entries) {
const makeDirectory = (dirPath, name, depth) => ({
type: 'directory',
key: `dir::${area}::${dirPath}`,
name,
path: dirPath,
area,
depth,
fileCount: 0,
children: [],
directoryChildren: new Map()
})
const root = makeDirectory('', '', -1)
for (const entry of entries) {
const normalizedPath = normalizeRelativePath(entry.path)
const segments = splitPathSegments(normalizedPath)
if (segments.length === 0) {
continue
}
let parent = root
for (let index = 0; index < segments.length - 1; index += 1) {
const name = segments[index]
const dirPath = segments.slice(0, index + 1).join('/')
let dir = parent.directoryChildren.get(name)
if (!dir) {
dir = makeDirectory(dirPath, name, index)
parent.directoryChildren.set(name, dir)
parent.children.push(dir)
}
parent = dir
}
parent.children.push({
type: 'file',
key: `${area}::${entry.path}`,
name: segments.at(-1),
path: normalizedPath,
entry,
area,
depth: segments.length - 1
})
}
const finalize = (node) => {
const directories = node.children.filter((child) => child.type === 'directory').map(finalize)
const files = node.children.filter((child) => child.type === 'file')
directories.sort((a, b) => compareFileNames(a.name, b.name))
files.sort((a, b) => compareFileNames(a.entry.path, b.entry.path))
const { directoryChildren: _, ...rest } = node
return {
...rest,
fileCount: files.length + directories.reduce((count, dir) => count + dir.fileCount, 0),
children: [...directories, ...files]
}
}
return finalize(root).children
}
compare({
label: 'source-control tree build (per filter keystroke)',
scale: `${CHANGED_FILES} changed files`,
drives: 'production',
before: () => buildSourceControlTreeBefore('unstaged', changedEntries),
after: () => buildSourceControlTree('unstaged', changedEntries)
})
// ------------------------------------------------- 4. sidebar header boundaries
const { getRepoHeaderSectionEndByRepoId } = await importRenderer(
'components/sidebar/worktree-header-section-boundaries.ts'
)
const { estimateRenderRowSize } = await importRenderer(
'components/sidebar/worktree-list/viewport/virtual-rows.ts'
)
const headerRowIndexes = new Set(
Array.from({ length: SIDEBAR_REPOS }, (_, repo) =>
Math.floor((repo * SIDEBAR_ROWS) / SIDEBAR_REPOS)
)
)
const sidebarRows = Array.from({ length: SIDEBAR_ROWS }, (_, index) =>
headerRowIndexes.has(index)
? {
type: 'header',
key: `repo:${index}`,
label: '',
count: 0,
tone: '',
repo: { id: `repo-${index}` }
}
: { type: 'item', rowKey: `wt:${index}`, sectionKey: '', depth: 0, groupDepth: 0 }
)
const headerRepoIds = sidebarRows.filter((row) => row.type === 'header').map((row) => row.repo.id)
const boundaryArgs = {
rows: sidebarRows,
firstHeaderIndex: 0,
// What `getSidebarOrderedRepoHeaderIdsByBucket` yields for repos outside any project group.
sidebarRepoHeaderIdsByBucket: new Map([['ungrouped', headerRepoIds]]),
repoHeaderBucketByRepoId: new Map(headerRepoIds.map((id) => [id, 'ungrouped']))
}
// Pre-change `getRepoHeaderSectionEndByRepoId`: a findIndex and an indexOf per header row.
function getRepoHeaderSectionEndByRepoIdBefore(args) {
const rowStarts = []
let offset = 0
for (let index = 0; index < args.rows.length; index += 1) {
rowStarts[index] = offset
offset += estimateRenderRowSize(args.rows, index, args.firstHeaderIndex, null)
}
rowStarts[args.rows.length] = offset
const sectionEndByRepoId = new Map()
for (let index = 0; index < args.rows.length; index += 1) {
const row = args.rows[index]
const repoId = row?.type === 'header' ? row.repo?.id : undefined
if (!repoId) {
continue
}
const bucketKey = args.repoHeaderBucketByRepoId.get(repoId)
const bucketRepoIds = bucketKey ? args.sidebarRepoHeaderIdsByBucket.get(bucketKey) : undefined
const bucketIndex = bucketRepoIds?.indexOf(repoId) ?? -1
const nextRepoId = bucketIndex >= 0 ? bucketRepoIds?.[bucketIndex + 1] : undefined
let endIndex = -1
if (nextRepoId) {
endIndex = args.rows.findIndex((r) => r.type === 'header' && r.repo?.id === nextRepoId)
} else {
endIndex = args.rows.length
for (let next = index + 1; next < args.rows.length; next += 1) {
if (args.rows[next]?.type === 'header' || args.rows[next]?.type === 'host-header') {
endIndex = next
break
}
}
}
sectionEndByRepoId.set(
repoId,
rowStarts[endIndex >= 0 ? endIndex : args.rows.length] ?? rowStarts[args.rows.length] ?? 0
)
}
return sectionEndByRepoId
}
compare({
label: 'sidebar header boundaries (per row-model rebuild)',
scale: `${SIDEBAR_REPOS} repos x ${SIDEBAR_ROWS} rows`,
drives: 'production',
before: repeat(50, () => [...getRepoHeaderSectionEndByRepoIdBefore(boundaryArgs)]),
after: repeat(50, () => [...getRepoHeaderSectionEndByRepoId(boundaryArgs)])
})
// -------------------------------------------------
console.log('Renderer quadratic-scan removals\n')
console.log('| projection | drives | scale | before | after | |')
console.log('| --- | --- | --- | --- | --- | --- |')
for (const row of results) {
console.log(
`| ${row.label} | ${row.drives} | ${row.scale} | ${row.beforeMs.toFixed(2)} ms | ${row.afterMs.toFixed(2)} ms | ${(row.beforeMs / row.afterMs).toFixed(1)}x |`
)
}
@@ -70,7 +70,7 @@ function valueAfter(flag) {
function buildImage(image) {
console.log(`Building ${image.name} fixture...`)
docker([
const buildArgs = [
'build',
'--build-arg',
`BASE_IMAGE=${image.base}`,
@@ -81,7 +81,16 @@ function buildImage(image) {
'-t',
image.tag,
'.'
])
]
// Why: apt fetches from archive.ubuntu.com stall or fail mid-sync; a second build usually lands on a healthy index.
try {
docker(buildArgs)
} catch (error) {
console.error(
`${error instanceof Error ? error.message : String(error)}\nRetrying docker build once...`
)
docker(buildArgs)
}
}
function extractAppImage(image) {
@@ -43,7 +43,7 @@ const artifactVolume = `orca-headless-serve-shutdown-${suffix}`
const sha256 = createHash('sha256').update(readFileSync(appImage)).digest('hex')
try {
docker([
const buildArgs = [
'build',
'--platform',
platform,
@@ -52,7 +52,15 @@ try {
'-t',
image,
shutdownDockerDirectory
])
]
// Why: apt fetches from archive.ubuntu.com stall or fail mid-sync; a second build usually lands on a healthy index.
const firstBuild = docker(buildArgs, { allowFailure: true })
if (firstBuild.status !== 0) {
process.stderr.write(
`${firstBuild.stdout}${firstBuild.stderr}\ndocker build failed with status ${firstBuild.status}; retrying once...\n`
)
docker(buildArgs)
}
docker(['volume', 'create', artifactVolume])
runDesktopStartupOracle({ image, appImage, platform })
docker([
@@ -173,20 +173,26 @@ function runCase(caseName) {
function buildImage() {
console.log(`Building ${tag}…`)
docker(
[
'build',
...dockerPlatformArgs,
'--build-arg',
`BASE_IMAGE=${base}`,
'-f',
'config/docker/cli-launch-contract/Dockerfile',
'-t',
tag,
'config/docker/cli-launch-contract'
],
{ timeoutMs: BUILD_TIMEOUT_MS }
)
const buildArgs = [
'build',
...dockerPlatformArgs,
'--build-arg',
`BASE_IMAGE=${base}`,
'-f',
'config/docker/cli-launch-contract/Dockerfile',
'-t',
tag,
'config/docker/cli-launch-contract'
]
// Why: apt fetches from archive.ubuntu.com stall or fail mid-sync; a second build usually lands on a healthy index.
try {
docker(buildArgs, { timeoutMs: BUILD_TIMEOUT_MS })
} catch (error) {
console.error(
`${error instanceof Error ? error.message : String(error)}\nRetrying docker build once…`
)
docker(buildArgs, { timeoutMs: BUILD_TIMEOUT_MS })
}
}
// Extract unprivileged so chrome-sandbox is not root-owned setuid.
@@ -0,0 +1,221 @@
#!/usr/bin/env node
// Benchmarks two CPU costs `setLocalWorkspaceSession` pays on every session write — the write
// that fires on something as ordinary as clicking between two terminal split panes.
//
// 1. capTerminalScrollbackSessionBuffer — UTF-8 budget scan per retained scrollback buffer
// 2. remapPaneKeys — pane-key map rebuild that steady state throws away
//
// The snapshot disk rewrite on the same path is measured separately (#18764).
//
// Each scenario runs the production export against a baseline that reproduces the pre-change
// shape, so the reported speedup cannot drift away from what production actually does.
import { spawnSync } from 'node:child_process'
import { performance } from 'node:perf_hooks'
import fs from 'node:fs'
import nodeModule from 'node:module'
import path from 'node:path'
import process from 'node:process'
import { fileURLToPath } from 'node:url'
if (!process.execArgv.includes('--experimental-transform-types')) {
const result = spawnSync(
process.execPath,
['--experimental-transform-types', '--no-warnings', import.meta.filename],
{ stdio: 'inherit' }
)
process.exit(result.status ?? 1)
}
// The app's TS sources import siblings without an extension; Node's ESM resolver needs it.
nodeModule.registerHooks({
resolve(specifier, context, nextResolve) {
if (specifier.startsWith('.') && !/\.[cm]?[jt]s$/.test(specifier) && context.parentURL) {
const candidate = new URL(`${specifier}.ts`, context.parentURL)
if (fs.existsSync(fileURLToPath(candidate))) {
return { url: candidate.href, shortCircuit: true }
}
}
return nextResolve(specifier, context)
}
})
const ROOT = path.resolve(import.meta.dirname, '../..')
const ROUNDS = Number(process.env.ORCA_SESSION_WRITE_BENCH_ROUNDS ?? '9')
const LEAVES = Number(process.env.ORCA_SESSION_WRITE_BENCH_LEAVES ?? '8')
const PANE_KEYS = Number(process.env.ORCA_SESSION_WRITE_BENCH_PANE_KEYS ?? '2000')
for (const [name, value] of [
['ORCA_SESSION_WRITE_BENCH_ROUNDS', ROUNDS],
['ORCA_SESSION_WRITE_BENCH_LEAVES', LEAVES],
['ORCA_SESSION_WRITE_BENCH_PANE_KEYS', PANE_KEYS]
]) {
if (!Number.isSafeInteger(value) || value <= 0) {
throw new Error(`${name} must be a positive integer, got ${value}`)
}
}
const { capTerminalScrollbackSessionBuffer } = await import(
path.join(ROOT, 'src/shared/workspace-session-terminal-buffers.ts')
)
const { TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT } = await import(
path.join(ROOT, 'src/shared/terminal-scrollback-limits.ts')
)
const { remapAcknowledgedAgentPaneKeys } = await import(
path.join(ROOT, 'src/main/persistence/restoring-sessions/pane-key-remapping.ts')
)
const { clampUtf8TextTail, measureUtf8ByteLength } = await import(
path.join(ROOT, 'src/shared/utf8-byte-limits.ts')
)
const { isTerminalLeafId, makePaneKey, parsePaneKey } = await import(
path.join(ROOT, 'src/shared/stable-pane-id.ts')
)
function median(samples) {
const sorted = [...samples].sort((left, right) => left - right)
return sorted[Math.floor(sorted.length / 2)]
}
function timeRounds(run) {
const samples = []
run()
for (let round = 0; round < ROUNDS; round += 1) {
const start = performance.now()
run()
samples.push(performance.now() - start)
}
return median(samples)
}
function report(label, baselineMs, currentMs, extra = '') {
const speedup = baselineMs / currentMs
console.log(
`${label}\n before ${baselineMs.toFixed(3)} ms → after ${currentMs.toFixed(3)} ms (${speedup.toFixed(1)}x)${extra}`
)
return speedup
}
// ---------------------------------------------------------------- scenario 1
// Verbatim pre-change capTerminalScrollbackSessionBuffer; measureUtf8ByteLength itself is unchanged.
function baselineCapScrollbackBuffer(buffer) {
if (
buffer.length <= TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT &&
!measureUtf8ByteLength(buffer, {
stopAfterBytes: TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT
}).exceededLimit
) {
return buffer
}
return clampUtf8TextTail(buffer, TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT).text
}
// A terminal that has been running a while sits at the cap, which is the case that scanned in full.
const scrollbackLine = `${''}build output line with a path /Users/dev/project/src/index.ts and a status ok\n`
let atCapBuffer = ''
while (atCapBuffer.length < TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT) {
atCapBuffer += scrollbackLine
}
atCapBuffer = atCapBuffer.slice(0, TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT)
if (capTerminalScrollbackSessionBuffer(atCapBuffer) !== baselineCapScrollbackBuffer(atCapBuffer)) {
throw new Error('scrollback cap disagreed with the baseline implementation')
}
// The session write runs the prune twice, once per retained leaf.
const CAP_CALLS_PER_WRITE = LEAVES * 2
const capBaselineMs = timeRounds(() => {
for (let call = 0; call < CAP_CALLS_PER_WRITE; call += 1) {
baselineCapScrollbackBuffer(atCapBuffer)
}
})
const capCurrentMs = timeRounds(() => {
for (let call = 0; call < CAP_CALLS_PER_WRITE; call += 1) {
capTerminalScrollbackSessionBuffer(atCapBuffer)
}
})
console.log(
`Session-write hot path — ${LEAVES} retained scrollback leaves, ${PANE_KEYS} accumulated pane keys\n`
)
report(
`1. scrollback UTF-8 budget scan (${CAP_CALLS_PER_WRITE} calls/write @ ${(atCapBuffer.length / 1024).toFixed(0)} KB)`,
capBaselineMs,
capCurrentMs
)
// ---------------------------------------------------------------- scenario 2
const paneKeys = {}
const leafIdByInputLeafIdByTabId = new Map()
for (let index = 0; index < PANE_KEYS; index += 1) {
const tabId = `tab-${index % 64}`
const leafId = `${(index % 64).toString(16).padStart(8, '0')}-0000-4000-8000-${index.toString(16).padStart(12, '0')}`
paneKeys[makePaneKey(tabId, leafId)] = index
let leaves = leafIdByInputLeafIdByTabId.get(tabId)
if (!leaves) {
leaves = new Map()
leafIdByInputLeafIdByTabId.set(tabId, leaves)
}
// Steady state: a stable UUID leaf maps to itself.
leaves.set(leafId, leafId)
}
// Verbatim pre-change remapPaneKeys: parses every key, then rebuilds the object regardless.
function baselineRemapPaneKeys(values, remap) {
if (!values || Object.keys(values).length === 0) {
return { values, changed: false }
}
let changed = false
const next = {}
const setValue = (paneKey, value) => {
const existing = next[paneKey]
next[paneKey] = existing === undefined ? value : Math.max(existing, value)
}
for (const [paneKey, value] of Object.entries(values)) {
if (parsePaneKey(paneKey)) {
setValue(paneKey, value)
continue
}
const delimiter = paneKey.indexOf(':')
if (delimiter <= 0 || delimiter === paneKey.length - 1) {
setValue(paneKey, value)
continue
}
const tabId = paneKey.slice(0, delimiter)
const remappedLeafId = remap.get(tabId)?.get(paneKey.slice(delimiter + 1))
if (!remappedLeafId || !isTerminalLeafId(remappedLeafId)) {
setValue(paneKey, value)
continue
}
try {
setValue(makePaneKey(tabId, remappedLeafId), value)
changed = true
} catch {
setValue(paneKey, value)
}
}
return { values: next, changed }
}
// The write remaps three of these maps: acknowledgements, activity cutoffs, manual unread.
const REMAP_CALLS_PER_WRITE = 3
const remapBaselineMs = timeRounds(() => {
for (let call = 0; call < REMAP_CALLS_PER_WRITE; call += 1) {
baselineRemapPaneKeys(paneKeys, leafIdByInputLeafIdByTabId)
}
})
const remapCurrentMs = timeRounds(() => {
for (let call = 0; call < REMAP_CALLS_PER_WRITE; call += 1) {
remapAcknowledgedAgentPaneKeys(paneKeys, leafIdByInputLeafIdByTabId)
}
})
const remapResult = remapAcknowledgedAgentPaneKeys(paneKeys, leafIdByInputLeafIdByTabId)
if (remapResult.changed || remapResult.acknowledgements !== paneKeys) {
throw new Error('steady-state remap should return the input map untouched')
}
report(
`2. pane-key remap (${REMAP_CALLS_PER_WRITE} maps/write @ ${PANE_KEYS} keys)`,
remapBaselineMs,
remapCurrentMs,
' — and 3 discarded objects/write become 0'
)
@@ -0,0 +1,88 @@
#!/usr/bin/env node
// Times the partial-escape-tail fold that runs once per PTY chunk for every terminal against a
// baseline with the pre-change shape (unconditional concat + per-code-unit walk). Equivalence is
// proven over a corpus first, so the reported speedup cannot come from the gate changing the answer.
import { performance } from 'node:perf_hooks'
import {
advancePartialEscapeTail,
extractPartialEscapeTail,
MAX_PARTIAL_ESCAPE_TAIL_LENGTH
} from '../../src/shared/terminal-partial-escape-tail.ts'
const CHUNK_BYTES = 16 * 1024
const CHUNKS = 640
const ROUNDS = 7
function baselineAdvance(pendingTail, chunk) {
const tail = extractPartialEscapeTail(pendingTail + chunk)
return tail.length > MAX_PARTIAL_ESCAPE_TAIL_LENGTH ? '' : tail
}
const chunkOf = (line) => line.repeat(Math.ceil(CHUNK_BYTES / line.length)).slice(0, CHUNK_BYTES)
const escFreeChunk = chunkOf('[build] compiled src/renderer/src/components/thing.tsx in 12ms\n')
const colouredChunk = chunkOf(
'\x1b[32m[build]\x1b[0m compiled src/renderer/src/components/thing.tsx in 12ms\n'
)
// Every state the scanner can be left in, plus the boundaries the gate must not swallow.
const PIECES = [
'',
'plain output\n',
'\x1b[32mgreen\x1b[0m',
'\x1b[3',
'\x1b]0;title\x07',
'\x1b]0;partial',
'\x1bP dcs payload',
'\x1b',
'\x18',
'\x1a',
'\x1b]8;;https://example.com\x1b\\',
'\x1b]8;;https://example.com\x1b',
'\x1b(B',
'\x1b(',
'\x1b[1;2;3',
escFreeChunk
]
let checked = 0
for (const pending of PIECES.map((piece) => extractPartialEscapeTail(piece))) {
for (const chunk of PIECES) {
const expected = baselineAdvance(pending, chunk)
const actual = advancePartialEscapeTail(pending, chunk)
if (expected !== actual) {
throw new Error(
`gate changed the tracked tail: ${JSON.stringify({ pending, chunk, expected, actual })}`
)
}
checked += 1
}
}
function medianMs(advance, chunk) {
// First sample is the warm-up and is discarded.
const samples = Array.from({ length: ROUNDS + 1 }, () => {
const start = performance.now()
let tail = ''
for (let index = 0; index < CHUNKS; index += 1) {
tail = advance(tail, chunk)
}
return performance.now() - start
})
return samples.slice(1).sort((left, right) => left - right)[Math.floor(ROUNDS / 2)]
}
const megabytes = ((CHUNK_BYTES * CHUNKS) / 1024 / 1024).toFixed(1)
console.log(
`Partial-escape-tail fold: ${CHUNKS} x ${CHUNK_BYTES / 1024} KB chunks (${megabytes} MB), ${checked} equivalence cases verified\n`
)
console.log('| stream shape | before | after | |')
console.log('| --- | --- | --- | --- |')
for (const [label, chunk] of [
['ESC-free (build logs, `cat`, piped output)', escFreeChunk],
['SGR-coloured output (gate does not apply)', colouredChunk]
]) {
const before = medianMs(baselineAdvance, chunk)
const after = medianMs(advancePartialEscapeTail, chunk)
console.log(
`| ${label} | ${before.toFixed(2)} ms | ${after.toFixed(2)} ms | ${(before / after).toFixed(1)}x |`
)
}
@@ -11,7 +11,12 @@ import { repairTranslatedValue } from './locale-translation-policy.mjs'
const SOURCE_EXTENSIONS = new Set(['.ts', '.tsx', '.js', '.jsx', '.mts', '.cts'])
const SKIP_PATH_PARTS = new Set(['.git', 'dist', 'node_modules', 'out', '__snapshots__', 'assets'])
const LOCALIZATION_FUNCTION_NAMES = new Set(['t', 'translate', 'translateMain', 'translateSearchKeyword'])
const LOCALIZATION_FUNCTION_NAMES = new Set([
't',
'translate',
'translateMain',
'translateSearchKeyword'
])
const PLACEHOLDER_RE = /\{\{[^}]+\}\}/g
const LOCALES_RELATIVE_DIR = path.join('src', 'renderer', 'src', 'i18n', 'locales')
export const LOCALIZATION_SOURCE_ROOTS = [
+4 -4
View File
@@ -1,5 +1,5 @@
<svg xmlns="http://www.w3.org/2000/svg" width="106" height="20" role="img" aria-label="downloads: 39m">
<title>downloads: 39m</title>
<svg xmlns="http://www.w3.org/2000/svg" width="106" height="20" role="img" aria-label="downloads: 40m">
<title>downloads: 40m</title>
<linearGradient id="s" x2="0" y2="100%">
<stop offset="0" stop-color="#bbb" stop-opacity=".1"/>
<stop offset="1" stop-opacity=".1"/>
@@ -15,7 +15,7 @@
<g fill="#fff" text-anchor="middle" font-family="Verdana,Geneva,DejaVu Sans,sans-serif" text-rendering="geometricPrecision" font-size="11">
<text x="37" y="15" fill="#010101" fill-opacity=".3">downloads</text>
<text x="37" y="14">downloads</text>
<text x="90" y="15" fill="#010101" fill-opacity=".3">39m</text>
<text x="90" y="14">39m</text>
<text x="90" y="15" fill="#010101" fill-opacity=".3">40m</text>
<text x="90" y="14">40m</text>
</g>
</svg>

Before

Width:  |  Height:  |  Size: 935 B

After

Width:  |  Height:  |  Size: 935 B

+1 -3
View File
@@ -9,9 +9,7 @@ Browser-use profiles let you run the Orca browser with a specific identity — a
1. Open [Settings → Browser → Profiles](/docs/settings).
1. Click **Add profile**, give it a name.
1. Optionally seed it with cookies, a user-agent, and a viewport size.
1. For sites that reject Orca's default Chrome-shaped UA (some Google sign-in flows), create a profile that keeps the **native Electron user agent** instead of spoofing. Default profiles still use the cleaned Chrome UA for broader Cloudflare compatibility.
You can also create a no-spoof profile from the CLI with `orca tab profile create --no-ua-spoof` when you script browser setup.
1. Every profile presents Electron's own user agent. Orca no longer rewrites it to look like Chrome, because Cloudflare Turnstile rejects a Chrome-shaped UA that sends no client hints and accepts a declared Electron client. The only exception is Google's sign-in hosts, where Orca presents a Firefox identity so Google issues cookies bound to the embedded browser. A **native user agent** profile (`orca tab profile create --no-ua-spoof`) also skips that Google exception.
## Cookie import and Google sign-in
+1 -1
View File
@@ -2,7 +2,7 @@
"expo": {
"name": "Orca",
"slug": "orca-mobile",
"version": "0.0.47",
"version": "0.0.48",
"orientation": "default",
"icon": "./assets/icon.png",
"userInterfaceStyle": "automatic",
+10 -5
View File
@@ -192,6 +192,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile
return (
<Text
key={index}
selectable
style={[styles.heading, block.level <= 2 ? styles.headingLarge : null]}
>
{renderInline(block.text, onOpenFile)}
@@ -201,7 +202,9 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile
if (block.type === 'quote') {
return (
<View key={index} style={styles.quote}>
<Text style={styles.quoteText}>{renderInline(block.text, onOpenFile)}</Text>
<Text selectable style={styles.quoteText}>
{renderInline(block.text, onOpenFile)}
</Text>
</View>
)
}
@@ -223,7 +226,9 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile
return (
<View key={index} style={styles.codeBlock}>
{block.language ? <Text style={styles.codeLanguage}>{block.language}</Text> : null}
<Text style={styles.codeText}>{block.text}</Text>
<Text selectable style={styles.codeText}>
{block.text}
</Text>
</View>
)
}
@@ -251,7 +256,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile
<View style={styles.table}>
<View style={styles.tableRow}>
{visibleHeaders.map((header, cellIndex) => (
<Text key={cellIndex} style={[styles.tableCell, styles.tableHeader]}>
<Text key={cellIndex} selectable style={[styles.tableCell, styles.tableHeader]}>
{renderInline(header, onOpenFile)}
</Text>
))}
@@ -259,7 +264,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile
{visibleRows.map((row, rowIndex) => (
<View key={rowIndex} style={styles.tableRow}>
{visibleHeaders.map((_, cellIndex) => (
<Text key={cellIndex} style={styles.tableCell}>
<Text key={cellIndex} selectable style={styles.tableCell}>
{renderInline(row[cellIndex] ?? '', onOpenFile)}
</Text>
))}
@@ -290,7 +295,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile
? '[x]'
: '[ ]'}
</Text>
<Text style={[styles.listText, listScale]}>
<Text selectable style={[styles.listText, listScale]}>
{renderInline(item.text, onOpenFile)}
</Text>
</View>
@@ -6,11 +6,21 @@ import type { NativeChatMessage } from '../../../src/shared/native-chat-types'
vi.mock('react-native', async () => {
const React = await import('react')
const Text = ({ children, ...props }: { children?: unknown }): unknown =>
React.createElement('Text', props, children)
return {
Animated: {
Text,
Value: class {
setValue(): void {}
},
loop: (animation: unknown) => animation,
sequence: () => ({ start: vi.fn(), stop: vi.fn() }),
timing: () => ({ start: vi.fn(), stop: vi.fn() })
},
Image: 'Image',
Pressable: 'Pressable',
Text: ({ children, ...props }: { children?: unknown }) =>
React.createElement('Text', props, children),
Text,
View: ({ children, ...props }: { children?: unknown }) =>
React.createElement('View', props, children),
StyleSheet: { create: (styles: unknown) => styles, hairlineWidth: 1 }
@@ -21,7 +31,10 @@ vi.mock('lucide-react-native', () => ({
ArrowUp: 'ArrowUp',
ChevronDown: 'ChevronDown',
Copy: 'Copy',
SquareChevronRight: 'SquareChevronRight'
SquareChevronRight: 'SquareChevronRight',
SquareTerminal: 'SquareTerminal',
Wrench: 'Wrench',
ChevronRight: 'ChevronRight'
}))
vi.mock('../components/MobileMarkdown', () => ({ MobileMarkdown: 'MobileMarkdown' }))
@@ -45,7 +58,18 @@ describe('MobileNativeChatMessage', () => {
function render(
message: NativeChatMessage,
props: { toolsExpanded?: boolean } = {}
props: {
toolsExpanded?: boolean
structuredActivityUi?: boolean
activeTurnIsWorking?: boolean
turnExpanded?: boolean
turnStatus?: {
startedAt: number | null
thinking: boolean
workedSeconds: number | null
} | null
onToggleTurn?: () => void
} = {}
): ReactTestRenderer {
act(() => {
renderer = create(createElement(MobileNativeChatMessage, { message, ...props }))
@@ -152,4 +176,96 @@ describe('MobileNativeChatMessage', () => {
expect(tree.root.findAllByType('ChevronDown' as never)).toHaveLength(1)
expect(tree.root.findAllByType('SquareChevronRight' as never)).toHaveLength(1)
})
describe('structured activity UI', () => {
const runningCall = {
type: 'tool-call' as const,
name: 'Bash',
input: { command: 'npm test' },
state: 'running' as const
}
const settledCall = {
type: 'tool-call' as const,
name: 'Read',
input: { file_path: 'a/b.ts' },
state: 'completed' as const
}
it('shows the live tool label with a terminal glyph while a command runs', () => {
const tree = render(toolMessage([runningCall]), {
structuredActivityUi: true,
activeTurnIsWorking: true
})
expect(textIn(tree.root)).toContain('Running npm test')
expect(tree.root.findAllByType('SquareTerminal' as never)).toHaveLength(1)
expect(tree.root.findAllByType('Wrench' as never)).toHaveLength(0)
})
it('uses the wrench glyph for a non-command tool', () => {
const tree = render(
toolMessage([
{ type: 'tool-call', name: 'Read', input: { file_path: 'a/b.ts' }, state: 'running' }
]),
{ structuredActivityUi: true, activeTurnIsWorking: true }
)
expect(textIn(tree.root)).toContain('Running Read a/b.ts')
expect(tree.root.findAllByType('Wrench' as never)).toHaveLength(1)
})
it('falls back to the collapsed count row once the run settles', () => {
const tree = render(toolMessage([settledCall]), {
structuredActivityUi: true,
activeTurnIsWorking: true
})
expect(textIn(tree.root)).not.toContain('Running Read a/b.ts')
expect(textIn(tree.root)).toContain('1×')
})
it("hides a completed turn's activity until the turn caret discloses it", () => {
const collapsed = render(toolMessage([settledCall]), {
structuredActivityUi: true,
activeTurnIsWorking: false
})
expect(textIn(collapsed.root)).not.toContain('1×')
act(() => collapsed.unmount())
const disclosed = render(toolMessage([settledCall]), {
structuredActivityUi: true,
activeTurnIsWorking: false,
turnExpanded: true
})
expect(textIn(disclosed.root)).toContain('1×')
})
it('lets the global Tools toggle reveal a hidden settled run', () => {
// Otherwise the composer's Tools control is a no-op on every settled turn.
const tree = render(toolMessage([settledCall]), {
structuredActivityUi: true,
activeTurnIsWorking: false,
toolsExpanded: true
})
expect(textIn(tree.root)).toContain('1\u00d7')
})
it('keeps the bridge lane on its always-visible tool run', () => {
const tree = render(toolMessage([settledCall]), { activeTurnIsWorking: false })
expect(textIn(tree.root)).toContain('1×')
expect(tree.root.findAllByType('Wrench' as never)).toHaveLength(0)
})
it('renders the turn status row under a user message', () => {
const tree = render(userMessage([{ type: 'text', text: 'go' }]), {
structuredActivityUi: true,
turnStatus: { startedAt: Date.now(), thinking: true, workedSeconds: null }
})
expect(textIn(tree.root)).toContain('Thinking')
})
it('does not render a turn status row without one', () => {
const tree = render(userMessage([{ type: 'text', text: 'go' }]), {
structuredActivityUi: true
})
expect(textIn(tree.root)).toEqual(['go'])
})
})
})
+87 -220
View File
@@ -1,139 +1,20 @@
import { memo, useEffect, useRef, useState } from 'react'
import { Image, Pressable, Text, View } from 'react-native'
import * as Clipboard from 'expo-clipboard'
import { ArrowUp, ChevronDown, Copy, SquareChevronRight } from 'lucide-react-native'
import { diffFromText, diffFromToolCall } from '../../../src/shared/native-chat-diff'
import type { NativeChatDiffLine as DiffLine } from '../../../src/shared/native-chat-diff'
import { pairToolBlocks, splitNativeChatBlocks } from '../../../src/shared/native-chat-tool-fold'
import type { NativeChatToolPair as ToolPair } from '../../../src/shared/native-chat-tool-fold'
import {
createToolInputDisplay,
summarizeToolRun,
truncateToolDetail
} from '../../../src/shared/native-chat-tool-summary'
import { ArrowUp, Copy } from 'lucide-react-native'
import { splitNativeChatBlocks } from '../../../src/shared/native-chat-tool-fold'
import { selectActiveToolCall } from '../../../src/shared/native-chat-tool-activity'
import { isImageRefBlock, isTextBlock } from '../../../src/shared/native-chat-types'
import type { NativeChatBlock, NativeChatMessage } from '../../../src/shared/native-chat-types'
import { MobileMarkdown } from '../components/MobileMarkdown'
import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus'
import { ToolRun } from './MobileNativeChatToolRun'
import type { NativeChatTurnStatus } from './use-mobile-native-chat-turn-status'
import { colors } from '../theme/mobile-theme'
import { isRenderableImageUri } from './mobile-native-chat-image-preview'
import { styles, TEXT_SIZE } from './mobile-native-chat-message-styles'
import { nativeChatMessageText } from './mobile-native-chat-message-text'
const MAX_VISIBLE_TOOL_PAIRS = 6
const MAX_TOOL_RUN_DIFF_ROWS = 240
function DiffView({ lines }: { lines: DiffLine[] }): React.JSX.Element {
return (
<View style={styles.diff}>
{lines.map((line, i) => (
<Text
key={i}
style={[
styles.diffLine,
line.kind === 'add' && styles.diffAdd,
line.kind === 'del' && styles.diffDel,
line.kind === 'meta' && styles.diffMeta
]}
>
{line.kind === 'add' ? '+' : line.kind === 'del' ? '-' : ' '}
{line.text}
</Text>
))}
</View>
)
}
/** A single inline tool line — `▸ ToolName preview` — that expands in place to
* show the call's diff/input or the result's body. Mirrors the reference design
* where tool calls read as flat lines in the conversation, not boxed blocks. */
function ResultBody({
output,
isError,
diff
}: {
output: string
isError?: boolean
diff: DiffLine[] | null
}): React.JSX.Element {
if (diff) {
return <DiffView lines={diff} />
}
return (
<View style={[styles.toolResult, isError && styles.toolResultError]}>
<Text style={styles.mono}>{truncateToolDetail(output)}</Text>
</View>
)
}
/** One request: a tool call and its result rendered together as a single
* expandable line. `defaultExpanded` lets the group toggle open every line. */
function ToolLine({
pair,
defaultExpanded,
diffLineLimit,
onOpenFile
}: {
pair: ToolPair
defaultExpanded: boolean
diffLineLimit: number
onOpenFile?: (relativePath: string) => void
}): React.JSX.Element {
const [expanded, setExpanded] = useState(defaultExpanded)
const { call, result } = pair
const name = call ? call.name : 'Result'
const inputDisplay = call ? createToolInputDisplay(call.input) : null
const preview = inputDisplay?.label ?? result?.output.split('\n')[0]?.slice(0, 80) ?? ''
// Why: collapsed tool rows are the common path; defer bounded diff parsing
// and detail formatting until the user asks to reveal the detail.
const callDiff = expanded && call ? diffFromToolCall(call.name, call.input, diffLineLimit) : null
const resultDiff = expanded && result ? diffFromText(result.output, diffLineLimit) : null
const callDetail = expanded && inputDisplay && !callDiff ? inputDisplay.formatDetail() : undefined
const hasDetail = callDiff !== null || result !== undefined || inputDisplay?.hasDetail === true
// The group toggle opens every line at once, bypassing the tap guard, so the
// panel has to consult it too — else a detail-less row echoes its own label
// under itself and no tap can dismiss it.
const showDetail = hasDetail && expanded
// A tool that targets a file (Read/Edit/Write…) renders its preview as a
// tappable link that opens the file, independent of the line's expand tap.
const filePath = inputDisplay?.filePath ?? null
const openable = filePath !== null && onOpenFile !== undefined
return (
<View>
<Pressable
style={styles.toolLine}
onPress={() => hasDetail && setExpanded((v) => !v)}
hitSlop={6}
>
{showDetail ? (
<ChevronDown size={15} color={colors.textMuted} strokeWidth={2} />
) : (
<SquareChevronRight size={15} color={colors.textMuted} strokeWidth={2} />
)}
<Text style={styles.toolName}>{name}</Text>
{preview ? (
<Text
style={[styles.toolPreview, openable && styles.toolPreviewLink]}
numberOfLines={1}
onPress={openable ? () => onOpenFile!(filePath!) : undefined}
suppressHighlighting={!openable}
>
{preview}
</Text>
) : null}
</Pressable>
{showDetail ? (
<View style={styles.toolDetail}>
{callDiff ? <DiffView lines={callDiff} /> : null}
{callDetail ? <Text style={styles.mono}>{callDetail}</Text> : null}
{result ? (
<ResultBody output={result.output} isError={result.isError} diff={resultDiff} />
) : null}
</View>
) : null}
</View>
)
}
function Prose({
block,
invert,
@@ -150,7 +31,9 @@ function Prose({
// markdown renderer's light-on-dark palette.
if (invert) {
return (
<Text style={[styles.userText, { fontSize: TEXT_SIZE * fontScale }]}>{block.text}</Text>
<Text selectable style={[styles.userText, { fontSize: TEXT_SIZE * fontScale }]}>
{block.text}
</Text>
)
}
return (
@@ -180,67 +63,6 @@ function Prose({
return null
}
/** A run of a message's tool calls/results, collapsed to a one-line summary that
* expands to the individual inline tool lines. `defaultExpanded` lets the global
* toolbar toggle drive every run at once while still allowing per-run override. */
function ToolRun({
blocks,
defaultExpanded,
trailing,
onOpenFile
}: {
blocks: NativeChatBlock[]
defaultExpanded: boolean
trailing?: React.ReactNode
onOpenFile?: (relativePath: string) => void
}): React.JSX.Element {
const [open, setOpen] = useState(defaultExpanded)
const pairs = pairToolBlocks(blocks, MAX_VISIBLE_TOOL_PAIRS)
const diffLineLimit = Math.max(1, Math.floor(MAX_TOOL_RUN_DIFF_ROWS / (pairs.length * 2 || 1)))
let callCount = 0
for (const block of blocks) {
if (block.type === 'tool-call') {
callCount++
}
}
callCount ||= pairs.length
const summary = summarizeToolRun(blocks)
return (
<View style={styles.toolRun}>
<View style={styles.toolRunHeader}>
<Pressable style={styles.toolRunToggle} onPress={() => setOpen((v) => !v)} hitSlop={6}>
{open ? (
<ChevronDown size={15} color={colors.textMuted} strokeWidth={2} />
) : (
<SquareChevronRight size={15} color={colors.textMuted} strokeWidth={2} />
)}
<Text style={styles.toolRunCount}>{callCount}×</Text>
<Text style={styles.toolRunLabel} numberOfLines={1}>
{summary || `${callCount} tool ${callCount === 1 ? 'call' : 'calls'}`}
</Text>
</Pressable>
{trailing}
</View>
{open ? (
<View style={styles.toolRunBody}>
{pairs.map((pair, i) => (
<ToolLine
key={i}
pair={pair}
defaultExpanded={defaultExpanded}
diffLineLimit={diffLineLimit}
onOpenFile={onOpenFile}
/>
))}
{callCount > pairs.length ? (
<Text style={styles.toolPreview}>… {callCount - pairs.length} more tool calls</Text>
) : null}
</View>
) : null}
</View>
)
}
/** Subtle top-right controls for an agent message: copy its prose, or scroll so
* this message's top aligns to the top of the viewport. */
function AgentControls({
@@ -280,7 +102,13 @@ function MobileNativeChatMessageImpl({
fontScale = 1,
messageIndex,
onScrollToMessage,
onOpenFile
onOpenFile,
turnStatus,
turnExpanded,
turnKey,
onToggleTurn,
activeTurnIsWorking,
structuredActivityUi = false
}: {
message: NativeChatMessage
toolsExpanded?: boolean
@@ -291,6 +119,18 @@ function MobileNativeChatMessageImpl({
/** Ask the list to align this message's top to the top of the viewport. */
onScrollToMessage?: (index: number) => void
onOpenFile?: (relativePath: string) => void
/** This turn's status row, rendered under a user message (desktop parity). */
turnStatus?: NativeChatTurnStatus | null
/** Whether the turn caret has disclosed this turn's activity. */
turnExpanded?: boolean
/** Set only when this row's turn has settled and can disclose its activity. */
turnKey?: string
/** Stable across renders; the row supplies its own key when tapped. */
onToggleTurn?: (turnKey: string) => void
/** Session-level working state for this message's turn; gates the live tool row. */
activeTurnIsWorking?: boolean
/** Structured lane only: live tool progress plus the turn-status disclosure. */
structuredActivityUi?: boolean
}): React.JSX.Element {
const isUser = message.role === 'user'
const isReasoning = message.role === 'reasoning'
@@ -310,6 +150,20 @@ function MobileNativeChatMessageImpl({
// tool calls fold into a collapsible run beneath. The user's own messages get
// an inverted (filled accent) bubble so they stand apart from agent prose.
const { prose, tools } = splitNativeChatBlocks(message.blocks)
const activeCall = structuredActivityUi
? selectActiveToolCall(tools, { activeTurnIsWorking })
: null
// A completed turn's activity belongs behind the turn-status caret. Leaving the
// grouped row visible made a failed child command read as a failed response.
// The composer's global Tools toggle still overrides this, or it would silently
// do nothing on every settled turn.
const settledToolsHidden =
structuredActivityUi &&
activeCall == null &&
activeTurnIsWorking === false &&
!turnExpanded &&
!toolsExpanded
const showToolRun = tools.length > 0 && !settledToolsHidden
const handleCopy = (): void => {
const text = nativeChatMessageText(message.blocks)
@@ -338,39 +192,52 @@ function MobileNativeChatMessageImpl({
) : null
return (
<View style={[styles.row, isUser && styles.rowUser]}>
<View
style={[
styles.content,
isUser && styles.userBubble,
isReasoning && styles.reasoning,
copied && styles.copied
]}
>
{prose.map((block, index) => (
<Prose
key={index}
block={block}
invert={isUser}
fontScale={fontScale}
onOpenFile={onOpenFile}
/>
))}
{tools.length > 0 ? (
<ToolRun
// Why: a global toggle intentionally resets all per-run/per-line
// overrides in one remount, avoiding an effect-driven second render.
key={toolsExpanded ? 'expanded' : 'collapsed'}
blocks={tools}
defaultExpanded={toolsExpanded}
trailing={controls}
onOpenFile={onOpenFile}
/>
) : controls ? (
<View style={styles.controlsRow}>{controls}</View>
) : null}
<>
<View style={[styles.row, isUser && styles.rowUser]}>
<View
style={[
styles.content,
isUser && styles.userBubble,
isReasoning && styles.reasoning,
copied && styles.copied
]}
>
{prose.map((block, index) => (
<Prose
key={index}
block={block}
invert={isUser}
fontScale={fontScale}
onOpenFile={onOpenFile}
/>
))}
{showToolRun ? (
<ToolRun
// Why: a global toggle intentionally resets all per-run/per-line
// overrides in one remount, avoiding an effect-driven second render.
key={`${toolsExpanded ? 'expanded' : 'collapsed'}:${turnExpanded ? 'turn' : 'flat'}`}
blocks={tools}
defaultExpanded={turnExpanded || toolsExpanded}
expandChildren={turnExpanded ? false : toolsExpanded}
activeCall={activeCall}
trailing={controls}
onOpenFile={onOpenFile}
/>
) : controls ? (
<View style={styles.controlsRow}>{controls}</View>
) : null}
</View>
</View>
</View>
{turnStatus ? (
<MobileNativeChatTurnStatus
startedAt={turnStatus.startedAt}
thinking={turnStatus.thinking}
workedSeconds={turnStatus.workedSeconds}
expanded={turnExpanded ?? false}
onToggleExpanded={turnKey && onToggleTurn ? () => onToggleTurn(turnKey) : undefined}
/>
) : null}
</>
)
}
@@ -71,6 +71,7 @@ export function MobileNativeChatOverlay({
error={session.error}
agent={controller.nativeChatAgent}
agentWorking={controller.nativeChatAgentWorking}
structuredActivityUi={controller.nativeChatStructured}
streaming={streaming}
onStop={controller.handleNativeChatStop}
ask={controller.nativeChatAsk}
@@ -0,0 +1,74 @@
import type { AskAnswerSelection, AskPrompt } from '../../../src/shared/native-chat-ask'
import { MobileNativeChatAsk } from './MobileNativeChatAsk'
import { MobileNativeChatPermission } from './MobileNativeChatPermission'
import type { MobileChatPermission } from './mobile-native-chat-permission'
import { MobileNativeChatQuestion } from './MobileNativeChatQuestion'
import { mobileChatQuestionKey, type MobileChatQuestion } from './mobile-native-chat-question'
/** The one pending agent prompt shown above the composer: a structured
* AskUserQuestion wins, then a heuristic permission, then a heuristic question.
* The controller owns dismissal (it must survive this subtree unmounting on a
* view toggle); `ask` arrives already nulled while dismissed. */
export function MobileNativeChatPromptCard({
ask,
askKey,
onDismissAsk,
onAnswerAsk,
onCancelAsk,
permission,
onRespondPermission,
question,
onAnswerQuestion
}: {
ask?: AskPrompt | null
askKey?: string | null
onDismissAsk?: () => void
onAnswerAsk?: (prompt: AskPrompt, selections: AskAnswerSelection[]) => Promise<boolean>
onCancelAsk?: () => Promise<boolean>
permission?: MobileChatPermission | null
onRespondPermission?: (send: string) => Promise<boolean>
question?: MobileChatQuestion | null
onAnswerQuestion?: (text: string) => Promise<boolean>
}): React.JSX.Element | null {
if (ask) {
return (
<MobileNativeChatAsk
key={askKey ?? 'ask'}
prompt={ask}
onAnswer={async (selections) => {
const accepted = (await onAnswerAsk?.(ask, selections)) ?? false
if (accepted) {
onDismissAsk?.()
}
return accepted
}}
onCancel={async () => {
const accepted = (await onCancelAsk?.()) ?? false
if (accepted) {
onDismissAsk?.()
}
return accepted
}}
/>
)
}
if (permission) {
return (
<MobileNativeChatPermission
key={JSON.stringify(permission)}
permission={permission}
onRespond={async (send) => (await onRespondPermission?.(send)) ?? false}
/>
)
}
if (question) {
return (
<MobileNativeChatQuestion
key={mobileChatQuestionKey(question)}
question={question}
onAnswer={async (text) => (await onAnswerQuestion?.(text)) ?? false}
/>
)
}
return null
}
@@ -41,6 +41,7 @@ const MODEL_DESCRIPTOR: SessionOptionDescriptor = {
]
},
valueSource: 'reported',
transport: 'catalog',
settable: true
}
@@ -57,6 +58,7 @@ const EFFORT_DESCRIPTOR: SessionOptionDescriptor = {
]
},
valueSource: 'dispatched',
transport: 'catalog',
settable: true
}
@@ -66,6 +68,7 @@ const FAST_MODE_DESCRIPTOR: SessionOptionDescriptor = {
category: 'mode',
kind: { type: 'boolean', currentValue: false },
valueSource: 'reported',
transport: 'catalog',
settable: true
}
@@ -227,6 +230,7 @@ describe('MobileNativeChatSessionOptionPickers', () => {
...MODEL_DESCRIPTOR,
kind: { type: 'select', choices: [] },
valueSource: 'unknown',
transport: 'catalog',
action: { type: 'agent-picker' }
}
])
@@ -236,6 +240,46 @@ describe('MobileNativeChatSessionOptionPickers', () => {
expect(invokeAction).toHaveBeenCalledWith('model')
})
// The terminal transport can only learn the outcome by parsing the screen back,
// so the sheet admits the value is unconfirmed; the structured transport reports
// it every turn, which makes the same caption noise there.
it.each([
{ transport: 'catalog' as const, caption: true },
{ transport: 'agent-session' as const, caption: false }
])('captions a dispatched value only on the terminal transport', async (scenario) => {
mount([
MODEL_DESCRIPTOR,
{ ...EFFORT_DESCRIPTOR, valueSource: 'dispatched', transport: scenario.transport }
])
await act(async () => pill('Model').props.onPress())
await act(async () => rowByText('Effort').props.onPress())
const captions = renderer!.root
.findAll((node) => node.type === 'Text')
.filter(
(node) =>
(node.props as { children?: unknown }).children === 'Sent to the agent — not confirmed'
)
expect(captions.length > 0).toBe(scenario.caption)
})
it.each(['catalog', 'agent-session'] as const)(
'does not caption a reported value on the %s transport',
async (transport) => {
mount([MODEL_DESCRIPTOR, { ...EFFORT_DESCRIPTOR, valueSource: 'reported', transport }])
await act(async () => pill('Model').props.onPress())
await act(async () => rowByText('Effort').props.onPress())
expect(
renderer!.root
.findAll((node) => node.type === 'Text')
.some(
(node) =>
(node.props as { children?: unknown }).children ===
'Sent to the agent — not confirmed'
)
).toBe(false)
}
)
it('locks the pills while the agent is working', () => {
mount([MODEL_DESCRIPTOR, EFFORT_DESCRIPTOR], true)
expect(pill('Model').props).toMatchObject({ disabled: true })
@@ -3,9 +3,10 @@ import { ActivityIndicator, Keyboard, Pressable, StyleSheet, Text, View } from '
import { ChevronLeft, X } from 'lucide-react-native'
import { BottomDrawer } from '../components/BottomDrawer'
import { colors, radii, spacing, typography } from '../theme/mobile-theme'
import type {
SessionOptionDescriptor,
SessionOptionValue
import {
sessionOptionDispatchUnconfirmed,
type SessionOptionDescriptor,
type SessionOptionValue
} from '../../../src/shared/native-chat-session-options'
import {
mobileModelPillLabel,
@@ -119,7 +120,7 @@ export function MobileNativeChatSessionOptionPickers({
) : null}
</View>
</View>
{activeDescriptor.valueSource === 'dispatched' ? (
{sessionOptionDispatchUnconfirmed(activeDescriptor) ? (
<SessionOptionCaption>Sent to the agent — not confirmed</SessionOptionCaption>
) : null}
{reason ? <SessionOptionCaption>{reason}</SessionOptionCaption> : null}
@@ -0,0 +1,250 @@
import { useEffect, useRef, useState } from 'react'
import { Animated, Pressable, Text, View } from 'react-native'
import { ChevronDown, SquareChevronRight, SquareTerminal, Wrench } from 'lucide-react-native'
import { diffFromText, diffFromToolCall } from '../../../src/shared/native-chat-diff'
import type { NativeChatDiffLine as DiffLine } from '../../../src/shared/native-chat-diff'
import { pairToolBlocks } from '../../../src/shared/native-chat-tool-fold'
import type { NativeChatToolPair as ToolPair } from '../../../src/shared/native-chat-tool-fold'
import {
createToolInputDisplay,
summarizeToolRun,
truncateToolDetail
} from '../../../src/shared/native-chat-tool-summary'
import {
describeActiveToolCall,
formatActiveToolLabel,
formatToolCallCount,
isCommandToolName,
selectActiveToolCall
} from '../../../src/shared/native-chat-tool-activity'
import type { NativeChatBlock } from '../../../src/shared/native-chat-types'
import { colors } from '../theme/mobile-theme'
import { styles } from './mobile-native-chat-message-styles'
const MAX_VISIBLE_TOOL_PAIRS = 6
const MAX_TOOL_RUN_DIFF_ROWS = 240
function DiffView({ lines }: { lines: DiffLine[] }): React.JSX.Element {
return (
<View style={styles.diff}>
{lines.map((line, i) => (
<Text
key={i}
style={[
styles.diffLine,
line.kind === 'add' && styles.diffAdd,
line.kind === 'del' && styles.diffDel,
line.kind === 'meta' && styles.diffMeta
]}
>
{line.kind === 'add' ? '+' : line.kind === 'del' ? '-' : ' '}
{line.text}
</Text>
))}
</View>
)
}
/** A single inline tool line — `▸ ToolName preview` — that expands in place to
* show the call's diff/input or the result's body. Mirrors the reference design
* where tool calls read as flat lines in the conversation, not boxed blocks. */
function ResultBody({
output,
isError,
diff
}: {
output: string
isError?: boolean
diff: DiffLine[] | null
}): React.JSX.Element {
if (diff) {
return <DiffView lines={diff} />
}
return (
<View style={[styles.toolResult, isError && styles.toolResultError]}>
<Text style={styles.mono}>{truncateToolDetail(output)}</Text>
</View>
)
}
/** One request: a tool call and its result rendered together as a single
* expandable line. `defaultExpanded` lets the group toggle open every line. */
function ToolLine({
pair,
defaultExpanded,
diffLineLimit,
onOpenFile
}: {
pair: ToolPair
defaultExpanded: boolean
diffLineLimit: number
onOpenFile?: (relativePath: string) => void
}): React.JSX.Element {
const [expanded, setExpanded] = useState(defaultExpanded)
const { call, result } = pair
const name = call ? call.name : 'Result'
const inputDisplay = call ? createToolInputDisplay(call.input) : null
const preview = inputDisplay?.label ?? result?.output.split('\n')[0]?.slice(0, 80) ?? ''
// Why: collapsed tool rows are the common path; defer bounded diff parsing
// and detail formatting until the user asks to reveal the detail.
const callDiff = expanded && call ? diffFromToolCall(call.name, call.input, diffLineLimit) : null
const resultDiff = expanded && result ? diffFromText(result.output, diffLineLimit) : null
const callDetail = expanded && inputDisplay && !callDiff ? inputDisplay.formatDetail() : undefined
const hasDetail = callDiff !== null || result !== undefined || inputDisplay?.hasDetail === true
// The group toggle opens every line at once, bypassing the tap guard, so the
// panel has to consult it too — else a detail-less row echoes its own label
// under itself and no tap can dismiss it.
const showDetail = hasDetail && expanded
// A tool that targets a file (Read/Edit/Write…) renders its preview as a
// tappable link that opens the file, independent of the line's expand tap.
const filePath = inputDisplay?.filePath ?? null
const openable = filePath !== null && onOpenFile !== undefined
return (
<View>
<Pressable
style={styles.toolLine}
onPress={() => hasDetail && setExpanded((v) => !v)}
hitSlop={6}
>
{showDetail ? (
<ChevronDown size={15} color={colors.textMuted} strokeWidth={2} />
) : (
<SquareChevronRight size={15} color={colors.textMuted} strokeWidth={2} />
)}
<Text style={styles.toolName}>{name}</Text>
{preview ? (
<Text
style={[styles.toolPreview, openable && styles.toolPreviewLink]}
numberOfLines={1}
onPress={openable ? () => onOpenFile!(filePath!) : undefined}
suppressHighlighting={!openable}
>
{preview}
</Text>
) : null}
</Pressable>
{showDetail ? (
<View style={styles.toolDetail}>
{callDiff ? <DiffView lines={callDiff} /> : null}
{callDetail ? <Text style={styles.mono}>{callDetail}</Text> : null}
{result ? (
<ResultBody output={result.output} isError={result.isError} diff={resultDiff} />
) : null}
</View>
) : null}
</View>
)
}
/** Breathing label for a still-running tool, matching desktop's `animate-pulse`. */
function PulsingText({
style,
numberOfLines,
children
}: {
style?: React.ComponentProps<typeof Animated.Text>['style']
numberOfLines?: number
children: React.ReactNode
}): React.JSX.Element {
const pulse = useRef(new Animated.Value(1)).current
useEffect(() => {
const animation = Animated.loop(
Animated.sequence([
Animated.timing(pulse, { toValue: 0.45, duration: 700, useNativeDriver: true }),
Animated.timing(pulse, { toValue: 1, duration: 700, useNativeDriver: true })
])
)
animation.start()
return () => animation.stop()
}, [pulse])
return (
<Animated.Text style={[style, { opacity: pulse }]} numberOfLines={numberOfLines}>
{children}
</Animated.Text>
)
}
/** A run of a message's tool calls/results, collapsed to a one-line summary that
* expands to the individual inline tool lines. `defaultExpanded` lets the global
* toolbar toggle drive every run at once while still allowing per-run override. */
export function ToolRun({
blocks,
defaultExpanded,
expandChildren,
activeCall,
trailing,
onOpenFile
}: {
blocks: NativeChatBlock[]
defaultExpanded: boolean
/** Child tool lines stay collapsed when the turn caret drove the run open. */
expandChildren: boolean
/** The still-running call, when the turn is live (desktop parity). */
activeCall: ReturnType<typeof selectActiveToolCall>
trailing?: React.ReactNode
onOpenFile?: (relativePath: string) => void
}): React.JSX.Element {
const [open, setOpen] = useState(defaultExpanded)
const pairs = pairToolBlocks(blocks, MAX_VISIBLE_TOOL_PAIRS)
const diffLineLimit = Math.max(1, Math.floor(MAX_TOOL_RUN_DIFF_ROWS / (pairs.length * 2 || 1)))
let callCount = 0
for (const block of blocks) {
if (block.type === 'tool-call') {
callCount++
}
}
callCount ||= pairs.length
const summary = summarizeToolRun(blocks)
const ActiveToolIcon = activeCall && isCommandToolName(activeCall.name) ? SquareTerminal : Wrench
return (
<View style={styles.toolRun}>
<View style={styles.toolRunHeader}>
{activeCall ? (
<Pressable
style={styles.toolRunActive}
onPress={() => setOpen((v) => !v)}
hitSlop={6}
accessibilityRole="button"
accessibilityState={{ expanded: open }}
accessibilityLiveRegion="polite"
>
<ActiveToolIcon size={15} color={colors.textMuted} strokeWidth={2} />
<PulsingText style={styles.toolRunActiveLabel} numberOfLines={1}>
{formatActiveToolLabel(describeActiveToolCall(activeCall))}
</PulsingText>
{open ? <ChevronDown size={15} color={colors.textMuted} strokeWidth={2} /> : null}
</Pressable>
) : (
<Pressable style={styles.toolRunToggle} onPress={() => setOpen((v) => !v)} hitSlop={6}>
{open ? (
<ChevronDown size={15} color={colors.textMuted} strokeWidth={2} />
) : (
<SquareChevronRight size={15} color={colors.textMuted} strokeWidth={2} />
)}
<Text style={styles.toolRunCount}>{callCount}×</Text>
<Text style={styles.toolRunLabel} numberOfLines={1}>
{summary || formatToolCallCount(callCount)}
</Text>
</Pressable>
)}
{trailing}
</View>
{open ? (
<View style={styles.toolRunBody}>
{pairs.map((pair, i) => (
<ToolLine
key={i}
pair={pair}
defaultExpanded={expandChildren}
diffLineLimit={diffLineLimit}
onOpenFile={onOpenFile}
/>
))}
{callCount > pairs.length ? (
<Text style={styles.toolPreview}>… {callCount - pairs.length} more tool calls</Text>
) : null}
</View>
) : null}
</View>
)
}
@@ -0,0 +1,112 @@
import { createElement } from 'react'
import { act, create, type ReactTestInstance, type ReactTestRenderer } from 'react-test-renderer'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
vi.mock('react-native', async () => {
const React = await import('react')
const Text = ({ children, ...props }: { children?: unknown }): unknown =>
React.createElement('Text', props, children)
return {
Animated: {
Text,
Value: class {
constructor(private value: number) {}
setValue(next: number): void {
this.value = next
}
},
loop: (animation: unknown) => animation,
sequence: () => ({ start: vi.fn(), stop: vi.fn() }),
timing: () => ({ start: vi.fn(), stop: vi.fn() })
},
Pressable: ({ children, ...props }: { children?: unknown }) =>
React.createElement('Pressable', props, children),
Text,
View: ({ children, ...props }: { children?: unknown }) =>
React.createElement('View', props, children),
StyleSheet: { create: (styles: unknown) => styles, hairlineWidth: 1 }
}
})
vi.mock('lucide-react-native', () => ({ ChevronRight: 'ChevronRight' }))
import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus'
describe('MobileNativeChatTurnStatus', () => {
let renderer: ReactTestRenderer | null = null
beforeEach(() => {
vi.useFakeTimers()
vi.setSystemTime(new Date('2026-09-04T00:00:00Z'))
})
afterEach(() => {
act(() => renderer?.unmount())
renderer = null
vi.useRealTimers()
})
function render(props: {
startedAt: number | null
thinking: boolean
workedSeconds?: number | null
expanded?: boolean
onToggleExpanded?: () => void
}): ReactTestRenderer {
act(() => {
renderer = create(createElement(MobileNativeChatTurnStatus, props))
})
return renderer!
}
const labels = (node: ReactTestInstance): string[] =>
node.findAllByType('Text' as never).map((text) => String(text.children.join('')))
it('reads "Thinking" before the turn produces output', () => {
const tree = render({ startedAt: Date.now(), thinking: true })
expect(labels(tree.root)).toEqual(['Thinking'])
})
it('counts up once the turn is producing output', () => {
const startedAt = Date.now()
const tree = render({ startedAt, thinking: false })
expect(labels(tree.root)).toEqual(['Working for 0s'])
act(() => {
vi.advanceTimersByTime(12_000)
})
expect(labels(tree.root)).toEqual(['Working for 12s'])
})
it('settles to a tappable "Worked for" row that toggles the turn', () => {
const onToggleExpanded = vi.fn()
const tree = render({
startedAt: Date.now(),
thinking: false,
workedSeconds: 184,
onToggleExpanded
})
expect(labels(tree.root)).toEqual(['Worked for 3m 4s'])
const button = tree.root.findByType('Pressable' as never)
expect(button.props.accessibilityLabel).toBe('Toggle turn details')
expect(button.props.accessibilityState).toEqual({ expanded: false })
act(() => button.props.onPress())
expect(onToggleExpanded).toHaveBeenCalledOnce()
})
it('stays a plain row when the settled turn has nothing to disclose', () => {
const tree = render({ startedAt: Date.now(), thinking: false, workedSeconds: 5 })
expect(tree.root.findAllByType('Pressable' as never)).toHaveLength(0)
expect(labels(tree.root)).toEqual(['Worked for 5s'])
})
it('holds no interval once the turn has settled', () => {
render({ startedAt: Date.now(), thinking: false, workedSeconds: 5 })
expect(vi.getTimerCount()).toBe(0)
})
it('announces the live row to assistive tech', () => {
const tree = render({ startedAt: Date.now(), thinking: true })
const row = tree.root.findByType('View' as never)
expect(row.props.accessibilityLiveRegion).toBe('polite')
expect(row.props.accessibilityLabel).toBe('Agent is responding')
})
})
@@ -0,0 +1,117 @@
import { useEffect, useRef, useState } from 'react'
import { Animated, Pressable, StyleSheet, Text, View } from 'react-native'
import { ChevronRight } from 'lucide-react-native'
import {
formatNativeChatTurnStatusLabel,
NATIVE_CHAT_TURN_STATUS_COPY,
nativeChatElapsedSeconds
} from '../../../src/shared/native-chat-turn-status'
import { colors, spacing, typography } from '../theme/mobile-theme'
/** Seconds tick only while a turn is actually counting, so a settled transcript
* holds no timers. */
function useElapsedSeconds(startedAt: number | null, counting: boolean): number {
// Preserves the pre-stamp epoch for the frame before the turn's startedAt lands.
const [mountedAt] = useState(() => Date.now())
const [now, setNow] = useState(() => Date.now())
useEffect(() => {
if (!counting) {
return
}
setNow(Date.now())
const timer = setInterval(() => setNow(Date.now()), 1_000)
return () => clearInterval(timer)
}, [counting])
return counting ? nativeChatElapsedSeconds(startedAt, mountedAt, now) : 0
}
/** The per-turn status row — "Thinking", then "Working for 12s" while the turn
* runs, settling to a tappable "Worked for 3m 4s" that discloses the turn's
* tool activity. Desktop parity: `NativeChatWorkingStatus`. */
export function MobileNativeChatTurnStatus({
startedAt,
thinking,
workedSeconds,
expanded = false,
onToggleExpanded
}: {
startedAt: number | null
thinking: boolean
workedSeconds?: number | null
expanded?: boolean
onToggleExpanded?: () => void
}): React.JSX.Element {
const counting = !thinking && workedSeconds == null
const elapsedSeconds = useElapsedSeconds(startedAt, counting)
const label = formatNativeChatTurnStatusLabel({ thinking, workedSeconds, elapsedSeconds })
const pulse = useRef(new Animated.Value(1)).current
useEffect(() => {
if (!thinking) {
pulse.setValue(1)
return
}
const animation = Animated.loop(
Animated.sequence([
Animated.timing(pulse, { toValue: 0.45, duration: 700, useNativeDriver: true }),
Animated.timing(pulse, { toValue: 1, duration: 700, useNativeDriver: true })
])
)
animation.start()
return () => animation.stop()
}, [pulse, thinking])
const rowStyle = [styles.row, thinking ? null : styles.rowSettled]
if (workedSeconds != null && onToggleExpanded) {
return (
<Pressable
style={({ pressed }) => [...rowStyle, pressed && styles.pressed]}
onPress={onToggleExpanded}
hitSlop={6}
accessibilityRole="button"
accessibilityState={{ expanded }}
accessibilityLabel={NATIVE_CHAT_TURN_STATUS_COPY.toggleDetails}
>
<Text style={styles.label}>{label}</Text>
<View style={expanded ? styles.caretOpen : undefined}>
<ChevronRight size={14} color={colors.textMuted} strokeWidth={2} />
</View>
</Pressable>
)
}
return (
<View
style={rowStyle}
accessibilityLiveRegion="polite"
accessibilityLabel={NATIVE_CHAT_TURN_STATUS_COPY.responding}
>
<Animated.Text style={[styles.label, thinking && { opacity: pulse }]}>{label}</Animated.Text>
</View>
)
}
const styles = StyleSheet.create({
row: {
flexDirection: 'row',
alignItems: 'center',
gap: spacing.xs,
minHeight: 28,
paddingHorizontal: spacing.md
},
rowSettled: {
borderBottomWidth: StyleSheet.hairlineWidth,
borderBottomColor: colors.borderSubtle
},
pressed: {
opacity: 0.6
},
label: {
color: colors.textMuted,
fontSize: typography.bodySize
},
caretOpen: {
transform: [{ rotate: '90deg' }]
}
})
@@ -72,6 +72,9 @@ type Overrides = {
inputLockReason?: 'disconnected' | 'waiting' | null
onSend?: (text: string) => Promise<boolean>
pending?: Parameters<typeof MobileNativeChatView>[0]['pending']
structuredActivityUi?: boolean
agentWorking?: boolean
sendSurfaceId?: string
}
function assistantTurn(id: string, text: string): NativeChatMessage {
@@ -243,4 +246,111 @@ describe('MobileNativeChatView', () => {
vi.useRealTimers()
}
})
describe('structured turn status wiring', () => {
const userTurn = (id: string, text: string): NativeChatMessage => ({
id,
role: 'user',
blocks: [{ type: 'text', text }],
timestamp: 0,
source: 'transcript'
})
function rowProps(id: string): Record<string, unknown> {
return (renderedRow(id) as { props: Record<string, unknown> }).props
}
function workingIndicators(): ReactTestInstance[] {
return renderer!.root.findAll((node) => node.type === 'WorkingIndicator')
}
it('gives the live user turn a status row and drops the three-dot indicator', async () => {
const folded = [userTurn('u1', 'go')]
await render({ messages: folded, folded, structuredActivityUi: true, agentWorking: true })
const props = rowProps('u1')
expect(props.structuredActivityUi).toBe(true)
expect(props.turnStatus).toMatchObject({ thinking: true, workedSeconds: null })
expect(props.activeTurnIsWorking).toBe(true)
expect(workingIndicators()).toHaveLength(0)
})
it('keeps the bridge lane on the three-dot indicator with no turn status', async () => {
const folded = [userTurn('u1', 'go')]
await render({ messages: folded, folded, agentWorking: true })
const props = rowProps('u1')
expect(props.structuredActivityUi).toBe(false)
expect(props.turnStatus).toBeNull()
expect(props.activeTurnIsWorking).toBe(false)
expect(workingIndicators()).toHaveLength(1)
})
it('settles the finished turn to a tappable duration', async () => {
const folded = [userTurn('u1', 'go'), assistantTurn('a1', 'done')]
await render({ messages: folded, folded, structuredActivityUi: true, agentWorking: true })
expect(rowProps('u1').turnStatus).toMatchObject({ thinking: false, workedSeconds: null })
await update({ messages: folded, folded, structuredActivityUi: true, agentWorking: false })
const settled = rowProps('u1')
expect(settled.turnStatus).toMatchObject({ thinking: false })
expect((settled.turnStatus as { workedSeconds: number | null }).workedSeconds).toBeTypeOf(
'number'
)
expect(settled.onToggleTurn).toBeTypeOf('function')
expect(settled.activeTurnIsWorking).toBe(false)
})
it('hangs no status row on an assistant row', async () => {
const folded = [userTurn('u1', 'go'), assistantTurn('a1', 'done')]
await render({ messages: folded, folded, structuredActivityUi: true, agentWorking: true })
expect(rowProps('a1').turnStatus).toBeNull()
// The assistant row still belongs to the live turn, so its tool row stays visible.
expect(rowProps('a1').activeTurnIsWorking).toBe(true)
})
it('does not carry a running turn clock across chat surfaces', async () => {
vi.useFakeTimers()
try {
vi.setSystemTime(1_000)
const firstTab = [userTurn('u1', 'first')]
await render({
messages: firstTab,
folded: firstTab,
structuredActivityUi: true,
agentWorking: true,
sendSurfaceId: 'host\0worktree\0tab-a'
})
expect(rowProps('u1').turnStatus).toMatchObject({ startedAt: 1_000 })
vi.setSystemTime(12_000)
const secondTab = [userTurn('u2', 'second')]
await update({
messages: secondTab,
folded: secondTab,
structuredActivityUi: true,
agentWorking: true,
sendSurfaceId: 'host\0worktree\0tab-b'
})
expect(rowProps('u2').turnStatus).toMatchObject({ startedAt: 12_000 })
} finally {
vi.useRealTimers()
}
})
it('does not treat pre-user history as part of the live turn', async () => {
const history = [
assistantTurn('a0', 'before the first prompt'),
userTurn('u1', 'go'),
assistantTurn('a1', 'working')
]
await render({
messages: history,
folded: history,
structuredActivityUi: true,
agentWorking: true
})
expect(rowProps('a0').activeTurnIsWorking).toBe(false)
expect(rowProps('a1').activeTurnIsWorking).toBe(true)
})
})
})
+43 -43
View File
@@ -21,16 +21,16 @@ import {
type MobileNativeChatPendingItem
} from './mobile-native-chat-render-data'
import { useMobileNativeChatPinchGesture } from './use-mobile-native-chat-pinch-gesture'
import { useMobileNativeChatTurnDisclosure } from './use-mobile-native-chat-turn-disclosure'
import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus'
import { MobileAgentWorkingIndicator } from './MobileAgentWorkingIndicator'
import type { PendingNativeChatImage } from './mobile-native-chat-image-attachment'
import { MobileNativeChatComposer } from './MobileNativeChatComposer'
import { MobileNativeChatPromptCard } from './MobileNativeChatPromptCard'
import type { MobileChatPermission } from './mobile-native-chat-permission'
import type { MobileChatQuestion } from './mobile-native-chat-question'
import type { MobileNativeChatSessionOptionPickersProps } from './MobileNativeChatSessionOptionPickers'
import { MobileNativeChatMessage } from './MobileNativeChatMessage'
import { MobileNativeChatAsk } from './MobileNativeChatAsk'
import { MobileNativeChatPermission } from './MobileNativeChatPermission'
import type { MobileChatPermission } from './mobile-native-chat-permission'
import { MobileNativeChatQuestion } from './MobileNativeChatQuestion'
import { mobileChatQuestionKey, type MobileChatQuestion } from './mobile-native-chat-question'
import type { MobileNativeChatStatus } from './use-mobile-native-chat-session'
const INPUT_LOCK_SETTLE_MS = 600
@@ -49,6 +49,9 @@ type Props = {
/** Resolved agent for this chat; names the empty-state copy (desktop parity). */
agent?: string | null
agentWorking?: boolean
/** Structured lane: per-turn "Working for N" status plus live tool progress,
* replacing the bridge lane's static three-dot working row (desktop parity). */
structuredActivityUi?: boolean
/** Interrupt the agent mid-turn (shown as a Stop button on the working bar). */
onStop?: () => void
/** Live partial assistant text to show as an in-progress bubble, already gated
@@ -126,6 +129,7 @@ export function MobileNativeChatView({
error,
agent,
agentWorking,
structuredActivityUi = false,
onStop,
streaming,
hasMore,
@@ -252,6 +256,15 @@ export function MobileNativeChatView({
listRef.current?.scrollToIndex({ index, viewPosition: 0, animated: true })
}, [])
// Per-turn "Thinking / Working for N / Worked for N" rows. The structured lane
// owns them; the bridge lane keeps its three-dot indicator.
const turns = useMobileNativeChatTurnDisclosure({
messages: data,
enabled: structuredActivityUi,
isWorking: agentWorking === true,
scopeKey: sendSurfaceId
})
const renderItem = useCallback(
({ item, index }: { item: NativeChatMessage; index: number }) => (
<MobileNativeChatMessage
@@ -261,9 +274,12 @@ export function MobileNativeChatView({
messageIndex={index}
onScrollToMessage={onScrollToMessage}
onOpenFile={onOpenFile}
structuredActivityUi={structuredActivityUi}
onToggleTurn={turns.onToggleTurn}
{...turns.resolveRow(index, item)}
/>
),
[toolsExpanded, fontScale, onScrollToMessage, onOpenFile]
[toolsExpanded, fontScale, onScrollToMessage, onOpenFile, structuredActivityUi, turns]
)
const emptyState = mobileNativeChatEmptyState(status, agent ?? null, error)
@@ -337,6 +353,15 @@ export function MobileNativeChatView({
</Pressable>
) : null
}
ListFooterComponent={
turns.activeTurnIsUnanchored && turns.active ? (
<MobileNativeChatTurnStatus
startedAt={turns.active.startedAt}
thinking={turns.active.thinking}
workedSeconds={turns.active.workedSeconds}
/>
) : null
}
ListEmptyComponent={
emptyState ? (
<View style={styles.center}>
@@ -360,47 +385,22 @@ export function MobileNativeChatView({
) : null}
</GestureHandlerRootView>
)}
{/* Pending agent prompt: a structured AskUserQuestion wins, then a
heuristic permission, then a heuristic question. The controller owns
dismissal (it must survive this subtree unmounting on a view toggle);
`ask` arrives already nulled while dismissed. */}
{ask ? (
<MobileNativeChatAsk
key={askKey ?? 'ask'}
prompt={ask}
onAnswer={async (selections) => {
const accepted = (await onAnswerAsk?.(ask, selections)) ?? false
if (accepted) {
onDismissAsk?.()
}
return accepted
}}
onCancel={async () => {
const accepted = (await onCancelAsk?.()) ?? false
if (accepted) {
onDismissAsk?.()
}
return accepted
}}
/>
) : permission ? (
<MobileNativeChatPermission
key={JSON.stringify(permission)}
permission={permission}
onRespond={async (send) => (await onRespondPermission?.(send)) ?? false}
/>
) : question ? (
<MobileNativeChatQuestion
key={mobileChatQuestionKey(question)}
question={question}
onAnswer={async (text) => (await onAnswerQuestion?.(text)) ?? false}
/>
) : null}
<MobileNativeChatPromptCard
ask={ask}
askKey={askKey}
onDismissAsk={onDismissAsk}
onAnswerAsk={onAnswerAsk}
onCancelAsk={onCancelAsk}
permission={permission}
onRespondPermission={onRespondPermission}
question={question}
onAnswerQuestion={onAnswerQuestion}
/>
{/* Chrome row above the composer: the working indicator and the global
tool-calls expand/collapse toggle on the left, Stop in the far corner. */}
<View style={styles.chromeRow}>
<View style={styles.chromeLeft}>
{agentWorking ? <MobileAgentWorkingIndicator /> : null}
{agentWorking && !structuredActivityUi ? <MobileAgentWorkingIndicator /> : null}
<Pressable
style={({ pressed }) => [styles.chromeToggle, pressed && styles.pressed]}
onPress={() => setToolsExpanded((v) => !v)}
@@ -25,6 +25,8 @@ export type MobileNativeChatController = {
chatPending: MobileNativeChatPendingMessage[]
chatImagePreviewsByMessageId: Record<string, string[]>
nativeChatSession: ReturnType<typeof useMobileNativeChatSession>
/** Structured lane: drives the per-turn status row and live tool progress. */
nativeChatStructured: boolean
nativeChatAgentWorking: boolean
nativeChatStreamingText?: string
/** Agent mid-turn, regardless of whether chat is the visible view. */
@@ -80,6 +80,18 @@ export const styles = StyleSheet.create({
fontFamily: typography.monoFamily,
fontSize: MONO_SIZE
},
toolRunActive: {
flex: 1,
flexDirection: 'row',
alignItems: 'center',
gap: spacing.sm,
paddingVertical: 3
},
toolRunActiveLabel: {
flex: 1,
color: colors.textSecondary,
fontSize: typography.bodySize
},
toolRunBody: {
paddingLeft: spacing.sm,
borderLeftWidth: 2,
@@ -139,4 +139,76 @@ describe('mobile structured Codex launch', () => {
kind: 'unknown'
})
})
it.each(['structured_agent_session_unsupported', 'method_not_found'])(
'treats a top-level %s as a definitive refusal',
async (code) => {
const client = clientReturning(
{ ok: true, result: { supported: true } },
{ ok: false, error: { code, message: 'structured create unavailable' } }
)
await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({
kind: 'failed',
message: 'structured create unavailable'
})
}
)
it.each(['agent_session_operation_unknown', 'runtime_error', 'future_unknown_code'])(
'keeps a top-level %s outcome unknown',
async (code) => {
const client = clientReturning(
{ ok: true, result: { supported: true } },
{ ok: false, error: { code, message: 'create outcome ambiguous' } }
)
await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({
kind: 'unknown',
message: 'create outcome ambiguous'
})
}
)
it('treats an envelope unsupported refusal as definitive', async () => {
const client = clientReturning(
{ ok: true, result: { supported: true } },
{
ok: true,
result: {
ok: false,
refusal: {
code: 'structured_agent_session_unsupported',
message: 'structured create unavailable'
}
}
}
)
await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({
kind: 'failed',
message: 'structured create unavailable'
})
})
it.each(['agent_session_operation_unknown', 'agent_session_ownership_unknown', 'future_code'])(
'keeps an envelope %s refusal unknown',
async (code) => {
const client = clientReturning(
{ ok: true, result: { supported: true } },
{
ok: true,
result: {
ok: false,
refusal: { code, message: 'create outcome ambiguous' }
}
}
)
await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({
kind: 'unknown',
message: 'create outcome ambiguous'
})
}
)
})
@@ -2,6 +2,7 @@ import type {
AgentSessionAttachResult,
AgentSessionMutationResult
} from '../../../src/shared/agent-session-wire'
import { isDefinitiveAgentSessionCreateRefusal } from '../../../src/shared/agent-session-definitive-refusal'
import { structuredAgentSessionPayloadFingerprint } from '../../../src/shared/structured-agent-session-mutation'
import type { RpcClient } from '../transport/rpc-client'
import { structuredSessionOperationId } from './mobile-structured-agent-session-rpc'
@@ -66,6 +67,13 @@ function unknownCreateResult(error: unknown): MobileStructuredCodexLaunchResult
}
}
function classifyCreateRefusal(code: string, message: string): MobileStructuredCodexLaunchResult {
if (!isDefinitiveAgentSessionCreateRefusal(code)) {
return unknownCreateResult(new Error(message))
}
return { kind: 'failed', message: message || 'Could not open Codex chat.' }
}
export async function createMobileStructuredCodexSession(
client: RpcClient,
worktreeId: string
@@ -125,10 +133,7 @@ export async function createMobileStructuredCodexSession(
) {
return unknownCreateResult(new Error('The Codex chat result could not be confirmed.'))
}
if (response.error.code === 'agent_session_operation_unknown') {
return unknownCreateResult(new Error(response.error.message))
}
return { kind: 'failed', message: response.error.message || 'Could not open Codex chat.' }
return classifyCreateRefusal(response.error.code, response.error.message)
}
const result = response.result as AgentSessionMutationResult<AgentSessionAttachResult>
if (!result || typeof result !== 'object' || typeof result.ok !== 'boolean') {
@@ -142,10 +147,7 @@ export async function createMobileStructuredCodexSession(
) {
return unknownCreateResult(new Error('The Codex chat result could not be confirmed.'))
}
if (result.refusal.code === 'agent_session_operation_unknown') {
return unknownCreateResult(new Error(result.refusal.message))
}
return { kind: 'failed', message: result.refusal.message || 'Could not open Codex chat.' }
return classifyCreateRefusal(result.refusal.code, result.refusal.message)
}
if (
!result.value ||
@@ -10,9 +10,8 @@ import { useMobileNativeChatDrafts } from './use-mobile-native-chat-drafts'
import { useMobileNativeChatFileSearch } from './use-mobile-native-chat-file-search'
import { useMobileNativeChatMessageSend } from './use-mobile-native-chat-message-send'
import { mobileNativeChatStreamPreview } from './mobile-native-chat-streaming-gate'
import { useMobileNativeChatSession } from './use-mobile-native-chat-session'
import { useMobileNativeChatSessionOptionController } from './use-mobile-native-chat-session-option-controller'
import { useMobileStructuredAgentSession } from './use-mobile-structured-agent-session'
import { useMobileNativeChatSessionLane } from './use-mobile-native-chat-session-lane'
import { useMobileStructuredNativeChatSendBridge } from './use-mobile-structured-native-chat-send-bridge'
import { useMobileNativeChatPrompts } from './use-mobile-native-chat-prompts'
import { useMobileNativeChatStop } from './use-mobile-native-chat-stop'
@@ -82,27 +81,19 @@ export function useMobileNativeChatController(args: {
nativeChatTranscriptIsLocalReadable
})
const legacyNativeChatSession = useMobileNativeChatSession({
client,
sourceIdentity,
agent: activeChatStructured ? null : (activeChatResolution?.agent ?? null),
sessionId: activeChatStructured ? null : activeChatSessionId,
transcriptPath: activeChatStructured ? null : (activeChatResolution?.transcriptPath ?? null)
})
const structuredNativeChat = useMobileStructuredAgentSession({
client,
sessionId: activeChatStructured ? activeChatSessionId : null,
sourceIdentity,
enabled: showNativeChat,
// Holds are connection-scoped; dropping this on transport loss lets the hook
// reacquire the provider without clearing the cached transcript.
connected: connState === 'connected',
agent: activeChatStructured ? activeChatAgent : null,
onSendError
})
const nativeChatSession = activeChatStructured
? structuredNativeChat.session
: legacyNativeChatSession
const { structuredSession: structuredNativeChat, session: nativeChatSession } =
useMobileNativeChatSessionLane({
client,
structured: activeChatStructured,
agent: activeChatAgent,
resolvedAgent: activeChatResolution?.agent ?? null,
transcriptPath: activeChatResolution?.transcriptPath ?? null,
sessionId: activeChatSessionId,
sourceIdentity,
enabled: showNativeChat,
connState,
onSendError
})
const {
composerText: chatComposerText,
setComposerText: setChatComposerText,
@@ -303,6 +294,8 @@ export function useMobileNativeChatController(args: {
chatPending,
chatImagePreviewsByMessageId,
nativeChatSession,
/** Structured lane: drives the per-turn status row and live tool progress. */
nativeChatStructured: activeChatStructured,
nativeChatAgentWorking,
nativeChatStreamingText,
nativeChatStreamLive,
@@ -0,0 +1,59 @@
import type { RpcClient } from '../transport/rpc-client'
import type { ConnectionState } from '../transport/types'
import { useMobileNativeChatSession } from './use-mobile-native-chat-session'
import { useMobileStructuredAgentSession } from './use-mobile-structured-agent-session'
/** Mounts both transcript sources and hands back the one this tab's lane owns.
* Both hooks always run (hook order is fixed); the inactive lane is starved of
* its identity inputs rather than unmounted, so a lane flip keeps its cache. */
export function useMobileNativeChatSessionLane({
client,
structured,
agent,
resolvedAgent,
transcriptPath,
sessionId,
sourceIdentity,
enabled,
connState,
onSendError
}: {
client: RpcClient | null
structured: boolean
/** Agent id for the structured provider session. */
agent: string | null
/** Agent resolved from the terminal, for the bridge transcript reader. */
resolvedAgent: string | null
transcriptPath: string | null
sessionId: string | null
sourceIdentity: Parameters<typeof useMobileNativeChatSession>[0]['sourceIdentity']
enabled: boolean
connState: ConnectionState
onSendError: (message: string) => void
}): {
structuredSession: ReturnType<typeof useMobileStructuredAgentSession>
session: ReturnType<typeof useMobileNativeChatSession>
} {
const bridgeSession = useMobileNativeChatSession({
client,
sourceIdentity,
agent: structured ? null : resolvedAgent,
sessionId: structured ? null : sessionId,
transcriptPath: structured ? null : transcriptPath
})
const structuredSession = useMobileStructuredAgentSession({
client,
sessionId: structured ? sessionId : null,
sourceIdentity,
enabled,
// Holds are connection-scoped; dropping this on transport loss lets the hook
// reacquire the provider without clearing the cached transcript.
connected: connState === 'connected',
agent: structured ? agent : null,
onSendError
})
return {
structuredSession,
session: structured ? structuredSession.session : bridgeSession
}
}
@@ -168,7 +168,8 @@ export function useMobileNativeChatSessionOptions(args: {
models: activeModels(catalog, record),
record,
mode: 'live',
modelLabel: 'Model'
modelLabel: 'Model',
liveTransport: 'catalog'
})
}, [agent, catalog, scopeKey, version])
@@ -0,0 +1,153 @@
import { createElement } from 'react'
import { act, create, type ReactTestRenderer } from 'react-test-renderer'
import { afterEach, describe, expect, it, vi } from 'vitest'
import type { NativeChatMessage } from '../../../src/shared/native-chat-types'
import { useMobileNativeChatTurnDisclosure } from './use-mobile-native-chat-turn-disclosure'
function userMessage(id: string): NativeChatMessage {
return {
id,
role: 'user',
blocks: [{ type: 'text', text: id }],
timestamp: null,
source: 'transcript'
}
}
function Harness({
messages,
enabled,
isWorking = true,
scopeKey = 'host\0worktree\0tab-a'
}: {
messages: readonly NativeChatMessage[]
enabled: boolean
isWorking?: boolean
scopeKey?: string
}): React.JSX.Element {
const disclosure = useMobileNativeChatTurnDisclosure({
messages,
enabled,
isWorking,
scopeKey
})
return createElement('result', { disclosure })
}
describe('useMobileNativeChatTurnDisclosure', () => {
let renderer: ReactTestRenderer | null = null
afterEach(() => {
act(() => renderer?.unmount())
renderer = null
})
it('does not scan bridge-lane transcripts', () => {
const messages: NativeChatMessage[] = [
{
id: 'u1',
role: 'user',
blocks: [{ type: 'text', text: 'go' }],
timestamp: null,
source: 'transcript'
}
]
const findLastIndex = vi.spyOn(messages, 'findLastIndex')
const slice = vi.spyOn(messages, 'slice')
const filter = vi.spyOn(messages, 'filter')
const map = vi.spyOn(messages, 'map')
act(() => {
renderer = create(createElement(Harness, { messages, enabled: false }))
})
expect(findLastIndex).not.toHaveBeenCalled()
expect(slice).not.toHaveBeenCalled()
expect(filter).not.toHaveBeenCalled()
expect(map).not.toHaveBeenCalled()
})
it('keeps a settled turn handler stable for NUL-delimited scope keys', () => {
vi.useFakeTimers()
try {
vi.setSystemTime(1_000)
const messages: NativeChatMessage[] = [
{
id: 'u1',
role: 'user',
blocks: [{ type: 'text', text: 'go' }],
timestamp: null,
source: 'transcript'
}
]
act(() => {
renderer = create(createElement(Harness, { messages, enabled: true }))
})
vi.setSystemTime(6_000)
act(() => {
renderer?.update(createElement(Harness, { messages, enabled: true, isWorking: false }))
})
const first = renderer!.root.findByType('result').props.disclosure.resolveRow(0, messages[0])
const refreshed = [...messages]
act(() => {
renderer?.update(
createElement(Harness, { messages: refreshed, enabled: true, isWorking: false })
)
})
const second = renderer!.root
.findByType('result')
.props.disclosure.resolveRow(0, refreshed[0])
// The row carries the key; the handler itself lives on the hook and stays
// stable for the scope, so a re-render never disturbs a row's memo.
expect(first.turnKey).toBe('u1')
expect(second.turnKey).toBe('u1')
const firstHandler = renderer!.root.findByType('result').props.disclosure.onToggleTurn
expect(firstHandler).toBeTypeOf('function')
act(() => {
renderer?.update(
createElement(Harness, { messages: [...refreshed], enabled: true, isWorking: false })
)
})
expect(renderer!.root.findByType('result').props.disclosure.onToggleTurn).toBe(firstHandler)
} finally {
vi.useRealTimers()
}
})
it('keeps at most the latest 128 turns expanded', () => {
vi.useFakeTimers()
try {
let messages: NativeChatMessage[] = []
for (let index = 0; index < 129; index++) {
messages = messages.concat(userMessage(`u${index}`))
vi.setSystemTime(index * 2_000)
act(() => {
if (renderer) {
renderer.update(createElement(Harness, { messages, enabled: true }))
} else {
renderer = create(createElement(Harness, { messages, enabled: true }))
}
})
vi.setSystemTime(index * 2_000 + 1_000)
act(() => {
renderer?.update(createElement(Harness, { messages, enabled: true, isWorking: false }))
})
const disclosureNow = renderer!.root.findByType('result').props.disclosure
const row = disclosureNow.resolveRow(index, messages[index])
act(() => disclosureNow.onToggleTurn(row.turnKey))
}
const disclosure = renderer!.root.findByType('result').props.disclosure
const expanded = messages.filter(
(message, index) => disclosure.resolveRow(index, message).turnExpanded
)
expect(expanded).toHaveLength(128)
expect(disclosure.resolveRow(0, messages[0]).turnExpanded).toBe(false)
expect(disclosure.resolveRow(128, messages[128]).turnExpanded).toBe(true)
} finally {
vi.useRealTimers()
}
})
})
@@ -0,0 +1,126 @@
import { useCallback, useMemo, useState } from 'react'
import type { NativeChatMessage } from '../../../src/shared/native-chat-types'
import {
MOBILE_UNANCHORED_TURN_KEY,
useMobileNativeChatTurnStatus,
type NativeChatTurnStatus
} from './use-mobile-native-chat-turn-status'
const EMPTY_TURN_IDS: ReadonlySet<string> = new Set()
const EMPTY_TURN_KEYS: readonly undefined[] = []
const MAX_EXPANDED_TURNS = 128
export type MobileNativeChatTurnRow = {
turnStatus: NativeChatTurnStatus | null
turnExpanded: boolean
/** Set only on a settled turn — the one row that has activity to disclose. */
turnKey?: string
activeTurnIsWorking: boolean
}
/** Owns the transcript's per-turn status rows and their disclosure state, and
* resolves what one list row needs. Bridge-lane chats pass `enabled: false` and
* keep their single three-dot working indicator instead. */
export function useMobileNativeChatTurnDisclosure({
messages,
enabled,
isWorking,
scopeKey
}: {
messages: readonly NativeChatMessage[]
enabled: boolean
isWorking: boolean
/** Host/worktree/tab identity for timing and disclosure isolation. */
scopeKey: string
}): {
active: NativeChatTurnStatus | null
/** True when the live turn has no user message to hang its status row under. */
activeTurnIsUnanchored: boolean
onToggleTurn: (turnKey: string) => void
resolveRow: (index: number, message: NativeChatMessage) => MobileNativeChatTurnRow
} {
const turnStatuses = useMobileNativeChatTurnStatus({
messages,
enabled,
isWorking,
scopeKey
})
const [expandedTurns, setExpandedTurns] = useState<{
scopeKey: string
turnIds: ReadonlySet<string>
}>(() => ({ scopeKey, turnIds: new Set() }))
const expandedTurnIds =
expandedTurns.scopeKey === scopeKey ? expandedTurns.turnIds : EMPTY_TURN_IDS
const toggleExpandedTurn = useCallback(
(turnKey: string) => {
setExpandedTurns((current) => {
const next = new Set(current.scopeKey === scopeKey ? current.turnIds : [])
if (!next.delete(turnKey)) {
if (next.size >= MAX_EXPANDED_TURNS) {
const oldest = next.values().next().value
if (oldest) {
next.delete(oldest)
}
}
next.add(turnKey)
}
return { scopeKey, turnIds: next }
})
},
[scopeKey]
)
// Resolve each row's turn boundary once — a findLast per row is quadratic on a
// long transcript.
const turnKeys = useMemo(() => {
if (!enabled) {
return EMPTY_TURN_KEYS
}
let turnKey: string | undefined
return messages.map((message) => {
if (message.role === 'user') {
turnKey = message.id
}
return turnKey
})
}, [enabled, messages])
const { active, activeTurnKey, completedByTurn } = turnStatuses
const resolveRow = useCallback(
(index: number, message: NativeChatMessage): MobileNativeChatTurnRow => {
const turnKey = turnKeys[index]
const turnStatus =
!enabled || message.role !== 'user'
? null
: turnKey === activeTurnKey
? active
: turnKey
? (completedByTurn[turnKey] ?? null)
: null
return {
turnStatus,
turnExpanded: turnKey ? expandedTurnIds.has(turnKey) : false,
// Why: the key travels and the row calls one stable handler with it. A
// closure per row would be a new identity every render of a streaming
// transcript, defeating the row's memo; caching one per turn would mean
// writing a ref during render, which react-freeze can discard.
turnKey: turnKey && turnStatus?.workedSeconds != null ? turnKey : undefined,
// With no user boundary at all, the session's working state stays authoritative.
activeTurnIsWorking:
enabled &&
isWorking &&
(turnKey === activeTurnKey ||
(turnKey === undefined && activeTurnKey === MOBILE_UNANCHORED_TURN_KEY))
}
},
[turnKeys, enabled, activeTurnKey, active, completedByTurn, expandedTurnIds, isWorking]
)
return {
active,
/** Stable for a given chat scope, so it never disturbs a row's memo. */
onToggleTurn: toggleExpandedTurn,
activeTurnIsUnanchored:
enabled && active != null && activeTurnKey === MOBILE_UNANCHORED_TURN_KEY,
resolveRow
}
}
@@ -0,0 +1,105 @@
import { useEffect, useMemo, useRef, useState } from 'react'
import type { NativeChatMessage } from '../../../src/shared/native-chat-types'
import {
nativeChatTurnHasResponse,
reduceNativeChatTurnTiming,
selectNativeChatTurnStatuses,
type NativeChatTurnStatus,
type NativeChatTurnTimingByTurn
} from '../../../src/shared/native-chat-turn-status'
export type { NativeChatTurnStatus }
export const MOBILE_UNANCHORED_TURN_KEY = '__unanchored__'
const EMPTY_TURN_TIMING_BY_TURN: NativeChatTurnTimingByTurn = Object.freeze({})
type ScopedTurnTiming = {
scopeKey: string
timingByTurn: NativeChatTurnTimingByTurn
}
/** Per-turn "Thinking / Working for N / Worked for N" timing, on the same shared
* state machine the desktop renderer uses so the two surfaces stamp turns alike. */
export function useMobileNativeChatTurnStatus({
messages,
enabled,
isWorking,
workingStartedAt,
scopeKey
}: {
messages: readonly NativeChatMessage[]
enabled: boolean
isWorking: boolean
workingStartedAt?: number | null
/** Host/worktree/tab identity. Timings never carry across chat surfaces. */
scopeKey: string
}): {
active: NativeChatTurnStatus | null
completedByTurn: Readonly<Record<string, NativeChatTurnStatus>>
activeTurnKey: string
} {
const latestUserIndex = enabled
? messages.findLastIndex((message) => message.role === 'user')
: -1
const hasCurrentTurnResponse = enabled && nativeChatTurnHasResponse(messages, latestUserIndex)
const latestUserId = latestUserIndex !== -1 ? (messages[latestUserIndex]?.id ?? null) : null
const activeTurnKey = latestUserId ?? MOBILE_UNANCHORED_TURN_KEY
const [scopedTiming, setScopedTiming] = useState<ScopedTurnTiming>(() => ({
scopeKey,
timingByTurn: {}
}))
// Do not expose the previous surface's state during the render before the
// timing effect adopts the new scope, or scan it while this UI is disabled.
const timingByTurn =
enabled && scopedTiming.scopeKey === scopeKey
? scopedTiming.timingByTurn
: EMPTY_TURN_TIMING_BY_TURN
// An accepted send renders as `pending-N` until the transcript echo lands under
// its real id. That is one turn under two keys, so the clock must survive the swap.
const previousActiveTurn = useRef<{ scopeKey: string; turnKey: string } | null>(null)
useEffect(() => {
if (!enabled) {
return
}
const validTurnKeys = new Set(
messages.filter((message) => message.role === 'user').map((message) => message.id)
)
const previousActiveTurnKey =
previousActiveTurn.current?.scopeKey === scopeKey
? previousActiveTurn.current.turnKey
: undefined
previousActiveTurn.current = { scopeKey, turnKey: activeTurnKey }
setScopedTiming((current) => {
const currentTiming =
current.scopeKey === scopeKey ? current.timingByTurn : EMPTY_TURN_TIMING_BY_TURN
const nextTiming = reduceNativeChatTurnTiming(currentTiming, {
activeTurnKey,
previousActiveTurnKey,
validTurnKeys,
isWorking,
workingStartedAt,
now: Date.now()
})
return current.scopeKey === scopeKey && nextTiming === currentTiming
? current
: { scopeKey, timingByTurn: nextTiming }
})
}, [activeTurnKey, enabled, isWorking, messages, scopeKey, workingStartedAt])
// Why: the selection rebuilds its status objects on every call, and a streaming
// turn re-renders ~20x/s. Without this, every settled turn's row gets fresh
// props each tick and the memoized message rows all re-render.
const turnIsWorking = enabled && isWorking
const statuses = useMemo(
() =>
selectNativeChatTurnStatuses(timingByTurn, {
activeTurnKey,
isWorking: turnIsWorking,
workingStartedAt,
hasCurrentTurnResponse
}),
[timingByTurn, activeTurnKey, turnIsWorking, workingStartedAt, hasCurrentTurnResponse]
)
return { ...statuses, activeTurnKey }
}
@@ -137,14 +137,17 @@ describe('mobile + Codex tab creation routing', () => {
expect(scope.setActiveSessionTabId).toHaveBeenCalledWith('terminal-tab-1')
})
it('falls back to a terminal when structured creation is refused', async () => {
it('falls back to a terminal when structured creation is definitively refused', async () => {
const client = clientReturning(
{ ok: true, result: { supported: true } },
{
ok: true,
result: {
ok: false,
refusal: { code: 'agent_session_ownership_unknown', message: 'provider unavailable' }
refusal: {
code: 'structured_agent_session_unsupported',
message: 'provider unavailable'
}
}
},
terminalCreateResponse()
@@ -228,4 +231,34 @@ describe('mobile + Codex tab creation routing', () => {
expect(scope.setCreateError).toHaveBeenCalledWith('still unknown')
expect(scope.showToast).toHaveBeenCalledWith('still unknown', 1800)
})
it.each(['agent_session_operation_unknown', 'runtime_error', 'future_unknown_code'])(
'does not create a legacy sibling after a top-level %s response',
async (code) => {
const client = clientReturning(
{ ok: true, result: { supported: true } },
{ ok: false, error: { code, message: 'create outcome ambiguous' } }
)
const scope = createScope(client)
let actions: ReturnType<typeof useMobileSessionTerminalCreateActions> | undefined
function Harness() {
actions = useMobileSessionTerminalCreateActions(scope as never)
return null
}
await act(async () => {
renderer = create(createElement(Harness))
})
await act(async () => {
await actions?.handleCreateTerminal('codex')
})
const sendRequest = client.sendRequest as unknown as ReturnType<typeof vi.fn>
expect(sendRequest.mock.calls.map(([method]) => method)).toEqual([
'agentSession.createSupport',
'agentSession.create'
])
expect(scope.setCreateError).toHaveBeenCalledWith('create outcome ambiguous')
expect(scope.showToast).toHaveBeenCalledWith('create outcome ambiguous', 1800)
}
)
})
+5
View File
@@ -141,6 +141,10 @@
"bench:main-thread-jank": "pnpm run ensure:electron-runtime && node tests/tools/benchmarks/main-thread-jank-bench.mjs",
"bench:worktree-deletion": "node tests/tools/benchmarks/worktree-deletion-dev-bench.mjs",
"bench:zustand-selector-fanout": "node config/scripts/zustand-selector-fanout-benchmark.mjs",
"bench:agent-inspection-cadence": "node config/scripts/agent-inspection-cadence-batching-benchmark.mjs",
"bench:renderer-quadratic-scans": "node config/scripts/renderer-quadratic-scan-benchmark.mjs",
"bench:session-write-hot-path": "node config/scripts/session-write-hot-path-benchmark.mjs",
"bench:terminal-partial-escape-tail": "node config/scripts/terminal-partial-escape-tail-benchmark.mjs",
"bench:worktree-refresh-churn": "node --disable-warning=MODULE_TYPELESS_PACKAGE_JSON config/scripts/worktree-refresh-churn-benchmark.mjs",
"bench:multi-workspace-typing": "pnpm run ensure:electron-runtime && node config/scripts/run-multi-workspace-typing-bench.mjs",
"bench:ai-vault-typing": "pnpm run ensure:electron-runtime && node config/scripts/run-ai-vault-typing-bench.mjs",
@@ -153,6 +157,7 @@
"repro:live-remote-realistic-freeze": "node config/scripts/live-remote-realistic-freeze-repro.mjs"
},
"dependencies": {
"@anthropic-ai/claude-agent-sdk": "0.3.251",
"@electron-toolkit/preload": "^3.0.2",
"@electron-toolkit/utils": "^4.0.0",
"@floating-ui/dom": "1.7.6",
+117 -16
View File
@@ -122,6 +122,9 @@ importers:
.:
dependencies:
'@anthropic-ai/claude-agent-sdk':
specifier: 0.3.251
version: 0.3.251(@anthropic-ai/sdk@0.122.0(zod@4.5.4))(@modelcontextprotocol/sdk@1.30.0(supports-color@7.2.0)(zod@4.5.4))(zod@4.5.4)
'@electron-toolkit/preload':
specifier: ^3.0.2
version: 3.0.2(electron@43.4.1(supports-color@7.2.0))
@@ -535,6 +538,23 @@ packages:
'@antfu/install-pkg@1.1.0':
resolution: {integrity: sha512-MGQsmw10ZyI+EJo45CdSER4zEb+p31LpDAFp2Z3gkSd1yqVZGi0Ebx++YTEMonJy4oChEMLsxZ64j8FH6sSqtQ==}
'@anthropic-ai/claude-agent-sdk@0.3.251':
resolution: {integrity: sha512-DqSi8mH2tQYRlVV0G+lJnQ/WbjJZ/a+8cJ3vPuYoqh8esIIvXHm1ZOXV1UPGsFYRnbBytEoiSGitguEXd+sQ+Q==}
engines: {node: '>=18.0.0'}
peerDependencies:
'@anthropic-ai/sdk': '>=0.93.0'
'@modelcontextprotocol/sdk': ^1.29.0
zod: ^4.0.0
'@anthropic-ai/sdk@0.122.0':
resolution: {integrity: sha512-GGPNftt0caaz9MDlmNQGHX8855Ojaduyy5pm9Sm1h7HalCn0cWNb5/bweadJF+4yzbal+QL6ztBa09WAAOzLmQ==}
hasBin: true
peerDependencies:
zod: ^3.25.0 || ^4.0.0
peerDependenciesMeta:
zod:
optional: true
'@babel/code-frame@7.29.7':
resolution: {integrity: sha512-Aup7aUOfpbAUg2ROOJN6Iw5f9DMBlzu0mIkm/malLQFN/YQgO48wCj0Kxa3sEHJvPVFg7siR+qRInwXd2qhQKw==}
engines: {node: '>=6.9.0'}
@@ -2628,6 +2648,9 @@ packages:
resolution: {integrity: sha512-tlqY9xq5ukxTUZBmoOp+m61cqwQD5pHJtFY3Mn8CA8ps6yghLH/Hw8UPdqg4OLmFW3IFlcXnQNmo/dh8HzXYIQ==}
engines: {node: '>=18'}
'@stablelib/base64@1.0.1':
resolution: {integrity: sha512-1bnPQqSxSuc3Ii6MhBysoWCg58j97aUjuCSZrGSmDxNqtytIi0k8utUenAwTZN4V5mXXYGsVUI9zeBqy+jBOSQ==}
'@stablyai/playwright-base@2.1.14':
resolution: {integrity: sha512-/iAgMW5tC0ETDo3mFyTzszRrD7rGFIT4fgDgtZxqa9vPhiTLix/1+GeOOBNY0uS+XRLFY0Uc/irsC3XProL47g==}
engines: {node: '>=18'}
@@ -4493,6 +4516,9 @@ packages:
resolution: {integrity: sha512-7MptL8U0cqcFdzIzwOTHoilX9x5BrNqye7Z/LuC7kCMRio1EMSyqRK3BEAUD7sXRq4iT4AzTVuZdhgQ2TCvYLg==}
engines: {node: '>=8.6.0'}
fast-sha256@1.3.0:
resolution: {integrity: sha512-n11RGP/lrWEFI/bWdygLxhI+pVeo1ZYIVwvvPkW7azl/rOy+F3HYRZ2K5zeE9mmkhQppyv9sQFx0JM9UabnpPQ==}
fast-string-truncated-width@3.0.3:
resolution: {integrity: sha512-0jjjIEL6+0jag3l2XWWizO64/aZVtpiGE3t0Zgqxv0DPuxiMjvB3M24fCyhZUO4KomJQPj3LTSUnDP3GpdwC0g==}
@@ -5039,6 +5065,10 @@ packages:
json-parse-even-better-errors@2.3.1:
resolution: {integrity: sha512-xyFwyhro/JEof6Ghe2iz2NcXoj2sloNsWr/XsERDK/oiPCfaNhl5ONfp+jQdAZRQQ0IJWNzH9zIZF7li91kh2w==}
json-schema-to-ts@3.1.1:
resolution: {integrity: sha512-+DWg8jCJG2TEnpy7kOm/7/AxaYoaRbjVB4LFZLySZlWn8exGs3A4OLJR966cVvU26N7X9TWxl+Jsw7dzAqKT6g==}
engines: {node: '>=16'}
json-schema-traverse@1.0.0:
resolution: {integrity: sha512-NM8/P9n3XjXhIZn1lLhkFaACTOURQXjWhV4BA/RnOv8xvgqtqpAX9IO4mRQxSx1Rlo4tqzeqb0sOlruaOy3dug==}
@@ -6425,6 +6455,9 @@ packages:
stackback@0.0.2:
resolution: {integrity: sha512-1XMJE5fQo1jGH6Y/7ebnwPOBEkIEnT4QF32d5R1+VXdXveM0IBMJt8zfaxX1P3QhVwrYe+576+jkANtSS2mBbw==}
standardwebhooks@1.1.1:
resolution: {integrity: sha512-bCbX9ZEyFkWPsRz7Bl3NuQUJohmwGSev/yhr7vhaGPlc4AfIrspIRa6cPTBuI1ItmrTDJ4d/S2hCsfe4+vQGnQ==}
stat-mode@1.0.0:
resolution: {integrity: sha512-jH9EhtKIjuXZ2cWxmXS8ZP80XyC3iasQxMDV8jzhNJpfDb7VbQLVW4Wvsxz9QZvzV+G4YoSfBUVKDOyxLzi/sg==}
engines: {node: '>= 6'}
@@ -6608,6 +6641,9 @@ packages:
truncate-utf8-bytes@1.0.2:
resolution: {integrity: sha512-95Pu1QXQvruGEhv62XCMO3Mm90GscOCClvrIUwCM0PYOXK3kaF3l3sIHxx71ThJfcbM2O5Au6SO3AWCSEfW4mQ==}
ts-algebra@2.0.0:
resolution: {integrity: sha512-FPAhNPFMrkwz76P7cdjdmiShwMynZYN6SgOujD1urY4oNm80Ou9oMdmbR45LotcKOXoy7wSmHkRFE6Mxbrhefw==}
ts-dedent@2.2.0:
resolution: {integrity: sha512-q5W7tVM71e2xjHZTlgfTDoPF/SmqKG5hddq9SzR49CH2hayqRKJtQ4mtRlSxKaJlR/+9rEM+mnBHf7I2/BQcpQ==}
engines: {node: '>=6.10'}
@@ -7007,6 +7043,16 @@ packages:
zwitch@2.0.4:
resolution: {integrity: sha512-bXE4cR/kVZhKZX/RjPEflHaKVhUVl85noU3v6b8apfQEc1x4A+zBxjZ4lN8LqGd6WZ3dl98pY4o717VFmoPp+A==}
ignoredOptionalDependencies:
- '@anthropic-ai/claude-agent-sdk-darwin-arm64'
- '@anthropic-ai/claude-agent-sdk-darwin-x64'
- '@anthropic-ai/claude-agent-sdk-linux-arm64'
- '@anthropic-ai/claude-agent-sdk-linux-arm64-musl'
- '@anthropic-ai/claude-agent-sdk-linux-x64'
- '@anthropic-ai/claude-agent-sdk-linux-x64-musl'
- '@anthropic-ai/claude-agent-sdk-win32-arm64'
- '@anthropic-ai/claude-agent-sdk-win32-x64'
snapshots:
'@adobe/css-tools@4.5.0': {}
@@ -7016,6 +7062,19 @@ snapshots:
package-manager-detector: 1.6.0
tinyexec: 1.1.2
'@anthropic-ai/claude-agent-sdk@0.3.251(@anthropic-ai/sdk@0.122.0(zod@4.5.4))(@modelcontextprotocol/sdk@1.30.0(supports-color@7.2.0)(zod@4.5.4))(zod@4.5.4)':
dependencies:
'@anthropic-ai/sdk': 0.122.0(zod@4.5.4)
'@modelcontextprotocol/sdk': 1.30.0(supports-color@7.2.0)(zod@4.5.4)
zod: 4.5.4
'@anthropic-ai/sdk@0.122.0(zod@4.5.4)':
dependencies:
json-schema-to-ts: 3.1.1
standardwebhooks: 1.1.1
optionalDependencies:
zod: 4.5.4
'@babel/code-frame@7.29.7':
dependencies:
'@babel/helper-validator-identifier': 7.29.7
@@ -7669,6 +7728,28 @@ snapshots:
dependencies:
'@chevrotain/types': 11.1.2
'@modelcontextprotocol/sdk@1.30.0(supports-color@7.2.0)(zod@4.5.4)':
dependencies:
'@hono/node-server': 2.1.0(hono@4.13.0)
ajv: 8.20.0
ajv-formats: 3.0.1(ajv@8.20.0)
content-type: 1.0.5
cors: 2.8.6
cross-spawn: 7.0.6
eventsource: 3.0.7
eventsource-parser: 3.0.8
express: 5.2.1(supports-color@7.2.0)
express-rate-limit: 8.5.2(express@5.2.1(supports-color@7.2.0))
hono: 4.13.0
jose: 6.2.3
json-schema-typed: 8.0.2
pkce-challenge: 5.0.1
raw-body: 3.0.2
zod: 4.5.4
zod-to-json-schema: 3.25.2(zod@4.5.4)
transitivePeerDependencies:
- supports-color
'@modelcontextprotocol/sdk@1.30.0(zod@3.25.76)':
dependencies:
'@hono/node-server': 2.1.0(hono@4.13.0)
@@ -7679,8 +7760,8 @@ snapshots:
cross-spawn: 7.0.6
eventsource: 3.0.7
eventsource-parser: 3.0.8
express: 5.2.1
express-rate-limit: 8.5.2(express@5.2.1)
express: 5.2.1(supports-color@7.2.0)
express-rate-limit: 8.5.2(express@5.2.1(supports-color@7.2.0))
hono: 4.13.0
jose: 6.2.3
json-schema-typed: 8.0.2
@@ -8917,6 +8998,8 @@ snapshots:
'@sindresorhus/merge-streams@4.0.0': {}
'@stablelib/base64@1.0.1': {}
'@stablyai/playwright-base@2.1.14(@playwright/test@1.59.1)(zod@4.5.4)':
dependencies:
'@playwright/test': 1.59.1
@@ -9923,7 +10006,7 @@ snapshots:
bluebird@3.7.2: {}
body-parser@2.3.0:
body-parser@2.3.0(supports-color@7.2.0):
dependencies:
bytes: 3.1.2
content-type: 2.0.0
@@ -10779,15 +10862,15 @@ snapshots:
exponential-backoff@3.1.3: {}
express-rate-limit@8.5.2(express@5.2.1):
express-rate-limit@8.5.2(express@5.2.1(supports-color@7.2.0)):
dependencies:
express: 5.2.1
express: 5.2.1(supports-color@7.2.0)
ip-address: 10.4.0
express@5.2.1:
express@5.2.1(supports-color@7.2.0):
dependencies:
accepts: 2.0.0
body-parser: 2.3.0
body-parser: 2.3.0(supports-color@7.2.0)
content-disposition: 1.1.0
content-type: 1.0.5
cookie: 0.7.2
@@ -10797,7 +10880,7 @@ snapshots:
encodeurl: 2.0.0
escape-html: 1.0.3
etag: 1.8.1
finalhandler: 2.1.1
finalhandler: 2.1.1(supports-color@7.2.0)
fresh: 2.0.0
http-errors: 2.0.1
merge-descriptors: 2.0.0
@@ -10808,9 +10891,9 @@ snapshots:
proxy-addr: 2.0.7
qs: 6.15.2
range-parser: 1.2.1
router: 2.2.0
send: 1.2.1
serve-static: 2.2.1
router: 2.2.0(supports-color@7.2.0)
send: 1.2.1(supports-color@7.2.0)
serve-static: 2.2.1(supports-color@7.2.0)
statuses: 2.0.2
type-is: 2.1.0
vary: 1.1.2
@@ -10831,6 +10914,8 @@ snapshots:
merge2: 1.4.1
micromatch: 4.0.8
fast-sha256@1.3.0: {}
fast-string-truncated-width@3.0.3: {}
fast-string-width@3.0.2:
@@ -10871,7 +10956,7 @@ snapshots:
dependencies:
to-regex-range: 5.0.1
finalhandler@2.1.1:
finalhandler@2.1.1(supports-color@7.2.0):
dependencies:
debug: 4.4.3(supports-color@7.2.0)
encodeurl: 2.0.0
@@ -11445,6 +11530,11 @@ snapshots:
json-parse-even-better-errors@2.3.1: {}
json-schema-to-ts@3.1.1:
dependencies:
'@babel/runtime': 7.29.7
ts-algebra: 2.0.0
json-schema-traverse@1.0.0: {}
json-schema-typed@8.0.2: {}
@@ -13037,7 +13127,7 @@ snapshots:
points-on-curve: 0.2.0
points-on-path: 0.2.1
router@2.2.0:
router@2.2.0(supports-color@7.2.0):
dependencies:
debug: 4.4.3(supports-color@7.2.0)
depd: 2.0.0
@@ -13080,7 +13170,7 @@ snapshots:
semver@7.8.1: {}
send@1.2.1:
send@1.2.1(supports-color@7.2.0):
dependencies:
debug: 4.4.3(supports-color@7.2.0)
encodeurl: 2.0.0
@@ -13107,12 +13197,12 @@ snapshots:
transitivePeerDependencies:
- typescript
serve-static@2.2.1:
serve-static@2.2.1(supports-color@7.2.0):
dependencies:
encodeurl: 2.0.0
escape-html: 1.0.3
parseurl: 1.3.3
send: 1.2.1
send: 1.2.1(supports-color@7.2.0)
transitivePeerDependencies:
- supports-color
@@ -13267,6 +13357,11 @@ snapshots:
stackback@0.0.2: {}
standardwebhooks@1.1.1:
dependencies:
'@stablelib/base64': 1.0.1
fast-sha256: 1.3.0
stat-mode@1.0.0: {}
state-local@1.0.7: {}
@@ -13437,6 +13532,8 @@ snapshots:
dependencies:
utf8-byte-length: 1.0.5
ts-algebra@2.0.0: {}
ts-dedent@2.2.0: {}
ts-morph@26.0.0:
@@ -13786,6 +13883,10 @@ snapshots:
dependencies:
zod: 3.25.76
zod-to-json-schema@3.25.2(zod@4.5.4):
dependencies:
zod: 4.5.4
zod@3.25.76: {}
zod@4.5.4: {}
+14
View File
@@ -12,6 +12,20 @@ minimumReleaseAgeExclude:
- zod@4.5.4
shamefullyHoist: true
# Orca always launches the user's own resolved Claude CLI via
# pathToClaudeCodeExecutable, so the SDK's bundled ~95 MB-per-platform CLI
# binaries must never be installed. Excluding them is what makes the path
# override mandatory rather than merely preferred.
ignoredOptionalDependencies:
- '@anthropic-ai/claude-agent-sdk-darwin-arm64'
- '@anthropic-ai/claude-agent-sdk-darwin-x64'
- '@anthropic-ai/claude-agent-sdk-linux-arm64'
- '@anthropic-ai/claude-agent-sdk-linux-arm64-musl'
- '@anthropic-ai/claude-agent-sdk-linux-x64'
- '@anthropic-ai/claude-agent-sdk-linux-x64-musl'
- '@anthropic-ai/claude-agent-sdk-win32-arm64'
- '@anthropic-ai/claude-agent-sdk-win32-x64'
supportedArchitectures:
os:
- current
@@ -127,10 +127,6 @@ __orca_osc133_precmd() {
unset __orca_in_command
fi
printf "\033]133;A\007"
# Why: emit the shell-ready marker here (not a trailing PROMPT_COMMAND entry)
# so a framework that must be last in PROMPT_COMMAND — bash-preexec — is not
# displaced by one of Orca's own hooks.
[[ -n "$__orca_ready_marker" ]] && printf "\033]777;orca-shell-ready\007"
return "$exit_code"
}
__orca_osc133_preexec() {
@@ -188,6 +184,11 @@ __orca_osc133_epilogue() {
unset __orca_in_prompt_command
__orca_adopt_outer_debug_trap
trap '__orca_osc133_preexec' DEBUG
# Readline renders PS1 after entering raw mode; prompt hooks still run in cooked mode.
if [[ -n "$__orca_ready_marker" ]]; then
PS1="${PS1-}"'\[\e]777;orca-shell-ready\a\]'
__orca_ready_marker=""
fi
}
__orca_normalize_prompt_command_part() {
local __orca_value="$1" __orca_output_name="$2" __orca_character __orca_chunk
+17 -1
View File
@@ -4,6 +4,7 @@ import type { AutomationPrecheck, AutomationPrecheckResult } from '../../shared/
import { MAX_AUTOMATION_PRECHECK_OUTPUT_CHARS } from '../../shared/automation-precheck'
import { getSshConnectionManager } from '../ipc/ssh'
import { shellEscape } from '../ssh/ssh-connection-utils'
import { admitSelfInitiatedTreeKill } from '../own-chromium-tree-kill-guard'
type AutomationPrecheckExecutionTarget =
| {
@@ -73,7 +74,10 @@ function failedPrecheckResult(
})
}
function killLocalPrecheckProcessTree(child: ChildProcess): ReturnType<typeof setTimeout> | null {
/** Exported for the refusal-fallback test; the timeout path is otherwise unreachable. */
export function killLocalPrecheckProcessTree(
child: ChildProcess
): ReturnType<typeof setTimeout> | null {
const pid = child.pid
if (!pid) {
child.kill()
@@ -81,6 +85,18 @@ function killLocalPrecheckProcessTree(child: ChildProcess): ReturnType<typeof se
}
if (process.platform === 'win32') {
if (
!admitSelfInitiatedTreeKill({
pid,
site: 'automation-precheck-timeout',
scope: 'win-taskkill-tree'
})
) {
// Refusal blocks the tree walk, not the termination: killing the root by
// handle cannot reach a recycled pid, and a timed-out precheck must stop.
child.kill()
return null
}
try {
// Why: shell prechecks can launch child processes; taskkill walks the
// Windows process tree so timeout means the command is actually stopped.
@@ -1,210 +0,0 @@
import { runInNewContext } from 'node:vm'
import { describe, expect, it } from 'vitest'
import { ANTI_DETECTION_SCRIPT } from './anti-detection'
type PermissionQueryResult = EventTarget & {
state: string
onchange: EventListener | null
marker: string
}
type PermissionStatusConstructor = {
new (): PermissionQueryResult
prototype: PermissionQueryResult
}
type AntiDetectionContext = {
Notification: {
permission: string
requestPermission: (callback?: (permission: string) => void) => Promise<string>
}
PermissionStatus: PermissionStatusConstructor
dispatchPermissionChange: (name: string) => void
navigator: {
permissions: {
query: (descriptor: { name: string }) => Promise<PermissionQueryResult>
}
}
}
function createContext(args: {
nativeNotificationPermission: string
requestedNotificationPermission: string
rejectedPermissions?: string[]
}): AntiDetectionContext & Record<string, unknown> {
class PermissionStatus extends EventTarget {
#state = 'denied'
#onchange: EventListener | null = null
marker = 'real-status'
get state(): string {
return this.#state
}
get onchange(): EventListener | null {
return this.#onchange
}
set onchange(listener: EventListener | null) {
if (this.#onchange) {
super.removeEventListener('change', this.#onchange)
}
this.#onchange = typeof listener === 'function' ? listener : null
if (this.#onchange) {
super.addEventListener('change', this.#onchange)
}
}
}
const statuses = new Map<string, PermissionStatus[]>()
const rejectedPermissions = new Set(args.rejectedPermissions)
class Permissions {
query(descriptor: { name: string }): Promise<PermissionStatus> {
if (rejectedPermissions.has(descriptor.name)) {
return Promise.reject(new Error('Unsupported permission'))
}
const status = new PermissionStatus()
const permissionStatuses = statuses.get(descriptor.name) ?? []
permissionStatuses.push(status)
statuses.set(descriptor.name, permissionStatuses)
return Promise.resolve(status)
}
}
const Notification = {
permission: args.nativeNotificationPermission,
requestPermission(callback?: (permission: string) => void): Promise<string> {
callback?.(args.requestedNotificationPermission)
return Promise.resolve(args.requestedNotificationPermission)
}
}
Object.defineProperty(Notification, 'permission', {
configurable: true,
get: () => args.nativeNotificationPermission
})
return {
Date,
Event,
EventTarget,
Object,
Promise,
Set,
performance: { now: () => 0 },
// Why: the script's Firefox gate reads navigator.userAgent, so a non-Firefox UA keeps these
// tests on the ordinary-page path where the PermissionStatus override applies.
window: { chrome: {} },
navigator: {
userAgent:
'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/150.0.0.0 Safari/537.36',
plugins: [],
languages: [],
permissions: new Permissions()
},
Permissions,
PermissionStatus,
Notification,
dispatchPermissionChange(name: string): void {
for (const status of statuses.get(name) ?? []) {
status.dispatchEvent(new Event('change'))
}
}
} as AntiDetectionContext & Record<string, unknown>
}
describe('ANTI_DETECTION_SCRIPT — PermissionStatus', () => {
it('keeps an existing notification status current after permission changes', async () => {
const context = createContext({
nativeNotificationPermission: 'denied',
requestedNotificationPermission: 'granted'
})
runInNewContext(ANTI_DETECTION_SCRIPT, context)
const status = await context.navigator.permissions.query({ name: 'notifications' })
expect(context.Notification.permission).toBe('default')
expect(status.state).toBe('prompt')
await expect(context.Notification.requestPermission()).resolves.toBe('granted')
expect(context.Notification.permission).toBe('granted')
expect(status.state).toBe('granted')
})
it('preserves native PermissionStatus identity and methods', async () => {
const context = createContext({
nativeNotificationPermission: 'denied',
requestedNotificationPermission: 'granted'
})
runInNewContext(ANTI_DETECTION_SCRIPT, context)
const status = await context.navigator.permissions.query({ name: 'camera' })
const expectedSource = Function.prototype.toString.call(
context.PermissionStatus.prototype.addEventListener
)
expect(status).toBeInstanceOf(context.PermissionStatus)
expect(status.state).toBe('prompt')
expect(status.constructor.name).toBe('PermissionStatus')
expect(status.marker).toBe('real-status')
expect(status.addEventListener.name).toBe('addEventListener')
expect(status.addEventListener).toBe(status.addEventListener)
expect(Function.prototype.toString.call(status.addEventListener)).toBe(expectedSource)
expect(expectedSource).toContain('addEventListener')
})
it('delivers change events through the returned status with the overridden state', async () => {
const context = createContext({
nativeNotificationPermission: 'denied',
requestedNotificationPermission: 'granted'
})
runInNewContext(ANTI_DETECTION_SCRIPT, context)
const status = await context.navigator.permissions.query({ name: 'notifications' })
const events: { receiver: EventTarget; target: EventTarget | null; state: string }[] = []
const recordEvent = function (this: EventTarget, event: Event): void {
events.push({
receiver: this,
target: event.target,
state: (event.target as PermissionQueryResult).state
})
}
status.addEventListener('change', recordEvent)
expect(() => {
status.onchange = function (this: EventTarget, event): void {
recordEvent.call(this, event)
}
}).not.toThrow()
await context.Notification.requestPermission()
context.dispatchPermissionChange('notifications')
expect(events).toHaveLength(2)
expect(events).toEqual([
{ receiver: status, target: status, state: 'granted' },
{ receiver: status, target: status, state: 'granted' }
])
})
// Why: 'camera' rather than 'storage-access' — #14685 narrowed the intercepted set to
// camera/microphone, so a name outside it falls through to the real query and never reaches
// the fallback at all.
it('uses a non-enumerable EventTarget fallback when the native query rejects', async () => {
const context = createContext({
nativeNotificationPermission: 'denied',
requestedNotificationPermission: 'granted',
rejectedPermissions: ['camera']
})
runInNewContext(ANTI_DETECTION_SCRIPT, context)
const status = await context.navigator.permissions.query({ name: 'camera' })
expect(status).toBeInstanceOf(EventTarget)
expect(status).not.toBeInstanceOf(context.PermissionStatus)
expect(status.state).toBe('prompt')
expect(Object.keys(status)).toEqual([])
})
})
-157
View File
@@ -1,157 +0,0 @@
import { runInNewContext } from 'node:vm'
import { describe, expect, it } from 'vitest'
import { ANTI_DETECTION_SCRIPT } from './anti-detection'
import { googleAuthUserAgent } from './browser-google-auth-ua'
type PermissionQueryResult = {
state: string
onchange: null
}
type AntiDetectionContext = {
Notification: {
permission: string
requestPermission: (callback?: (permission: string) => void) => Promise<string>
}
navigator: {
userAgent: string
permissions: {
query: (descriptor: { name: string }) => Promise<PermissionQueryResult>
}
}
window: {
chrome?: {
runtime?: unknown
csi?: () => unknown
loadTimes?: () => unknown
}
}
}
function createContext(args: {
nativeNotificationPermission: string
requestedNotificationPermission: string
userAgent?: string
}): AntiDetectionContext & Record<string, unknown> {
class Permissions {
query(): Promise<PermissionQueryResult> {
return Promise.resolve({ state: 'denied', onchange: null })
}
}
const Notification = {
permission: args.nativeNotificationPermission,
requestPermission(callback?: (permission: string) => void): Promise<string> {
callback?.(args.requestedNotificationPermission)
return Promise.resolve(args.requestedNotificationPermission)
}
}
Object.defineProperty(Notification, 'permission', {
configurable: true,
get: () => args.nativeNotificationPermission
})
return {
Date,
Object,
Promise,
Set,
performance: { now: () => 0 },
// Electron 43 exposes this native object before the anti-detection script runs.
window: { chrome: {} },
navigator: {
userAgent:
args.userAgent ??
'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko) Chrome/151.0.0.0 Safari/537.36',
plugins: [],
languages: [],
permissions: new Permissions()
},
Permissions,
Notification
} as AntiDetectionContext & Record<string, unknown>
}
describe('ANTI_DETECTION_SCRIPT', () => {
it('does not expose Chrome globals under a Firefox identity', () => {
const context = createContext({
nativeNotificationPermission: 'denied',
requestedNotificationPermission: 'denied',
userAgent: googleAuthUserAgent()
})
runInNewContext(ANTI_DETECTION_SCRIPT, context)
expect(context.window.chrome).toBeUndefined()
expect('chrome' in context.window).toBe(false)
})
it('keeps Chrome API stubs aligned with an ordinary Chrome page', () => {
const context = createContext({
nativeNotificationPermission: 'denied',
requestedNotificationPermission: 'denied'
})
runInNewContext(ANTI_DETECTION_SCRIPT, context)
expect(context.window.chrome?.runtime).toBeUndefined()
expect(context.window.chrome?.csi).toBeTypeOf('function')
expect(context.window.chrome?.loadTimes).toBeTypeOf('function')
})
it.each(['geolocation', 'idle-detection', 'midi', 'storage-access'])(
'passes non-intercepted permission queries through to the native state for %s',
async (name) => {
const context = createContext({
nativeNotificationPermission: 'denied',
requestedNotificationPermission: 'denied'
})
runInNewContext(ANTI_DETECTION_SCRIPT, context)
await expect(context.navigator.permissions.query({ name })).resolves.toEqual({
state: 'denied',
onchange: null
})
}
)
it('reports notification permission as granted after a site permission request succeeds', async () => {
const context = createContext({
nativeNotificationPermission: 'denied',
requestedNotificationPermission: 'granted'
})
runInNewContext(ANTI_DETECTION_SCRIPT, context)
expect(context.Notification.permission).toBe('default')
await expect(context.navigator.permissions.query({ name: 'notifications' })).resolves.toEqual({
state: 'prompt',
onchange: null
})
await expect(context.Notification.requestPermission()).resolves.toBe('granted')
expect(context.Notification.permission).toBe('granted')
await expect(context.navigator.permissions.query({ name: 'notifications' })).resolves.toEqual({
state: 'granted',
onchange: null
})
})
it('preserves notification permission when Electron already reports a grant', async () => {
const context = createContext({
nativeNotificationPermission: 'granted',
requestedNotificationPermission: 'granted'
})
runInNewContext(ANTI_DETECTION_SCRIPT, context)
expect(context.Notification.permission).toBe('granted')
await expect(context.navigator.permissions.query({ name: 'notifications' })).resolves.toEqual({
state: 'granted',
onchange: null
})
})
})
-161
View File
@@ -1,161 +0,0 @@
// Why: Cloudflare Turnstile and similar bot detectors probe multiple browser
// APIs beyond navigator.webdriver. This script runs via
// Page.addScriptToEvaluateOnNewDocument before any page JS to mask automation
// signals that CDP debugger attachment and Electron's webview expose.
export const ANTI_DETECTION_SCRIPT = `(function() {
Object.defineProperty(navigator, 'webdriver', { get: () => false });
// Why: Electron webviews expose an empty plugins array. Real Chrome always
// has at least a few default plugins (PDF Viewer, etc.). An empty array is
// a strong automation signal.
if (navigator.plugins.length === 0) {
Object.defineProperty(navigator, 'plugins', {
get: () => [
{ name: 'Chrome PDF Plugin', filename: 'internal-pdf-viewer' },
{ name: 'Chrome PDF Viewer', filename: 'mhjfbmdgcfjbbpaeojofohoefgiehjai' },
{ name: 'Native Client', filename: 'internal-nacl-plugin' }
]
});
}
// Why: auth hosts present Firefox, where Electron's native window.chrome is an identity mismatch.
if (navigator.userAgent.includes('Firefox/')) {
try {
delete window.chrome;
if ('chrome' in window) {
window.chrome = undefined;
}
} catch {}
} else {
// Why: Electron webviews may not have the window.chrome object that real
// Chrome exposes. Turnstile checks for its presence. The csi() and
// loadTimes() stubs satisfy deeper probes of Chrome-specific APIs.
if (!window.chrome) {
window.chrome = {};
}
if (!window.chrome.csi) {
window.chrome.csi = function() {
return {
startE: Date.now(),
onloadT: Date.now(),
pageT: performance.now(),
tran: 15
};
};
}
if (!window.chrome.loadTimes) {
window.chrome.loadTimes = function() {
return {
commitLoadTime: Date.now() / 1000,
connectionInfo: 'h2',
finishDocumentLoadTime: Date.now() / 1000,
finishLoadTime: Date.now() / 1000,
firstPaintAfterLoadTime: 0,
firstPaintTime: Date.now() / 1000,
navigationType: 'Other',
npnNegotiatedProtocol: 'h2',
requestTime: Date.now() / 1000 - 0.16,
startLoadTime: Date.now() / 1000 - 0.3,
wasAlternateProtocolAvailable: false,
wasFetchedViaSpdy: true,
wasNpnNegotiated: true
};
};
}
}
// Why: Electron's Permission API defaults to 'denied' for most permissions,
// but real Chrome returns 'prompt' for ungranted permissions. Returning
// 'denied' is a strong bot signal. Override the query result for common
// permissions that Turnstile and similar detectors probe.
var notificationPermission = 'default';
var setNotificationPermission = function(permission) {
if (permission === 'granted' || permission === 'denied') {
notificationPermission = permission;
return permission;
}
notificationPermission = 'default';
return 'default';
};
var notificationPermissionState = function() {
return notificationPermission === 'default' ? 'prompt' : notificationPermission;
};
try {
if (Notification.permission === 'granted') {
notificationPermission = 'granted';
}
} catch {}
const promptPerms = new Set([
'camera', 'microphone'
]);
const origQuery = Permissions.prototype.query;
// Why: sites must receive the genuine PermissionStatus so native events, brand checks and method
// identity survive. Shadow only state, and resolve it lazily so existing statuses stay current.
function withOverriddenState(realStatus, stateProvider) {
Object.defineProperty(realStatus, 'state', {
configurable: true,
get: stateProvider
});
return realStatus;
}
// Why: some names the real implementation rejects outright; fall back to an EventTarget so
// listener registration still works instead of throwing.
function fallbackStatus(stateProvider) {
const status = new EventTarget();
Object.defineProperties(status, {
state: { configurable: true, get: stateProvider },
onchange: { configurable: true, value: null, writable: true }
});
return status;
}
function queryWithState(permissions, desc, stateProvider) {
let real;
try {
real = origQuery.call(permissions, desc);
} catch {
return Promise.resolve(fallbackStatus(stateProvider));
}
return Promise.resolve(real).then(
(status) => withOverriddenState(status, stateProvider),
() => fallbackStatus(stateProvider)
);
}
Permissions.prototype.query = function(desc) {
if (desc.name === 'notifications') {
return queryWithState(this, desc, notificationPermissionState);
}
if (promptPerms.has(desc.name)) {
return queryWithState(this, desc, () => 'prompt');
}
return origQuery.call(this, desc);
};
// Why: Electron may report Notification.permission as 'denied' by default
// whereas real Chrome reports 'default' for sites that haven't been granted
// or blocked. Turnstile cross-references this with the Permissions API.
try {
Object.defineProperty(Notification, 'permission', {
get: () => notificationPermission
});
const origRequestPermission = Notification.requestPermission;
if (typeof origRequestPermission === 'function') {
Notification.requestPermission = function(callback) {
var wrappedCallback = typeof callback === 'function'
? function(permission) {
callback(setNotificationPermission(permission));
}
: undefined;
var result = origRequestPermission.call(Notification, wrappedCallback);
if (result && typeof result.then === 'function') {
return result.then(function(permission) {
return setNotificationPermission(permission);
});
}
return result;
};
}
} catch {}
// Why: Electron webviews may have an empty languages array. Real Chrome
// always has at least one entry. An empty array is an automation signal.
if (!navigator.languages || navigator.languages.length === 0) {
Object.defineProperty(navigator, 'languages', {
get: () => ['en-US', 'en']
});
}
})()`
+4 -4
View File
@@ -1,12 +1,12 @@
// Why: Google binds a signed-in session to the browser identity that created it.
// Cookies copied in from another browser (or sent under an Electron/Chrome-shaped
// UA that doesn't match a real first-party browser) get flagged by anti-fraud on
// accounts.google.com and expire within ~1h. Presenting a Firefox identity scoped
// Cookies copied in from another browser (or sent under a UA that doesn't match a
// real first-party browser) get flagged by anti-fraud on accounts.google.com and
// expire within ~1h. Presenting a Firefox identity scoped
// to Google's auth hosts lets the user sign in *inside* the embedded browser, so
// Google issues cookies bound to THIS browser that self-refresh — instead of us
// transplanting cookies that go stale. Scope is deliberately the auth hosts only:
// post-auth app surfaces (mail.google.com, myaccount.google.com, drive, etc.) keep
// the profile's real Chrome-shaped identity so nothing else about the session shifts.
// the profile's real identity so nothing else about the session shifts.
// Why: exact hostname match — subdomains such as myaccount.google.com are post-auth
// app surfaces, not the sign-in flow, and must retain the profile's real identity.
@@ -51,7 +51,7 @@ import {
import {
createViewportGuestFactory,
flushViewportOps,
GUEST_CLEAN_UA
GUEST_ELECTRON_UA
} from './browser-manager-viewport-test-fixtures'
const {
@@ -197,8 +197,9 @@ describe('browserManager', () => {
// Why: popup child windows get attachGuestPolicies but are never entered into tabIdByWebContentsId,
// so a direct lookup of the UA mode misses the native opt-out. That is worse than doing nothing —
// native sessions skip setupClientHintsOverride, so the popup would send the raw Electron UA on the
// wire while navigator.userAgent claimed Firefox. Google sign-in popups are a first-class surface.
// native sessions never install the header-level Firefox switch, so the popup would send the
// Electron UA on the wire while navigator.userAgent claimed Firefox. Google sign-in popups are a
// first-class surface.
it('leaves the UA untouched on auth hosts for a popup owned by a native-UA profile', () => {
const ownerGuest = {
id: 415,
@@ -543,7 +544,7 @@ describe('browserManager', () => {
)
expect(uaWrites.length).toBeGreaterThan(0)
for (const [, params] of uaWrites) {
expect((params as { userAgent: string }).userAgent).toBe(GUEST_CLEAN_UA)
expect((params as { userAgent: string }).userAgent).toBe(GUEST_ELECTRON_UA)
}
})
})
@@ -637,9 +637,9 @@ describe('browserManager', () => {
).toHaveLength(2)
})
it('cancels pending anti-detection reattach timers when unregistering a guest', () => {
vi.useFakeTimers()
// Why: a plain browsing tab must never attach a debugger (Cloudflare treats CDP as a bot signal);
// the only debugger wiring it keeps is the detach listener that invalidates the auth-host UA override.
it('never attaches a debugger to a browsing guest and drops its detach listener on unregister', () => {
const debuggerHandlers = new Map<string, () => void>()
const debuggerAttachMock = vi.fn()
const guest = {
@@ -670,18 +670,17 @@ describe('browserManager', () => {
browserManager.attachGuestPolicies(guest as never)
browserManager.registerGuest({
browserPageId: 'browser-reattach',
browserPageId: 'browser-no-debugger',
webContentsId: 809,
rendererWebContentsId
})
debuggerHandlers.get('detach')?.()
expect(vi.getTimerCount()).toBe(1)
expect(debuggerAttachMock).not.toHaveBeenCalled()
expect(guest.debugger.sendCommand).not.toHaveBeenCalled()
expect(debuggerHandlers.has('detach')).toBe(true)
browserManager.unregisterGuest('browser-reattach')
expect(vi.getTimerCount()).toBe(0)
vi.advanceTimersByTime(500)
expect(debuggerAttachMock).toHaveBeenCalledTimes(1)
browserManager.unregisterGuest('browser-no-debugger')
expect(debuggerHandlers.has('detach')).toBe(false)
expect(debuggerAttachMock).not.toHaveBeenCalled()
})
})
@@ -52,6 +52,8 @@ type GuestFake = {
isAttached: () => boolean
attach: ReturnType<typeof vi.fn>
sendCommand: ReturnType<typeof vi.fn>
on: ReturnType<typeof vi.fn>
off: ReturnType<typeof vi.fn>
}
on: (event: string, listener: (...args: never[]) => void) => void
once: (event: string, listener: (...args: never[]) => void) => void
@@ -81,7 +83,9 @@ function createGuest(id: number, url: string): GuestFake {
debugger: {
isAttached: () => true,
attach: vi.fn(),
sendCommand: vi.fn(async () => undefined)
sendCommand: vi.fn(async () => undefined),
on: vi.fn(),
off: vi.fn()
},
on: (event, listener) => {
listeners.set(event, [...(listeners.get(event) ?? []), listener])
@@ -143,7 +147,7 @@ describe('guest policy profiles', () => {
// The presence half of every absence below: a browsing guest observably takes all of it through
// the same method, so a profile that fenced nothing — or an attach path that stopped installing
// anything at all — cannot pass these by being uniformly empty.
it('gives a browsing guest link routing, popups and anti-detection', () => {
it('gives a browsing guest link routing, popups and auth-identity detach tracking', () => {
const guest = createGuest(300, 'https://example.com/')
browserManager.attachGuestPolicies(guest as never)
@@ -151,7 +155,10 @@ describe('guest policy profiles', () => {
expect(listenerCount(guest, 'dom-ready')).toBe(1)
expect(listenerCount(guest, 'frame-created')).toBe(1)
expect(listenerCount(guest, 'did-create-window')).toBe(1)
expect(guest.debugger.sendCommand).toHaveBeenCalled()
expect(guest.debugger.on).toHaveBeenCalledWith('detach', expect.any(Function))
// Why: a plain browsing tab must never attach a debugger; Cloudflare treats CDP as a bot signal.
expect(guest.debugger.attach).not.toHaveBeenCalled()
expect(guest.debugger.sendCommand).not.toHaveBeenCalled()
expect(navigateTo(guest, 'https://elsewhere.example/')).toBe(false)
})
@@ -161,6 +168,7 @@ describe('guest policy profiles', () => {
expect(listenerCount(guest, 'dom-ready')).toBe(0)
expect(listenerCount(guest, 'frame-created')).toBe(0)
expect(listenerCount(guest, 'did-create-window')).toBe(0)
expect(guest.debugger.on).not.toHaveBeenCalled()
expect(guest.debugger.sendCommand).not.toHaveBeenCalled()
expect(guest.executeJavaScriptInIsolatedWorld).not.toHaveBeenCalled()
})
@@ -35,8 +35,7 @@ export abstract class BrowserManagerGuestPolicy extends BrowserManagerGuestClean
this.clickedLinkFrameNameByGuestId.set(guest.id, clickedLinkFrameName)
}
// Why: bot detectors probe APIs that differ in Electron webviews; inject overrides each load so manual browsing passes.
const disposeAntiDetection = this.injectAntiDetection(guest)
const disposeAuthDetachTracking = this.trackDebuggerDetachForAuthUserAgent(guest)
// Why: disable throttling so background screenshots still get frames; else the compositor stalls and capture returns empty.
guest.setBackgroundThrottling(false)
const disposePopupPolicy = this.installGuestPopupPolicy(guest, clickedLinkFrameName)
@@ -44,14 +43,14 @@ export abstract class BrowserManagerGuestPolicy extends BrowserManagerGuestClean
// Why: store cleanup so unregisterGuest can drop these listeners on teardown and let the WebContents wrapper GC.
this.policyCleanupByGuestId.set(guest.id, () => {
disposeAntiDetection()
disposeAuthDetachTracking()
disposePopupPolicy()
disposeNavigationPolicy()
})
}
/**
* A workspace document is not the web: no popups, no link routing, no anti-detection, and no
* A workspace document is not the web: no popups, no link routing, no auth-identity tracking, and no
* navigation bookkeeping for chrome it does not have. What it does share with a browsing guest is
* this method's teardown, so a retired preview drops its listeners on the same path.
*/
+10 -12
View File
@@ -1,5 +1,4 @@
import { openPopupWithOriginBar, type PopupChildWindowOptions } from './popup-origin-bar-window'
import { cleanElectronUserAgent } from './browser-session-ua'
import { getBrowserSessionUserAgentMode } from './browser-session-user-agent-mode'
import { googleAuthUserAgent, isGoogleAuthUrl } from './browser-google-auth-ua'
import { buildViewportUserAgentOverride } from './browser-viewport-user-agent'
@@ -12,10 +11,10 @@ import { BrowserManagerVisibility } from './browser-manager-visibility'
export abstract class BrowserManagerNavigation extends BrowserManagerVisibility {
// Why: navigator.userAgent (read by Google's auth JS) reflects the WebContents UA,
// not the request header, so the header-level Firefox switch in setupClientHintsOverride
// not the request header, so the header-level Firefox switch in setupGoogleAuthUserAgentOverride
// must be matched here per navigation or the two layers disagree — itself a bot tell.
// Restores the session's base identity off the auth hosts. Native-UA profiles opt out
// of the whole clean-UA path, so they keep their untouched identity everywhere.
// of the Firefox switch, so they keep their untouched identity everywhere.
protected applyGoogleAuthUserAgent(
guest: Electron.WebContents,
url: string,
@@ -24,8 +23,8 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility
const browserPageId = this.tabIdByWebContentsId.get(guest.id)
// Why: popup child windows get these policies but are never in tabIdByWebContentsId, so a direct
// lookup misses the native-UA opt-out and would hand a native profile's popup the Firefox UA.
// That is worse than doing nothing: native sessions skip setupClientHintsOverride entirely, so
// the popup would send the raw Electron UA on the wire while navigator.userAgent claims Firefox.
// That is worse than doing nothing: native sessions never install the header-level Firefox
// switch, so the popup would send the Electron UA on the wire while navigator.userAgent claims Firefox.
const ownerTabId = this.resolveBrowserTabIdForGuestWebContentsId(guest.id)
// Session state is authoritative before renderer registration and after a native profile imports a source UA.
const mode =
@@ -56,15 +55,14 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility
// navigation (ERR_ABORTED) and replay the original request, which a POST-started OAuth chain
// cannot survive — the sign-in lands on a blank tab. CDP retargets navigator.userAgent without
// touching the navigation, and it outranks the WebContents UA from then on, so a guest that
// switches to it stays on it. The wire UA never depended on this write: setupClientHintsOverride
// rewrites User-Agent per request for auth-host URLs on its own.
// switches to it stays on it. The wire UA never depended on this write:
// setupGoogleAuthUserAgentOverride rewrites User-Agent per request for auth-host URLs on its own.
if (options.duringRedirect === true || overrideState !== undefined) {
if (this.canOverrideUserAgentOverCdp(guest)) {
authOverrideIssuedOverCdp = true
// Why: go through the viewport builder rather than writing nextUa raw, so both CDP writers
// resolve one identity for this URL — Firefox on auth hosts, the profile's clean base off
// them, any mobile preset preserved. Writing the session UA directly would put the
// unlaundered Electron token back on the wire.
// resolve one identity for this URL — Firefox on auth hosts, the session's base identity
// off them, any mobile preset preserved.
void this.applyAuthUserAgentOverrideOverCdp(
guest,
(browserPageId ? this.viewportUaOverrideMobileByTabId.get(browserPageId) : undefined) ??
@@ -188,7 +186,7 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility
// Why: Emulation.setUserAgentOverride is set once and stands across every later navigation,
// outranking setUserAgent for navigator.userAgent. A viewport preset applied before reaching an
// auth host would otherwise pin navigator.userAgent to the Chrome-shaped preset UA while the
// auth host would otherwise pin navigator.userAgent to the session's preset UA while the
// request header says Firefox — the two-layer disagreement this scope exists to remove.
protected reapplyViewportUserAgentOverride(
guest: Electron.WebContents,
@@ -222,7 +220,7 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility
// Why: the session UA is the profile's stable base identity. guest.getUserAgent() is not:
// applyGoogleAuthUserAgent leaves it pinned to the Firefox auth UA once a guest switches to
// the CDP override, so reading it back here would republish that identity on ordinary hosts.
baseUserAgent: cleanElectronUserAgent(baseUserAgent ?? guest.session.getUserAgent())
baseUserAgent: baseUserAgent ?? guest.session.getUserAgent()
})
)
}
+5 -43
View File
@@ -1,4 +1,3 @@
import { ANTI_DETECTION_SCRIPT } from './anti-detection'
import { BrowserGrabSessionController } from './browser-grab-session-controller'
import type { BrowserCertificateTrustController } from './browser-certificate-trust-controller'
import {
@@ -190,56 +189,19 @@ export abstract class BrowserManagerState extends BrowserManagerViewportScrollSt
this.settingsResolver = resolver
}
// Why: addScriptToEvaluateOnNewDocument (CDP) is the only reliable pre-page-script hook per nav; executeJavaScript ran on the old page context.
protected injectAntiDetection(guest: Electron.WebContents): () => void {
let disposed = false
let reattachTimer: ReturnType<typeof setTimeout> | null = null
const attach = (): void => {
if (disposed || guest.isDestroyed()) {
return
}
try {
if (!guest.debugger.isAttached()) {
guest.debugger.attach('1.3')
}
void guest.debugger
.sendCommand('Page.enable', {})
.then(() =>
guest.debugger.sendCommand('Page.addScriptToEvaluateOnNewDocument', {
source: ANTI_DETECTION_SCRIPT
})
)
.catch(() => {})
} catch {
/* best-effort — debugger may be unavailable */
}
}
// Why: proxy/bridge stop detaches the debugger and drops injections; re-attach (500ms delay to avoid racing a mid-restart) to keep overrides.
// Why: a debugger detach clears every CDP override Chromium holds, including the Google auth-host
// UA override, so the confirmed-override record must be dropped or the next auth navigation
// believes the identity is still installed and skips the write.
protected trackDebuggerDetachForAuthUserAgent(guest: Electron.WebContents): () => void {
const onDetach = (): void => {
this.authUserAgentOverrideStateByGuestId.delete(guest.id)
if (!disposed && !guest.isDestroyed() && reattachTimer === null) {
reattachTimer = setTimeout(() => {
reattachTimer = null
attach()
}, 500)
}
}
try {
attach()
guest.debugger.on('detach', onDetach)
} catch {
/* best-effort */
/* debugger may be unavailable */
}
return () => {
disposed = true
if (reattachTimer !== null) {
clearTimeout(reattachTimer)
reattachTimer = null
}
try {
guest.debugger.off('detach', onDetach)
} catch {
+1 -1
View File
@@ -117,7 +117,7 @@ export type PopupOwnerContext = {
/**
* What a guest is allowed to be. A browsing guest is the web — popups, clicked-link routing and
* anti-detection all apply. A workspace-document guest renders one granted document and gets none
* auth-identity tracking all apply. A workspace-document guest renders one granted document and gets none
* of that; `host` is the renderer that minted its grant, and the only sink for what it reports.
*/
export type BrowserGuestPolicy =
@@ -49,7 +49,6 @@ import {
import {
createViewportGuestFactory,
flushViewportOps,
GUEST_CLEAN_UA,
GUEST_ELECTRON_UA
} from './browser-manager-viewport-test-fixtures'
@@ -207,7 +206,7 @@ describe('browserManager', () => {
mobile: false
})
expect(debuggerSendCommand).toHaveBeenLastCalledWith('Emulation.setUserAgentOverride', {
userAgent: GUEST_CLEAN_UA
userAgent: GUEST_ELECTRON_UA
})
// Navigating to the auth host must move the standing override to the Firefox identity.
@@ -218,11 +217,11 @@ describe('browserManager', () => {
userAgent: googleAuthUserAgent()
})
// Leaving the auth host restores the clean Chrome-shaped preset UA.
// Leaving the auth host restores the session's own preset UA.
debuggerSendCommand.mockClear()
willRedirect({ preventDefault: vi.fn() }, 'https://example.com/', false, true)
await flushViewportOps()
expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA })
expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA })
})
// Why: not an ordering race — debugger.sendCommand dispatches in call order over one channel, so
@@ -241,9 +240,9 @@ describe('browserManager', () => {
}
// Why mobile: on the desktop branch the break is masked by coincidence — applyGoogleAuthUserAgent
// has already switched the WebContents UA to Firefox, and cleanElectronUserAgent passes a Firefox
// UA through untouched, so the stale-URL desktop path happens to emit Firefox anyway. The mobile
// branch derives a Chrome-shaped iPhone UA from that same base and exposes the real defect.
// has already switched the WebContents UA to Firefox, so the stale-URL desktop path happens to
// emit Firefox anyway. The mobile branch derives a Chrome-shaped iPhone UA from the session base
// and exposes the real defect.
it('does not leave the Chrome preset UA standing when a mobile preset lands mid-navigation onto an auth host', async () => {
const { guest, debuggerSendCommand } = makeGuest(4251, 'https://example.com/')
// Hold the preset's first CDP command open so the navigation lands inside its await window.
@@ -332,7 +331,7 @@ describe('browserManager', () => {
// Without the fix the resuming preset re-reads getURL() as the auth host and clobbers the
// navigation's correct write, stranding the Firefox UA on a non-auth page.
expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA })
expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA })
})
it('falls back to the committed URL once a navigation commits or fails', async () => {
@@ -378,7 +377,7 @@ describe('browserManager', () => {
await flushViewportOps()
expect(guest.setUserAgent).toHaveBeenLastCalledWith(GUEST_ELECTRON_UA)
expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA })
expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA })
// A later preset must also resolve the committed, non-auth URL.
debuggerSendCommand.mockClear()
@@ -457,7 +456,7 @@ describe('browserManager', () => {
expect(guest.setUserAgent).not.toHaveBeenCalled()
expect(debuggerSendCommand).not.toHaveBeenCalledWith(
'Emulation.setUserAgentOverride',
expect.objectContaining({ userAgent: GUEST_CLEAN_UA })
expect.objectContaining({ userAgent: GUEST_ELECTRON_UA })
)
})
@@ -517,7 +516,7 @@ describe('browserManager', () => {
didFailLoad(null, -3, 'Aborted', 'https://accounts.google.com/redirected', true)
await flushViewportOps()
expect(guest.setUserAgent).not.toHaveBeenCalled()
expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA })
expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA })
})
it('preserves the auth identity when a viewport preset is cleared after a redirect', async () => {
@@ -592,7 +591,7 @@ describe('browserManager', () => {
didStartNavigation(null, 'https://example.com/', false, true)
await flushViewportOps()
expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA })
expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA })
})
it('reapplies a preset when navigation starts during its final UA write', async () => {
@@ -849,8 +848,7 @@ describe('browserManager', () => {
expect(debuggerAttach).toHaveBeenCalledWith('1.3')
expect(debuggerSendCommand).toHaveBeenCalled()
// Why: detaching would clear Page.addScriptToEvaluateOnNewDocument
// (anti-detection). Guard regression.
// Why: detaching would clear every standing CDP override (viewport, auth UA). Guard regression.
expect((guest.debugger as { detach?: unknown }).detach ?? undefined).toBeUndefined()
})
@@ -3,8 +3,6 @@ import type { BrowserManagerMocks } from './browser-manager-test-harness'
export const GUEST_ELECTRON_UA =
'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) orca/1.0.0 Chrome/134.0.0.0 Electron/30.0.0 Safari/537.36'
export const GUEST_CLEAN_UA =
'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/134.0.0.0 Safari/537.36'
// Why: viewport UA writes are queued on the per-tab chain, so draining it takes more than one
// microtask hop; loop until the chain is empty rather than guessing a tick count.
@@ -53,7 +51,9 @@ export function createViewportGuestFactory(
debugger: {
isAttached: debuggerIsAttached,
attach: debuggerAttach,
sendCommand: debuggerSendCommand
sendCommand: debuggerSendCommand,
on: vi.fn(),
off: vi.fn()
}
}
return {
+1 -1
View File
@@ -30,7 +30,7 @@ export abstract class BrowserManagerViewport extends BrowserManagerDownloadLifec
return true
}
// Why: emulate viewport via CDP; never detach the debugger here or per-guest overrides (addScriptToEvaluateOnNewDocument) are cleared.
// Why: emulate viewport via CDP; never detach the debugger here or the agent bridge's per-guest state is cleared.
async setViewportOverride(
browserTabId: string,
override: BrowserViewportOverride | null
@@ -82,8 +82,7 @@ vi.mock('./browser-media-access', () => ({
requestSystemMediaAccess: async () => false
}))
vi.mock('./browser-session-ua', () => ({
cleanElectronUserAgent: (userAgent: string) => userAgent,
setupClientHintsOverride: vi.fn()
setupGoogleAuthUserAgentOverride: vi.fn()
}))
vi.mock('./browser-session-user-agent-mode', () => ({
setBrowserSessionUserAgentMode: vi.fn()
@@ -9,7 +9,7 @@ import {
} from './browser-session-proxy'
import { hasSystemMediaAccess, requestSystemMediaAccess } from './browser-media-access'
import { isAutoGrantedBrowserSessionPermission } from './browser-session-permission-policy'
import { cleanElectronUserAgent, setupClientHintsOverride } from './browser-session-ua'
import { setupGoogleAuthUserAgentOverride } from './browser-session-ua'
import { setBrowserSessionUserAgentMode } from './browser-session-user-agent-mode'
import {
allowsBrowserWebAuthnPermission,
@@ -92,10 +92,8 @@ export function installBrowserSessionPartitionPolicies(
}
browserManager.installCertificateRequestGuard(sess)
if (profile.userAgentMode !== 'native' && typeof sess.getUserAgent === 'function') {
const cleanUA = cleanElectronUserAgent(sess.getUserAgent())
sess.setUserAgent(cleanUA)
setupClientHintsOverride(sess, cleanUA)
if (profile.userAgentMode !== 'native') {
setupGoogleAuthUserAgentOverride(sess)
}
if (options?.permissions === 'deny') {
sess.setPermissionRequestHandler((_webContents, _permission, callback) => callback(false))
@@ -191,11 +189,7 @@ export function applyBrowserSessionUserAgentModes(profiles: BrowserSessionProfil
if (profile.userAgentMode === 'native') {
continue
}
// Why: the default Electron UA leaks "Electron/X.X.X" + app name, which trips Cloudflare Turnstile.
const cleanUA = cleanElectronUserAgent(sess.getUserAgent())
sess.setUserAgent(cleanUA)
setupClientHintsOverride(sess, cleanUA)
setupGoogleAuthUserAgentOverride(sess)
} catch {
/* session not available yet (e.g. unit tests or pre-ready) */
}
@@ -44,8 +44,7 @@ vi.mock('./browser-media-access', () => ({
requestSystemMediaAccess: vi.fn(async () => false)
}))
vi.mock('./browser-session-ua', () => ({
cleanElectronUserAgent: vi.fn((ua: string) => ua),
setupClientHintsOverride: vi.fn()
setupGoogleAuthUserAgentOverride: vi.fn()
}))
vi.mock('./browser-session-user-agent-mode', () => ({
setBrowserSessionUserAgentMode: vi.fn(),

Some files were not shown because too many files have changed in this diff Show More