diff --git a/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml b/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml index 6430c3f4793..8ef61507088 100644 --- a/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml +++ b/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml @@ -91,7 +91,11 @@ jobs: test -n "${CAPACITY_SERVICE_ACCOUNT}" test -n "${DIRECTOR_RUNTIME_SERVICE_ACCOUNT}" + # Full history: the monitor evidence this job verifies is sealed at an ancestor commit, + # and the provenance check fails closed on a commit a shallow clone left out. - uses: actions/checkout@v4 + with: + fetch-depth: 0 - uses: pnpm/action-setup@v4 with: { package_json_file: cloud/package.json } @@ -177,11 +181,12 @@ jobs: env: ORCA_RELAY_ADMIN_ID_TOKEN: ${{ steps.deploy-auth.outputs.id_token }} run: | - RETRY_ARGS=() - if test "${WAVE_INDEX}" != 0; then RETRY_ARGS=(--retry-freshness); fi + # Freshness-only failures are publish lag, not health, on every wave + # including the first; the CLI still caps the retry at the wave's + # evidence-age budget, so this cannot mutate on aged evidence. pnpm incident:relay-preflight -- \ --state-file "${OUTPUT_DIRECTORY}/relay-${MONITOR_RUN_ID}-dry-run.state.json" \ - --wave-index "${WAVE_INDEX}" "${RETRY_ARGS[@]}" + --wave-index "${WAVE_INDEX}" --retry-freshness - name: Require durable rehome disabled and exact selector env: @@ -271,10 +276,22 @@ jobs: env: ORCA_RELAY_ADMIN_ID_TOKEN: ${{ steps.deploy-auth.outputs.id_token }} run: | - CURRENT_RUNTIME="$(curl --fail-with-body --max-time 30 \ - --request POST "${CELL_ORIGIN}/v1/admin/runtime-status" \ - --header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \ - --header 'Content-Type: application/json' --data '{"v":1}')" + # A single transient 5xx (LB warm-up behind a fresh instance) must not + # fail a canary; 4xx (auth, generation mismatch) still fails fast. + admin_post() { + local out="${RUNNER_TEMP}/$1.json" + if ! curl --fail-with-body --max-time 30 \ + --retry 3 --retry-delay 2 --retry-connrefused --output "${out}" \ + --request POST "$2" \ + --header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \ + --header 'Content-Type: application/json' --data "$3"; then + cat "${out}" >&2 + return 1 + fi + cat "${out}" + } + CURRENT_RUNTIME="$(admin_post current-runtime \ + "${CELL_ORIGIN}/v1/admin/runtime-status" '{"v":1}')" # A rollback that failed between template apply and admission restore # leaves the cell already on the rollback image; resume from that # state instead of demanding the pre-rollback predecessor. @@ -370,11 +387,9 @@ jobs: if .regionalRehomeProtocol == null then "regionalRehomeProtocol" else empty end ] | if length > 0 then "runtime predecessor normalized legacy fields=" + join(",") else empty end' \ <<< "${CURRENT_RUNTIME}" - CURRENT_DIRECTOR_STATUS="$(curl --fail-with-body --max-time 30 \ - --request POST "${DIRECTOR_ORIGIN}/v1/admin/cell-status" \ - --header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \ - --header 'Content-Type: application/json' \ - --data "$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")" + CURRENT_DIRECTOR_STATUS="$(admin_post current-cell-status \ + "${DIRECTOR_ORIGIN}/v1/admin/cell-status" \ + "$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")" SOURCE_INCARNATION="$(jq -er '.status.runtime.cellIncarnation' \ <<< "${CURRENT_DIRECTOR_STATUS}")" if test "${ROLLBACK_RESUME}" = true && ! jq -e \ @@ -418,13 +433,13 @@ jobs: # result's generation is authoritative either way. ISOLATE_RESULT="$(node dev/scripts/prepare-relay-production-capacity-canary.mjs \ --director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \ - --cell-id "${TARGET_CELL_ID}" --mode isolate)" + --cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode isolate)" echo "${ISOLATE_RESULT}" ISOLATE_GENERATION="$(jq -er '.generation' <<< "${ISOLATE_RESULT}")" echo "SELECTOR_GENERATION_AFTER_ISOLATE=${ISOLATE_GENERATION}" >> "${GITHUB_ENV}" node dev/scripts/prepare-relay-production-capacity-canary.mjs \ --director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \ - --cell-id "${TARGET_CELL_ID}" --mode drain + --cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode drain node dev/scripts/verify-relay-capacity-transition.mjs \ --director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \ --cell-id "${TARGET_CELL_ID}" --hard-cap "${EXPECTED_HARD_CAP}" \ @@ -486,6 +501,7 @@ jobs: --rollback-image "${DESIRED_IMAGE}" \ --rehome-director-service-account "${DIRECTOR_RUNTIME_SERVICE_ACCOUNT}" \ --rehome-audience https://relay.onorca.dev/v1/admin/host-drain \ + --regional-rehome-protocol "${DESIRED_REHOME_PROTOCOL}" \ | jq -e '.changes == 2' >/dev/null fi gcloud compute instance-groups managed wait-until "${MIG_NAME}" --stable \ @@ -511,7 +527,8 @@ jobs: --unobserved-bound "${EXPECTED_UNOBSERVED_BOUND}" --image "${DESIRED_IMAGE}" \ --rollback-image "${IMAGE_REPOSITORY}@${CURRENT_IMAGE_DIGEST}" \ --rehome-director-service-account "${DIRECTOR_RUNTIME_SERVICE_ACCOUNT}" \ - --rehome-audience https://relay.onorca.dev/v1/admin/host-drain + --rehome-audience https://relay.onorca.dev/v1/admin/host-drain \ + --regional-rehome-protocol "${DESIRED_REHOME_PROTOCOL}" terraform -chdir=infra/terraform apply -auto-approve \ "${RUNNER_TEMP}/relay-same-cap.tfplan" gcloud compute instance-groups managed wait-until "${MIG_NAME}" --stable \ @@ -532,6 +549,20 @@ jobs: env: ORCA_RELAY_ADMIN_ID_TOKEN: ${{ steps.post-auth.outputs.id_token }} run: | + # A single transient 5xx (LB warm-up behind a fresh instance) must not + # fail a canary; 4xx (auth, generation mismatch) still fails fast. + admin_post() { + local out="${RUNNER_TEMP}/$1.json" + if ! curl --fail-with-body --max-time 30 \ + --retry 3 --retry-delay 2 --retry-connrefused --output "${out}" \ + --request POST "$2" \ + --header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \ + --header 'Content-Type: application/json' --data "$3"; then + cat "${out}" >&2 + return 1 + fi + cat "${out}" + } node dev/scripts/verify-relay-capacity-transition.mjs \ --director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \ --cell-id "${TARGET_CELL_ID}" --hard-cap "${EXPECTED_HARD_CAP}" \ @@ -539,19 +570,15 @@ jobs: --heartbeat fresh --admission migration-only --draining forbidden \ --activity allowed --expected-image-digests "${DESIRED_IMAGE_DIGEST}" \ --regional-rehome-protocol "${DESIRED_REHOME_PROTOCOL}" --timeout-ms 900000 - TARGET_RUNTIME="$(curl --fail-with-body --max-time 30 \ - --request POST "${CELL_ORIGIN}/v1/admin/runtime-status" \ - --header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \ - --header 'Content-Type: application/json' --data '{"v":1}')" + TARGET_RUNTIME="$(admin_post target-runtime \ + "${CELL_ORIGIN}/v1/admin/runtime-status" '{"v":1}')" jq -e --arg digest "${DESIRED_IMAGE_DIGEST}" \ --argjson protocol "${DESIRED_REHOME_PROTOCOL}" \ '.imageDigest == $digest and (.regionalRehomeProtocol // 0) == $protocol' \ <<< "${TARGET_RUNTIME}" >/dev/null - TARGET_DIRECTOR_STATUS="$(curl --fail-with-body --max-time 30 \ - --request POST "${DIRECTOR_ORIGIN}/v1/admin/cell-status" \ - --header "Authorization: Bearer ${ORCA_RELAY_ADMIN_ID_TOKEN}" \ - --header 'Content-Type: application/json' \ - --data "$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")" + TARGET_DIRECTOR_STATUS="$(admin_post target-cell-status \ + "${DIRECTOR_ORIGIN}/v1/admin/cell-status" \ + "$(jq -cn --arg cell "${TARGET_CELL_ID}" '{v:1,cellId:$cell}')")" TARGET_INCARNATION="$(jq -er '.status.runtime.cellIncarnation' \ <<< "${TARGET_DIRECTOR_STATUS}")" if test "${ROLLBACK_RESUME}" = true; then @@ -588,7 +615,7 @@ jobs: echo "MUTATION_STARTED=true" >> "${GITHUB_ENV}" ACTIVATE_RESULT="$(node dev/scripts/prepare-relay-production-capacity-canary.mjs \ --director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \ - --cell-id "${TARGET_CELL_ID}" --mode activate)" + --cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode activate)" echo "${ACTIVATE_RESULT}" SELECTOR_GENERATION_AFTER_ACTIVATE="$(jq -er '.generation' \ <<< "${ACTIVATE_RESULT}")" @@ -627,7 +654,7 @@ jobs: test "${MUTATION_STARTED:-false}" = true || exit 0 ISOLATE_RESULT="$(node dev/scripts/prepare-relay-production-capacity-canary.mjs \ --director-origin "${DIRECTOR_ORIGIN}" --cell-origin "${CELL_ORIGIN}" \ - --cell-id "${TARGET_CELL_ID}" --mode isolate)" + --cell-id "${TARGET_CELL_ID}" --approved-cells same-cap --mode isolate)" echo "${ISOLATE_RESULT}" # The isolate result carries the authoritative post-isolate generation; # fixed offsets are wrong whenever an earlier isolate was a no-op. diff --git a/.github/workflows/cloud-deploy-relay-production-same-cap.yml b/.github/workflows/cloud-deploy-relay-production-same-cap.yml index 1994966d083..fba5df0dcb9 100644 --- a/.github/workflows/cloud-deploy-relay-production-same-cap.yml +++ b/.github/workflows/cloud-deploy-relay-production-same-cap.yml @@ -87,13 +87,18 @@ jobs: gate: if: ${{ vars.ORCA_CLOUD_OPERATIONS_ENABLED == 'true' && (github.ref == 'refs/heads/main') }} runs-on: blacksmith-2vcpu-ubuntu-2204 - timeout-minutes: 10 + # Headroom for the full-history checkout the canary provenance check needs. + timeout-minutes: 15 environment: production outputs: cells: ${{ steps.wave.outputs.cells }} job-mode: ${{ steps.wave.outputs.job-mode }} steps: + # Full history: the canary authority a batch verifies is sealed at an ancestor commit, and + # the provenance check fails closed on a commit a shallow clone left out. - uses: actions/checkout@v4 + with: + fetch-depth: 0 - uses: actions/setup-node@v4 with: { node-version: 24 } diff --git a/.github/workflows/cloud-operate-relay-production-rehome-job.yml b/.github/workflows/cloud-operate-relay-production-rehome-job.yml index 682953af5e7..fdb1aca45e0 100644 --- a/.github/workflows/cloud-operate-relay-production-rehome-job.yml +++ b/.github/workflows/cloud-operate-relay-production-rehome-job.yml @@ -95,7 +95,11 @@ jobs: ;; esac + # Full history: the monitor evidence this job verifies is sealed at an ancestor commit, + # and the provenance check fails closed on a commit a shallow clone left out. - uses: actions/checkout@v4 + with: + fetch-depth: 0 - uses: actions/setup-node@v4 with: diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index ca1651325c9..a749214e232 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -807,6 +807,7 @@ jobs: src/main/agent-hooks/windows-hook-payload-delivery.test.ts src/main/windows/windows-pty-job.win32.test.ts src/main/windows/windows-host-job.win32.test.ts + src/main/windows-live-tree-kill.win32.test.ts src/main/wsl/wsl-runner.test.ts src/main/wsl/wsl-guest-environment.test.ts src/main/wsl/wsl-invocation-boundary.test.ts diff --git a/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts b/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts index 18ee4495078..412c2905408 100644 --- a/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts +++ b/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts @@ -6,7 +6,10 @@ import { livePreflightGcloud, runIncidentLivePreflight } from './incident-live-preflight-cli.js' -import type { IncidentSample } from './incident-monitor.js' +import { + INCIDENT_MONITOR_THRESHOLDS, + type IncidentSample +} from './incident-monitor.js' import type { AdmissionSelector } from './incident-selector.js' const directories: string[] = [] @@ -313,7 +316,7 @@ describe('relay incident live preflight', () => { it('retries freshness-only failures when explicitly requested', async () => { const stale = sample() stale.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = - new Date(now - 180_001).toISOString() + new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() const missing = sample() delete missing.sources['relay-logs'] const collect = vi.fn() @@ -331,11 +334,44 @@ describe('relay incident live preflight', () => { expect(wait).toHaveBeenNthCalledWith(2, 15_000) }) + it('retries a first-wave stale sample and passes on the fresh one', async () => { + const stale = sample() + stale.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = + new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() + const collect = vi.fn().mockResolvedValueOnce(stale).mockResolvedValueOnce(sample()) + const wait = vi.fn(async () => undefined) + await expect(runIncidentLivePreflight( + ['--state-file', stateFile(), '--wave-index', '0', '--retry-freshness'], + { now: () => now, collect, wait } + )).resolves.toBeUndefined() + expect(collect).toHaveBeenCalledTimes(2) + expect(wait).toHaveBeenCalledOnce() + }) + + it('stops retrying when the next wait would exceed the evidence-age bound', async () => { + const completedAt = now - 290_000 + const stale = sample() + stale.sources['cloud-monitoring']!.observedAt = new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() + const collect = vi.fn(async () => stale) + const wait = vi.fn(async () => undefined) + await expect(runIncidentLivePreflight( + ['--state-file', stateFile('strict', { + startedAt: new Date(completedAt - 17 * 60_000).toISOString(), + windowStartedAt: new Date(completedAt - 16 * 60_000).toISOString(), + lastSampleAt: new Date(completedAt - 30_000).toISOString(), + completedAt: new Date(completedAt).toISOString() + }), '--retry-freshness'], + { now: () => now, collect, wait } + )).rejects.toThrow('cloud-monitoring/source_stale') + expect(collect).toHaveBeenCalledOnce() + expect(wait).not.toHaveBeenCalled() + }) + it('does not retry a threshold failure', async () => { const unhealthy = sample() unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.value = 0.9 unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = - new Date(now - 180_001).toISOString() + new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() const collect = vi.fn(async () => unhealthy) const wait = vi.fn(async () => undefined) await expect(runIncidentLivePreflight( @@ -348,7 +384,7 @@ describe('relay incident live preflight', () => { it('fails closed after the bounded freshness retry window', async () => { const stale = sample() - stale.sources['cloud-monitoring']!.observedAt = new Date(now - 180_001).toISOString() + stale.sources['cloud-monitoring']!.observedAt = new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() const collect = vi.fn(async () => stale) const wait = vi.fn(async () => undefined) await expect(runIncidentLivePreflight( diff --git a/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts b/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts index e82627a3e80..2fcce3ed85d 100644 --- a/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts +++ b/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts @@ -7,6 +7,7 @@ import { suppliedIdentityToken } from './incident-monitor-cli.js' import { AdmissionSelectorSchema, type AdmissionSelector } from './incident-selector.js' import { evaluateIncidentSample, + FRESHNESS_FAILURE_CODES, preDrainDryRunPassed, type IncidentSample } from './incident-monitor.js' @@ -18,12 +19,6 @@ const MONITOR_EVIDENCE_MAX_AGE_MS = 5 * 60_000 // Matches the same-cap cell job timeout-minutes; bounds each predecessor wave. const WAVE_PREDECESSOR_TIMEOUT_MS = 75 * 60_000 const WAVE_INDEX_PATTERN = /^[0-3]$/ -const FRESHNESS_FAILURE_CODES = new Set([ - 'signal_missing', - 'signal_stale', - 'source_missing', - 'source_stale' -]) export function livePreflightGcloud( gcloud: ReturnType, @@ -173,7 +168,11 @@ export async function runIncidentLivePreflight( const freshnessOnly = evaluation.failures.every((failure) => FRESHNESS_FAILURE_CODES.has(failure.code) ) - if (!freshnessOnly || attempt === attempts) { + // Waiting must never carry the mutation past the same evidence-age bound + // the entry check enforces, so the wave budget also caps the retry window. + const budgetExhausted = + now() + FRESHNESS_RETRY_INTERVAL_MS - completedAt > maxEvidenceAgeMs + if (!freshnessOnly || attempt === attempts || budgetExhausted) { throw new Error( `relay live preflight failed: ${evaluation.failures .map((failure) => `${failure.source}/${failure.code}`) diff --git a/cloud/apps/relay-ops/src/incident-monitor-cli.ts b/cloud/apps/relay-ops/src/incident-monitor-cli.ts index e090be7ea58..adfe6cad480 100644 --- a/cloud/apps/relay-ops/src/incident-monitor-cli.ts +++ b/cloud/apps/relay-ops/src/incident-monitor-cli.ts @@ -50,6 +50,8 @@ const StateSchema = z.object({ continuityEvents: z.array(z.object({ recordedAt: z.string(), windowSequence: z.number().int().nonnegative(), + // Pre-2026-09-05 state files predate tolerated freshness gaps. + tolerated: z.boolean().default(false), failures: z.array(z.object({ code: z.string(), source: z.enum(['active-probe', 'cloud-monitoring', 'relay-logs', 'director-admin']), diff --git a/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts b/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts index 09b7b16fa45..74054b6c0ba 100644 --- a/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts +++ b/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts @@ -93,7 +93,7 @@ describe('incident monitor sources', () => { }) it('zero-fills an expired sparse lock-wait point', async () => { - let pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + let pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs const fetchImpl: typeof fetch = async () => Response.json({ timeSeries: [{ points: [{ @@ -141,7 +141,7 @@ describe('incident monitor sources', () => { it('freshens a sparse zero without masking a recent nonzero lock wait', async () => { let value = 0 - const pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + const pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs const readAt = now + 11_879 const fetchImpl: typeof fetch = async () => Response.json({ timeSeries: [{ diff --git a/cloud/apps/relay-ops/src/incident-monitor-sources.ts b/cloud/apps/relay-ops/src/incident-monitor-sources.ts index 0b78c2c4f6b..a97bfe3df43 100644 --- a/cloud/apps/relay-ops/src/incident-monitor-sources.ts +++ b/cloud/apps/relay-ops/src/incident-monitor-sources.ts @@ -95,7 +95,7 @@ export const GOOGLE_METRICS: GoogleMetricDefinition[] = [ 'resource.type="cloudsql_database" AND metric.label."wait_event_type"="Lock"', aggregation: 'latest-max', emptyIsZero: true, - zeroAfterMs: INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + zeroAfterMs: INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs }, { signal: 'cloud_sql.deadlocks', diff --git a/cloud/apps/relay-ops/src/incident-monitor.test.ts b/cloud/apps/relay-ops/src/incident-monitor.test.ts index 4e1da9fab26..076cff3de3b 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.test.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from 'vitest' import { evaluateIncidentSample, INCIDENT_CHECKPOINT_MINUTES, + INCIDENT_FRESHNESS_TOLERANCE_SAMPLES, INCIDENT_MONITOR_THRESHOLDS, INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS, initialIncidentMonitorState, @@ -182,12 +183,49 @@ describe('incident monitor evaluator', () => { code: 'source_missing', source: 'relay-logs' }) - const stale = healthySample(startedAt - 180_001) + const stale = healthySample( + startedAt - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) const failures = evaluateIncidentSample(stale, startedAt).failures expect(failures.some((failure) => failure.source === 'cloud-monitoring')).toBe(true) expect(failures.some((failure) => failure.source === 'active-probe')).toBe(true) }) + // Why: production run 33944873727 at 2026-09-05T04:46:09Z read + // cloud_sql.lock_waits 189 286 ms old and restarted a 15-minute window on + // Google's publish lag. Cloud SQL documents 60 s sampling plus up to 165 s of + // invisibility, so that age is Google's clock, not our fleet. + it('reads a 189-second cloud signal as fresh and holds the other sources at 180 s', () => { + const lagged = healthySample() + lagged.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, startedAt - 189_286) + expect(evaluateIncidentSample(lagged, startedAt)).toMatchObject({ + status: 'green', + failures: [] + }) + const laggedDirector = healthySample() + laggedDirector.sources['director-admin']!.observedAt = + new Date(startedAt - 189_286).toISOString() + expect(evaluateIncidentSample(laggedDirector, startedAt).failures).toContainEqual( + expect.objectContaining({ code: 'source_stale', source: 'director-admin' }) + ) + }) + + it('still fails a cloud signal past the documented publish lag', () => { + const dark = healthySample() + dark.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = signal( + 0, + startedAt - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) + expect(evaluateIncidentSample(dark, startedAt).failures).toContainEqual( + expect.objectContaining({ + code: 'signal_stale', + source: 'cloud-monitoring', + signal: 'cloud_sql.lock_waits' + }) + ) + }) + it('freezes on SQL, director, relay pool, heartbeat, and migration breaches', () => { const sample = healthySample() sample.sources['cloud-monitoring']!.signals['cloud_sql.cpu'] = signal(0.81) @@ -592,7 +630,7 @@ describe('incident monitor lifecycle', () => { 'restarts a %i-minute continuous window after stale telemetry', async (durationMinutes) => { let now = startedAt - let staleInjected = false + let staleSamples = INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1 const checkpoints: Array<[number, number]> = [] const state = initialIncidentMonitorState({ incidentId: 'incident-1', @@ -612,9 +650,11 @@ describe('incident monitor lifecycle', () => { now += ms }, collect: async () => { - if (!staleInjected && now === startedAt + 5 * 60_000) { - staleInjected = true - return healthySample(now - 180_001) + if (staleSamples > 0 && now >= startedAt + 5 * 60_000) { + staleSamples-- + return healthySample( + now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) } return healthySample(now) }, @@ -623,16 +663,20 @@ describe('incident monitor lifecycle', () => { checkpoints.push([summary.windowSequence, summary.checkpointMinute]) } }) + const restartMinute = 5 + INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1 expect(result.windowSequence).toBe(1) expect(result.windowStartedAt).toBe( - new Date(startedAt + 6 * 60_000).toISOString() + new Date(startedAt + restartMinute * 60_000).toISOString() ) expect(result.completedAt).toBe( - new Date(startedAt + (durationMinutes + 6) * 60_000).toISOString() + new Date(startedAt + (durationMinutes + restartMinute) * 60_000).toISOString() ) expect(result.sampleCount).toBe(durationMinutes + 1) - expect(result.continuityEvents).toHaveLength(1) - expect(result.continuityEvents[0]!.failures).toEqual( + expect(result.continuityEvents.map((event) => event.tolerated)).toEqual([ + ...Array(INCIDENT_FRESHNESS_TOLERANCE_SAMPLES).fill(true), + false + ]) + expect(result.continuityEvents.at(-1)!.failures).toEqual( expect.arrayContaining([ expect.objectContaining({ code: 'source_stale' }) ]) @@ -642,6 +686,188 @@ describe('incident monitor lifecycle', () => { } ) + // Why: run 33944873727 on 2026-09-05 restarted at 04:46:09Z on a single + // 189-second cloud reading and then blew the 25-minute lineage cap, so a + // green fleet produced no verdict at all. One unread sample now continues the + // window; the sample is still checked against every threshold it can read. + it('carries a 15-minute window through a single stale cloud sample', async () => { + let now = startedAt + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => { + const sample = healthySample(now) + if (now === startedAt + 10 * 60_000) { + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + } + return sample + }, + persist: async () => {}, + checkpoint: async () => {} + }) + expect(result.windowSequence).toBe(0) + expect(result.windowStartedAt).toBe(new Date(startedAt).toISOString()) + expect(result.completedAt).toBe(new Date(startedAt + 15 * 60_000).toISOString()) + expect(result.sampleCount).toBe(16) + expect(result.frozenAt).toBeNull() + expect(result.continuityEvents).toEqual([{ + recordedAt: new Date(startedAt + 10 * 60_000).toISOString(), + windowSequence: 0, + tolerated: true, + failures: [expect.objectContaining({ + code: 'signal_stale', + source: 'cloud-monitoring', + signal: 'cloud_sql.lock_waits' + })] + }]) + expect(preDrainDryRunPassed(result)).toBe(true) + }) + + it('gives a signal a fresh budget only after it reads fresh again', async () => { + let now = startedAt + const staleMinutes = new Set([3, 5, 6, 9, 10]) + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => { + const sample = healthySample(now) + if (staleMinutes.has((now - startedAt) / 60_000)) { + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + } + return sample + }, + persist: async () => {}, + checkpoint: async () => {} + }) + expect(result.windowSequence).toBe(0) + expect(result.continuityEvents).toHaveLength(staleMinutes.size) + expect(result.continuityEvents.every((event) => event.tolerated)).toBe(true) + expect(preDrainDryRunPassed(result)).toBe(true) + }) + + it('does not hand a resumed monitor a fresh tolerance budget', async () => { + let now = startedAt + 3 * 60_000 + const resumed = { + ...initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }), + windowStartedAt: new Date(startedAt).toISOString(), + lastSampleAt: new Date(startedAt + 2 * 60_000).toISOString(), + sampleCount: 3, + totalSampleCount: 3, + continuityEvents: Array.from( + { length: INCIDENT_FRESHNESS_TOLERANCE_SAMPLES }, + (_, index) => ({ + recordedAt: new Date(startedAt + (index + 1) * 60_000).toISOString(), + windowSequence: 0, + tolerated: true, + failures: [{ + code: 'signal_stale', + source: 'cloud-monitoring' as const, + signal: 'cloud_sql.lock_waits' + }] + }) + ) + } + const stop = new Error('stop after the resumed sample') + await expect(runIncidentMonitor(resumed, { + now: () => now, + wait: async () => { + throw stop + }, + collect: async () => { + const sample = healthySample(now) + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + return sample + }, + persist: async (state) => { + expect(state.windowSequence).toBe(1) + expect(state.windowStartedAt).toBeNull() + expect(state.continuityEvents.at(-1)!.tolerated).toBe(false) + }, + checkpoint: async () => {} + })).rejects.toThrow(stop) + }) + + it('freezes on a threshold breach that arrives with a tolerated stale signal', async () => { + let now = startedAt + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => { + const sample = healthySample(now) + if (now === startedAt + 2 * 60_000) { + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + sample.sources['cloud-monitoring']!.signals['cloud_sql.cpu'] = signal(0.81, now) + } + return sample + }, + persist: async () => {}, + checkpoint: async () => {} + }) + expect(result.frozenAt).toBe(new Date(startedAt + 2 * 60_000).toISOString()) + expect(result.failures).toContainEqual(expect.objectContaining({ + code: 'threshold_max', + signal: 'cloud_sql.cpu' + })) + expect(preDrainDryRunPassed(result)).toBe(false) + }) + it('resets at the next fresh sample after a runner gap', async () => { let now = startedAt + 10 * 60_000 const state = { @@ -690,13 +916,21 @@ describe('incident monitor lifecycle', () => { durationMinutes: 15, intervalMs: 60_000 }) + let staleSamples = INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1 const result = await runIncidentMonitor(state, { now: () => now, wait: async (ms) => { now += ms }, - collect: async () => - healthySample(now === startedAt + 10 * 60_000 ? now - 180_001 : now), + collect: async () => { + if (staleSamples > 0 && now >= startedAt + 10 * 60_000) { + staleSamples-- + return healthySample( + now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) + } + return healthySample(now) + }, persist: async () => {}, checkpoint: async () => {} }) @@ -706,7 +940,7 @@ describe('incident monitor lifecycle', () => { ) expect(result.frozenAt).not.toBeNull() expect(result.windowSequence).toBe(1) - expect(result.sampleCount).toBe(15) + expect(result.sampleCount).toBe(13) expect(result.failures).toContainEqual({ code: 'continuity_deadline_exceeded', source: 'active-probe', diff --git a/cloud/apps/relay-ops/src/incident-monitor.ts b/cloud/apps/relay-ops/src/incident-monitor.ts index a121568d918..2785eb573af 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.ts @@ -6,7 +6,27 @@ import { export const INCIDENT_MONITOR_THRESHOLDS = { activeProbeMaxAgeMs: 60_000, - cloudDataMaxAgeMs: 180_000, + // Why: Cloud Monitoring publishes on Google's clock, not ours. Per the metric + // list read 2026-09-05, Cloud Run instance_count / cpu / memory / + // max_request_concurrencies / request_count are "Sampled every 60 seconds. + // After sampling, data is not visible for up to 120 seconds" (60+120=180 s), + // and Cloud SQL cpu / memory / num_backends / backends_in_wait / + // deadlock_count say "up to 165 seconds" (60+165=225 s). Window-sum signals + // age differently: observedAt is the newest point in the 5-minute query + // window, so a label series that stops emitting reads as 300 s old while its + // summed value is still complete. 330 s clears the worst of the three (the + // 300 s query window) plus ~30 s of collect-to-evaluate latency. The old + // 180 s bar restarted healthy 15-minute windows at 181 s, 189 s and 255 s on + // 2026-09-04/05, once burning the whole 25-minute lineage with no verdict. + cloudDataMaxAgeMs: 330_000, + // Why: the director admin API answers live on our own request, so hold its + // freshness bar where it sat while it shared cloudDataMaxAgeMs. + directorAdminMaxAgeMs: 180_000, + // Why: how long a nonzero backends-in-wait point is carried before it reads as + // zero. Held at the pre-2026-09-05 cloud bar: carrying it for the full + // cloudDataMaxAgeMs would hand the evaluator a point older than its own + // freshness bar as soon as collection latency is added. + cloudLockWaitCarryMs: 180_000, relayLogMaxAgeMs: 180_000, heartbeatMaxAgeMs: 45_000, endpointLatencyMs: 2_000, @@ -175,6 +195,7 @@ export type IncidentMonitorState = { continuityEvents: { recordedAt: string windowSequence: number + tolerated: boolean failures: IncidentFailure[] }[] frozenAt: string | null @@ -307,7 +328,7 @@ const SOURCE_MAX_AGE: Record = { 'active-probe': INCIDENT_MONITOR_THRESHOLDS.activeProbeMaxAgeMs, 'cloud-monitoring': INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs, 'relay-logs': INCIDENT_MONITOR_THRESHOLDS.relayLogMaxAgeMs, - 'director-admin': INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 'director-admin': INCIDENT_MONITOR_THRESHOLDS.directorAdminMaxAgeMs } function ageMs(timestamp: string, nowMs: number): number { @@ -608,14 +629,59 @@ function checkpointMinutes(durationMinutes: number): number[] { return INCIDENT_CHECKPOINT_MINUTES.filter((minute) => minute <= durationMinutes) } -const CONTINUITY_FAILURE_CODES = new Set([ - 'collector_failed', - 'monitor_gap', +// Freshness-only failures: we could not read a signal this sample. Distinct from +// collector_failed / monitor_gap, where the whole sample is absent. +export const FRESHNESS_FAILURE_CODES = new Set([ + 'signal_missing', 'signal_stale', 'source_missing', 'source_stale' ]) +const CONTINUITY_FAILURE_CODES = new Set([ + 'collector_failed', + 'monitor_gap', + ...FRESHNESS_FAILURE_CODES +]) + +// Why: Cloud Monitoring overshoots its own publish bar, and one unread sample is +// not evidence of an unhealthy fleet. Under the 25-minute lineage cap a restart +// past minute 10 costs the entire verdict, so a healthy fleet produced none on +// 2026-09-05. A signal may miss this many consecutive samples before the window +// restarts; the sample is still evaluated against every threshold it can read, +// and a threshold breach still freezes the run outright. +export const INCIDENT_FRESHNESS_TOLERANCE_SAMPLES = 2 + +function freshnessKey(failure: IncidentFailure): string { + return `${failure.source}/${failure.signal ?? '*'}` +} + +// Rebuild the per-signal tolerated streak from the trailing continuity events so a +// resumed monitor cannot hand a signal a fresh budget. +function resumeFreshnessStreaks( + state: IncidentMonitorState +): Map { + const events = state.continuityEvents + const streaks = new Map() + const last = events[events.length - 1] + if (!last?.tolerated) return streaks + for (const key of new Set(last.failures.map(freshnessKey))) { + let streak = 0 + let laterAt: number | null = null + for (let index = events.length - 1; index >= 0; index--) { + const event = events[index]! + const recordedAt = Date.parse(event.recordedAt) + if (!event.tolerated) break + if (laterAt !== null && laterAt - recordedAt > state.intervalMs * 1.5) break + if (!event.failures.some((failure) => freshnessKey(failure) === key)) break + streak++ + laterAt = recordedAt + } + streaks.set(key, streak) + } + return streaks +} + function resetContinuousWindow( state: IncidentMonitorState, recordedAt: string, @@ -631,6 +697,7 @@ function resetContinuousWindow( state.continuityEvents.push({ recordedAt, windowSequence: state.windowSequence, + tolerated: false, failures }) } @@ -681,6 +748,7 @@ export async function runIncidentMonitor( await dependencies.persist(state) return state } + const freshnessStreaks = resumeFreshnessStreaks(state) while (state.completedAt === null) { if (dependencies.now() > lineageDeadlineMs) { completeContinuityDeadline(state, dependencies.now(), lineageStartMs) @@ -715,9 +783,34 @@ export async function runIncidentMonitor( const thresholdFailures = evaluation.failures.filter((failure) => !CONTINUITY_FAILURE_CODES.has(failure.code) ) - if (continuityFailures.length > 0) { + const toleratedKeys = new Set( + state.windowStartedAt !== null && + continuityFailures.length > 0 && + continuityFailures.every((failure) => FRESHNESS_FAILURE_CODES.has(failure.code)) + ? continuityFailures.map(freshnessKey) + : [] + ) + for (const key of [...freshnessStreaks.keys()]) { + if (!toleratedKeys.has(key)) freshnessStreaks.delete(key) + } + let tolerated = toleratedKeys.size > 0 + for (const key of toleratedKeys) { + const streak = (freshnessStreaks.get(key) ?? 0) + 1 + freshnessStreaks.set(key, streak) + if (streak > INCIDENT_FRESHNESS_TOLERANCE_SAMPLES) tolerated = false + } + if (continuityFailures.length > 0 && !tolerated) { + freshnessStreaks.clear() resetContinuousWindow(state, evaluation.evaluatedAt, continuityFailures) } else { + if (tolerated) { + state.continuityEvents.push({ + recordedAt: evaluation.evaluatedAt, + windowSequence: state.windowSequence, + tolerated: true, + failures: continuityFailures + }) + } if (state.windowStartedAt === null) { state.windowStartedAt = evaluation.evaluatedAt } diff --git a/cloud/apps/relay-ops/src/resource-inventory.test.ts b/cloud/apps/relay-ops/src/resource-inventory.test.ts index 4bf43fe9a7a..e2cfa13dccb 100644 --- a/cloud/apps/relay-ops/src/resource-inventory.test.ts +++ b/cloud/apps/relay-ops/src/resource-inventory.test.ts @@ -13,6 +13,41 @@ const runService = { latestReadyRevision: 'projects/project/revisions/revision-one' } +const sleepingStagingGcloud: GcloudClient = { accessToken: async () => 'a'.repeat(40) } + +// Staging's Cloud SQL is stopped, so this inventory reads REST only and probes no endpoint. +type MigOutcome = 'ok' | 'throw' | 'missing' +const sleepingStagingFetch = (migOutcome: (migName: string) => MigOutcome): typeof fetch => + async (input) => { + const url = new URL(String(input)) + if (url.hostname === 'run.googleapis.com') return Response.json(runService) + if (url.hostname === 'sqladmin.googleapis.com') return Response.json({ + state: 'STOPPED', + databaseVersion: 'POSTGRES_17', + settings: { activationPolicy: 'NEVER', availabilityType: 'ZONAL', tier: 'db-custom-1-3840' } + }) + if (url.hostname === 'certificatemanager.googleapis.com') return Response.json({ + managed: { domains: ['*.relay-staging.onorca.dev'], state: 'ACTIVE' } + }) + if (url.pathname.includes('/instanceGroupManagers/')) { + const name = url.pathname.split('/').at(-1)! + const outcome = migOutcome(name) + if (outcome === 'throw') throw new TypeError('fetch failed') + if (outcome === 'missing') return new Response(null, { status: 404 }) + return Response.json({ + name, + targetSize: 0, + size: '0', + instanceGroup: `projects/project/zones/zone/instanceGroups/${name}`, + instanceTemplate: `projects/project/global/instanceTemplates/template-${name}`, + status: { isStable: true } + }) + } + if (url.pathname.includes('/instanceTemplates/')) return Response.json({ properties: {} }) + if (url.pathname.endsWith('/getHealth')) return Response.json([]) + throw new Error(`Unexpected request to ${url.hostname}${url.pathname}`) + } + describe('readResourceInventory', () => { it('does not delay a healthy endpoint sample', async () => { let calls = 0 @@ -249,6 +284,74 @@ describe('readResourceInventory', () => { expect(JSON.stringify(result)).not.toContain('SECRET_TEXT') }) + it('re-asks a MIG read that failed once before calling a cell powered-unknown', async () => { + const parkedCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]! + const waits: number[] = [] + let parkedMigCalls = 0 + const result = await readResourceInventory( + RELAY_OPS_ENVIRONMENTS.staging, + sleepingStagingGcloud, + sleepingStagingFetch((migName) => { + if (!migName.endsWith(parkedCell.hostname)) return 'ok' + parkedMigCalls += 1 + return parkedMigCalls === 1 ? 'throw' : 'ok' + }), + { wait: async (ms) => { waits.push(ms) } } + ) + + const parked = result.cells.find((cell) => cell.cellId === parkedCell.cellId)! + // The MIG was fine and parked at zero; one transient read must not erase that reading. + expect(parked.targetSize).toBe(0) + expect(parkedMigCalls).toBe(2) + expect(waits).toEqual([1_000]) + expect(result.warnings).toEqual([]) + }) + + it('reports a MIG unavailable only when the retry fails too', async () => { + const parkedCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]! + const waits: number[] = [] + let parkedMigCalls = 0 + const result = await readResourceInventory( + RELAY_OPS_ENVIRONMENTS.staging, + sleepingStagingGcloud, + sleepingStagingFetch((migName) => { + if (!migName.endsWith(parkedCell.hostname)) return 'ok' + parkedMigCalls += 1 + return 'throw' + }), + { wait: async (ms) => { waits.push(ms) } } + ) + + const parked = result.cells.find((cell) => cell.cellId === parkedCell.cellId)! + expect(parked.targetSize).toBeNull() + expect(parked.backendHealth).toBe('unknown') + expect(parkedMigCalls).toBe(2) + expect(waits).toEqual([1_000]) + expect(result.warnings).toEqual([ + `${parkedCell.hostname.toUpperCase()} MIG inventory is unavailable.` + ]) + }) + + it('does not re-ask a MIG read the API answered with 404', async () => { + const missingCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]! + const waits: number[] = [] + let missingMigCalls = 0 + const result = await readResourceInventory( + RELAY_OPS_ENVIRONMENTS.staging, + sleepingStagingGcloud, + sleepingStagingFetch((migName) => { + if (!migName.endsWith(missingCell.hostname)) return 'ok' + missingMigCalls += 1 + return 'missing' + }), + { wait: async (ms) => { waits.push(ms) } } + ) + + expect(result.cells.find((cell) => cell.cellId === missingCell.cellId)!.targetSize).toBeNull() + expect(missingMigCalls).toBe(1) + expect(waits).toEqual([]) + }) + it('represents missing credentials as unknown inventory, never sleeping', async () => { const gcloud: GcloudClient = { accessToken: async () => { throw new Error('sensitive context') } diff --git a/cloud/apps/relay-ops/src/resource-inventory.ts b/cloud/apps/relay-ops/src/resource-inventory.ts index c1d01191baf..da490685198 100644 --- a/cloud/apps/relay-ops/src/resource-inventory.ts +++ b/cloud/apps/relay-ops/src/resource-inventory.ts @@ -103,6 +103,8 @@ export type ResourceInventory = { const unavailableEndpoint = (): EndpointHealth => ({ health: null, ready: null, latencyMs: null }) const independentEndpointRetryDelayMs = 11_000 const transientProbeRetryDelayMs = 1_000 +const sleep = async (ms: number): Promise => + await new Promise((resolvePromise) => setTimeout(resolvePromise, ms)) function finalSegment(value: string): string { return value.split('/').at(-1) ?? value @@ -121,6 +123,12 @@ function parseService(value: unknown): ServiceInventory { } } +class GoogleApiError extends Error { + constructor(readonly status: number) { + super(`Google API returned ${status}`) + } +} + async function googleRequest( fetchImpl: typeof fetch, token: string, @@ -135,10 +143,24 @@ async function googleRequest( }, signal: AbortSignal.timeout(30_000) }) - if (!response.ok) throw new Error(`Google API returned ${response.status}`) + if (!response.ok) throw new GoogleApiError(response.status) return await response.json() } +// A 404 is the API's answer about the resource; anything else is the absence of a reading, so re-ask. +async function readOnceMore( + read: () => Promise, + wait: (ms: number) => Promise +): Promise { + try { + return await read() + } catch (error) { + if (error instanceof GoogleApiError && error.status === 404) throw error + await wait(transientProbeRetryDelayMs) + return await read() + } +} + // A reading the endpoint actually produced: ok is its answer, latencyMs is that answer's round trip. type PathReading = { ok: boolean; latencyMs: number | null } @@ -201,8 +223,7 @@ export async function probeEndpointHealth( options: EndpointProbeOptions = {} ): Promise { const requiresReady = options.requiresReady ?? true - const wait = options.wait ?? - (async (ms: number) => await new Promise((resolvePromise) => setTimeout(resolvePromise, ms))) + const wait = options.wait ?? sleep const accepted = (probe: EndpointHealth): boolean => probe.health === true && (!requiresReady || probe.ready === true) && @@ -325,11 +346,17 @@ function unavailableInventory(environment: RelayOpsEnvironment, warning: string) } } +export type ResourceInventoryOptions = { + wait?: (ms: number) => Promise +} + export async function readResourceInventory( environment: RelayOpsEnvironment, gcloud: GcloudClient, - fetchImpl: typeof fetch = fetch + fetchImpl: typeof fetch = fetch, + options: ResourceInventoryOptions = {} ): Promise { + const wait = options.wait ?? sleep let token: string try { token = await gcloud.accessToken() @@ -356,7 +383,10 @@ export async function readResourceInventory( token, `https://certificatemanager.googleapis.com/v1/projects/${environment.project}/locations/global/certificates/${environment.certificateName}` ), - ...environment.cells.map((cell) => googleRequest(fetchImpl, token, migUrl(cell))) + // One transient Compute read must never become a verdict on a cell's power state. + ...environment.cells.map((cell) => + readOnceMore(async () => await googleRequest(fetchImpl, token, migUrl(cell)), wait) + ) ]) const warnings: string[] = [] const directorValue = parsed(settled[0]!, RunServiceSchema, 'Director service inventory is unavailable.', warnings) diff --git a/cloud/dev/scripts/operate-relay-regional-rehome.mjs b/cloud/dev/scripts/operate-relay-regional-rehome.mjs index 2ccce39926d..3887408520f 100644 --- a/cloud/dev/scripts/operate-relay-regional-rehome.mjs +++ b/cloud/dev/scripts/operate-relay-regional-rehome.mjs @@ -1,4 +1,5 @@ import { pathToFileURL } from 'node:url' +import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs' import { inspectAdmissionSelector } from './relay-admission-selector.mjs' const DIRECTOR_ORIGIN = 'https://relay.onorca.dev' @@ -229,15 +230,20 @@ export async function recoverRegionalRehomeEnable(config, post) { export async function operateRegionalRehome(config, dependencies = {}) { const fetchImpl = dependencies.fetch ?? fetch const post = dependencies.post ?? (async (path, body) => await responseJson( - await fetchImpl(`${config.directorOrigin}${path}`, { - method: 'POST', - headers: { - authorization: `Bearer ${config.token}`, - 'content-type': 'application/json' + // Generation-guarded writes make a retry a no-op or an explicit mismatch, never a double apply. + await fetchAdminOnceMore( + fetchImpl, + `${config.directorOrigin}${path}`, + { + method: 'POST', + headers: { + authorization: `Bearer ${config.token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify(body) }, - body: JSON.stringify(body), - signal: AbortSignal.timeout(30_000) - }), + { wait: dependencies.wait } + ), path )) if (config.mode === 'recover-enable') { diff --git a/cloud/dev/scripts/operate-relay-regional-rehome.test.mjs b/cloud/dev/scripts/operate-relay-regional-rehome.test.mjs index 8ffe38dfe09..bfea6769ec4 100644 --- a/cloud/dev/scripts/operate-relay-regional-rehome.test.mjs +++ b/cloud/dev/scripts/operate-relay-regional-rehome.test.mjs @@ -263,3 +263,55 @@ test('main executes recovery mode and emits verified disabled control', async () control: control(6, false) }) }) + +test('retries a transient 503 on the director control endpoint', async () => { + const config = parseRegionalRehomeArguments( + argumentsFor('inspect'), + { ORCA_RELAY_ADMIN_ID_TOKEN: 'token' } + ) + const paths = [] + let selectorCalls = 0 + const result = await operateRegionalRehome(config, { + wait: async () => {}, + fetch: async (url) => { + const path = new URL(url).pathname + paths.push(path) + if (path === '/v1/admin/admission-selector/status') { + selectorCalls += 1 + // The first read of each admin path 503s the way a warming instance does. + if (selectorCalls === 1) return new Response('warming up', { status: 503 }) + return Response.json({ selector: { generation: 11, membership } }) + } + if (paths.filter((value) => value === path).length === 1) { + return new Response('warming up', { status: 503 }) + } + return Response.json({ v: 1, control: control(4, false) }) + } + }) + assert.equal(result.control.generation, 4) + assert.deepEqual(paths, [ + '/v1/admin/admission-selector/status', + '/v1/admin/admission-selector/status', + '/v1/admin/regional-rehome-control', + '/v1/admin/regional-rehome-control' + ]) +}) + +test('fails when both attempts at the director control endpoint return 503', async () => { + const config = parseRegionalRehomeArguments( + argumentsFor('inspect'), + { ORCA_RELAY_ADMIN_ID_TOKEN: 'token' } + ) + let calls = 0 + await assert.rejects( + operateRegionalRehome(config, { + wait: async () => {}, + fetch: async () => { + calls += 1 + return new Response('warming up', { status: 503 }) + } + }), + /returned 503/ + ) + assert.equal(calls, 2) +}) diff --git a/cloud/dev/scripts/prepare-relay-production-capacity-canary.mjs b/cloud/dev/scripts/prepare-relay-production-capacity-canary.mjs index 5791c9f20e6..7967d164e4b 100644 --- a/cloud/dev/scripts/prepare-relay-production-capacity-canary.mjs +++ b/cloud/dev/scripts/prepare-relay-production-capacity-canary.mjs @@ -1,10 +1,12 @@ import { pathToFileURL } from 'node:url' +import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs' import { applyExactAdmissionSelector, inspectAdmissionSelector, membershipWithStates, selectorCellState } from './relay-admission-selector.mjs' +import { SAME_CAP_CELLS } from './relay-production-same-cap-wave.mjs' const DIRECTOR_ORIGIN = 'https://relay.onorca.dev' export const PRODUCTION_CAPACITY_CELL_IDS = [ @@ -30,6 +32,9 @@ function cellOrigin(cellId) { return `https://${cellId.slice('production-gce-'.length)}.relay.onorca.dev` } +// The same-cap roll covers the Asia cells the US-only capacity rollout never touches. +const APPROVED_CELL_LISTS = { 'same-cap': SAME_CAP_CELLS } + export function parseProductionCapacityCellArguments(argv) { const values = {} for (let index = 0; index < argv.length; index += 2) { @@ -41,8 +46,15 @@ export function parseProductionCapacityCellArguments(argv) { if (!['isolate', 'drain', 'activate'].includes(values.mode)) { throw new Error('--mode must be isolate, drain, or activate') } + const approvedList = values['approved-cells'] + if (approvedList !== undefined && !APPROVED_CELL_LISTS[approvedList]) { + throw new Error('--approved-cells is not a known allowlist') + } + const approvedCellIds = approvedList === undefined + ? PRODUCTION_CAPACITY_CELL_IDS + : APPROVED_CELL_LISTS[approvedList] const cellId = values['cell-id'] - if (!PRODUCTION_CAPACITY_CELL_IDS.includes(cellId)) { + if (!approvedCellIds.includes(cellId)) { throw new Error('production capacity target is not approved') } const expectedCellOrigin = cellOrigin(cellId) @@ -72,12 +84,16 @@ export async function prepareProductionCapacityCell(config, overrides = {}) { if (!token || token.length > 8_192) throw new Error('admin identity token is unavailable') const postAt = async (origin, path, body) => await responseJson( - await fetchImpl(`${origin}${path}`, { - method: 'POST', - headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, - body: JSON.stringify(body), - signal: AbortSignal.timeout(30_000) - }), + await fetchAdminOnceMore( + fetchImpl, + `${origin}${path}`, + { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify(body) + }, + { wait: overrides.wait } + ), path ) const post = async (path, body) => await postAt(config.directorOrigin, path, body) diff --git a/cloud/dev/scripts/prepare-relay-production-capacity-canary.test.mjs b/cloud/dev/scripts/prepare-relay-production-capacity-canary.test.mjs index c5d0a9db3bc..274a60d2198 100644 --- a/cloud/dev/scripts/prepare-relay-production-capacity-canary.test.mjs +++ b/cloud/dev/scripts/prepare-relay-production-capacity-canary.test.mjs @@ -104,6 +104,47 @@ describe('production Relay capacity cell admission', () => { '--cell-id', 'production-gce-c7', '--mode', 'isolate' ]), /origin is not exact/) + assert.throws(() => parseProductionCapacityCellArguments([ + '--director-origin', 'https://relay.onorca.dev', + '--cell-origin', 'https://c27.relay.onorca.dev', + '--cell-id', 'production-gce-c27', + '--mode', 'isolate' + ]), /not approved/) + }) + + it('admits the same-cap Asia cells only under the same-cap allowlist', () => { + for (const cellId of ['production-gce-c27', 'production-gce-c28', 'production-gce-c29']) { + const hostname = cellId.slice('production-gce-'.length) + assert.deepEqual(parseProductionCapacityCellArguments([ + '--director-origin', 'https://relay.onorca.dev', + '--cell-origin', `https://${hostname}.relay.onorca.dev`, + '--cell-id', cellId, + '--approved-cells', 'same-cap', + '--mode', 'isolate' + ]), { + directorOrigin: 'https://relay.onorca.dev', + cellOrigin: `https://${hostname}.relay.onorca.dev`, + cellId, + mode: 'isolate' + }) + } + for (const cellId of ['production-gce-c17', 'production-gce-c18', 'production-gce-c30']) { + const hostname = cellId.slice('production-gce-'.length) + assert.throws(() => parseProductionCapacityCellArguments([ + '--director-origin', 'https://relay.onorca.dev', + '--cell-origin', `https://${hostname}.relay.onorca.dev`, + '--cell-id', cellId, + '--approved-cells', 'same-cap', + '--mode', 'isolate' + ]), /not approved/) + } + assert.throws(() => parseProductionCapacityCellArguments([ + '--director-origin', 'https://relay.onorca.dev', + '--cell-origin', 'https://c27.relay.onorca.dev', + '--cell-id', 'production-gce-c27', + '--approved-cells', 'every-cell', + '--mode', 'isolate' + ]), /not a known allowlist/) }) it('isolates only the selected cell without depending on its runtime', async () => { @@ -170,4 +211,42 @@ describe('production Relay capacity cell admission', () => { /irreversible/ ) }) + + it('retries a transient 503 on the cell drain endpoint', async () => { + let calls = 0 + const result = await prepareProductionCapacityCell( + { ...config, mode: 'drain' }, + { + token: 'token', + wait: async () => {}, + fetch: async (url) => { + assert.equal(new URL(url).pathname, '/v1/admin/drain') + calls += 1 + if (calls === 1) return response({ error: 'warming up' }, 503) + return response({ v: 1, draining: true }) + } + } + ) + assert.equal(calls, 2) + assert.deepEqual(result, { changed: false, drained: true }) + }) + + it('fails when both drain attempts return a transient 503', async () => { + let calls = 0 + await assert.rejects( + prepareProductionCapacityCell( + { ...config, mode: 'drain' }, + { + token: 'token', + wait: async () => {}, + fetch: async () => { + calls += 1 + return response({ error: 'warming up' }, 503) + } + } + ), + /returned 503/ + ) + assert.equal(calls, 2) + }) }) diff --git a/cloud/dev/scripts/probe-relay-rehome-trust.mjs b/cloud/dev/scripts/probe-relay-rehome-trust.mjs index 7500d8bd14c..9a7d505d4bb 100644 --- a/cloud/dev/scripts/probe-relay-rehome-trust.mjs +++ b/cloud/dev/scripts/probe-relay-rehome-trust.mjs @@ -1,4 +1,5 @@ import { pathToFileURL } from 'node:url' +import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs' const PRODUCTION_CELL = /^production-gce-c(?:7|8|9|10|13|14|15|16|19|20|21|22|23|24|25|26)$/ const DIRECTOR_ORIGIN = 'https://relay.onorca.dev' @@ -35,7 +36,8 @@ export function parseRehomeTrustProbeArguments(argv, environment = process.env) export async function probeRehomeTrust(config, dependencies = {}) { const fetchImpl = dependencies.fetch ?? fetch - const response = await fetchImpl( + const response = await fetchAdminOnceMore( + fetchImpl, `${config.directorOrigin}/v1/admin/regional-rehome-trust-probe`, { method: 'POST', @@ -47,9 +49,9 @@ export async function probeRehomeTrust(config, dependencies = {}) { v: 1, sourceCellId: config.cellId, sourceCellIncarnation: config.cellIncarnation - }), - signal: AbortSignal.timeout(30_000) - } + }) + }, + { wait: dependencies.wait } ) const body = await response.json().catch(() => ({})) if (!response.ok) { diff --git a/cloud/dev/scripts/probe-relay-rehome-trust.test.mjs b/cloud/dev/scripts/probe-relay-rehome-trust.test.mjs index 509e9d53c7d..7d7b2cd95ac 100644 --- a/cloud/dev/scripts/probe-relay-rehome-trust.test.mjs +++ b/cloud/dev/scripts/probe-relay-rehome-trust.test.mjs @@ -68,3 +68,46 @@ test('rejects partial or mismatched proof', async () => { /incomplete/ ) }) + +const provenProbe = { + v: 1, + dedicatedIdentity: { + firstOutcome: 'host-not-connected', + secondOutcome: 'host-not-connected', + accepted: true, + idempotent: true + }, + sharedRuntimeIdentityRejected: true, + proven: true +} + +test('retries a transient 503 on the trust probe and proves on the second answer', async () => { + const config = parseRehomeTrustProbeArguments(argv, environment) + let calls = 0 + const result = await probeRehomeTrust(config, { + wait: async () => {}, + fetch: async () => { + calls += 1 + if (calls === 1) return new Response('warming up', { status: 503 }) + return Response.json(provenProbe) + } + }) + assert.equal(calls, 2) + assert.equal(result.proven, true) +}) + +test('fails when both trust-probe attempts return a transient 503', async () => { + const config = parseRehomeTrustProbeArguments(argv, environment) + let calls = 0 + await assert.rejects( + probeRehomeTrust(config, { + wait: async () => {}, + fetch: async () => { + calls += 1 + return new Response('warming up', { status: 503 }) + } + }), + /returned 503/ + ) + assert.equal(calls, 2) +}) diff --git a/cloud/dev/scripts/relay-admin-endpoint-retry-workflow.test.mjs b/cloud/dev/scripts/relay-admin-endpoint-retry-workflow.test.mjs new file mode 100644 index 00000000000..196fff9edf3 --- /dev/null +++ b/cloud/dev/scripts/relay-admin-endpoint-retry-workflow.test.mjs @@ -0,0 +1,43 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import { test } from 'node:test' +import { fileURLToPath } from 'node:url' +import { relayWorkflowUrl } from './relay-repository.mjs' + +const WORKFLOWS = [ + 'deploy-relay-production-same-cap-job.yml', + 'operate-relay-production-rehome-job.yml' +] + +function workflow(name) { + return readFileSync(fileURLToPath(relayWorkflowUrl(name)), 'utf8') +} + +// A single transient 5xx from a warming instance behind the global load balancer +// must not fail a canary, so no admin endpoint may be read by a bare curl. +test('no admin endpoint is reached by a curl without a bounded retry', () => { + for (const name of WORKFLOWS) { + for (const invocation of workflow(name).split(/\bcurl\b/).slice(1)) { + const flags = invocation.split('\n }')[0] + assert.match(flags, /--retry 3 --retry-delay 2 --retry-connrefused/, name) + assert.match(flags, /--max-time 30/, name) + // --retry-all-errors would also retry 401, 403, and 409, which are final. + assert.doesNotMatch(flags, /--retry-all-errors/, name) + } + } +}) + +test('every retried admin request captures only the final attempt body', () => { + const job = workflow('deploy-relay-production-same-cap-job.yml') + // --fail-with-body writes every failed attempt to stdout, so a retried + // request must land in a file curl truncates per attempt. + assert.match(job, /--output "\$\{out\}"/) + assert.equal(job.split('admin_post() {').length - 1, 2) + for (const call of [ + /CURRENT_RUNTIME="\$\(admin_post current-runtime/, + /CURRENT_DIRECTOR_STATUS="\$\(admin_post current-cell-status/, + /TARGET_RUNTIME="\$\(admin_post target-runtime/, + /TARGET_DIRECTOR_STATUS="\$\(admin_post target-cell-status/ + ]) assert.match(job, call) + assert.doesNotMatch(job, /\$\(curl /) +}) diff --git a/cloud/dev/scripts/relay-admin-transient-retry.mjs b/cloud/dev/scripts/relay-admin-transient-retry.mjs new file mode 100644 index 00000000000..9995be96a8f --- /dev/null +++ b/cloud/dev/scripts/relay-admin-transient-retry.mjs @@ -0,0 +1,29 @@ +// A single transient 5xx (load-balancer warm-up behind a fresh instance) must not fail a +// deploy step. 4xx is never retried: auth and generation-mismatch answers are final. +const TRANSIENT_STATUSES = [500, 502, 503, 504] +const RETRY_DELAY_MS = 2_000 +const REQUEST_TIMEOUT_MS = 30_000 + +export function isTransientAdminStatus(status) { + return TRANSIENT_STATUSES.includes(status) +} + +// Each attempt gets its own timeout budget, so a reused signal cannot abort the retry. +export async function fetchAdminOnceMore(fetchImpl, url, init, overrides = {}) { + const wait = overrides.wait ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms))) + const timeoutMs = overrides.timeoutMs ?? REQUEST_TIMEOUT_MS + const retryDelayMs = overrides.retryDelayMs ?? RETRY_DELAY_MS + const attempt = async () => + await fetchImpl(url, { ...init, signal: AbortSignal.timeout(timeoutMs) }) + let response + try { + response = await attempt() + } catch { + await wait(retryDelayMs) + return await attempt() + } + if (!isTransientAdminStatus(response.status)) return response + await response.arrayBuffer?.().catch(() => undefined) + await wait(retryDelayMs) + return await attempt() +} diff --git a/cloud/dev/scripts/relay-admin-transient-retry.test.mjs b/cloud/dev/scripts/relay-admin-transient-retry.test.mjs new file mode 100644 index 00000000000..ec041084344 --- /dev/null +++ b/cloud/dev/scripts/relay-admin-transient-retry.test.mjs @@ -0,0 +1,130 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs' + +const url = 'https://relay.onorca.dev/v1/admin/cell-status' +const init = { method: 'POST', body: '{"v":1}' } + +function recordingWait(waits) { + return async (ms) => { waits.push(ms) } +} + +test('a single transient 5xx is retried and the second answer is returned', async () => { + const waits = [] + const statuses = [503, 200] + let calls = 0 + const response = await fetchAdminOnceMore( + async () => { + calls += 1 + const status = statuses.shift() + return new Response(JSON.stringify({ ok: status === 200 }), { status }) + }, + url, + init, + { wait: recordingWait(waits) } + ) + assert.equal(calls, 2) + assert.equal(response.status, 200) + assert.deepEqual(waits, [2_000]) + assert.deepEqual(await response.json(), { ok: true }) +}) + +test('a connection failure is retried and the second answer is returned', async () => { + const waits = [] + let calls = 0 + const response = await fetchAdminOnceMore( + async () => { + calls += 1 + if (calls === 1) throw new TypeError('fetch failed') + return Response.json({ ok: true }) + }, + url, + init, + { wait: recordingWait(waits) } + ) + assert.equal(calls, 2) + assert.equal(response.status, 200) + assert.deepEqual(waits, [2_000]) +}) + +test('two transient failures surface the second answer without a third attempt', async () => { + let calls = 0 + const response = await fetchAdminOnceMore( + async () => { + calls += 1 + return new Response('down', { status: 503 }) + }, + url, + init, + { wait: async () => {} } + ) + assert.equal(calls, 2) + assert.equal(response.status, 503) +}) + +test('two connection failures rethrow the second error', async () => { + let calls = 0 + await assert.rejects( + fetchAdminOnceMore( + async () => { + calls += 1 + throw new TypeError(`fetch failed ${calls}`) + }, + url, + init, + { wait: async () => {} } + ), + /fetch failed 2/ + ) + assert.equal(calls, 2) +}) + +test('4xx is final: auth and generation-mismatch answers are never retried', async () => { + for (const status of [400, 401, 403, 404, 409, 429]) { + let calls = 0 + const response = await fetchAdminOnceMore( + async () => { + calls += 1 + return new Response('no', { status }) + }, + url, + init, + { wait: async () => { throw new Error('must not wait') } } + ) + assert.equal(calls, 1, `status ${status} must not be retried`) + assert.equal(response.status, status) + } +}) + +test('each attempt carries its own unexpired timeout signal', async () => { + const signals = [] + await fetchAdminOnceMore( + async (_url, attemptInit) => { + signals.push(attemptInit.signal) + return new Response('down', { status: 502 }) + }, + url, + init, + { wait: async () => {}, timeoutMs: 30_000 } + ) + assert.equal(signals.length, 2) + assert.notEqual(signals[0], signals[1]) + assert.equal(signals[1].aborted, false) +}) + +test('the caller init is forwarded unchanged apart from the signal', async () => { + let seen + await fetchAdminOnceMore( + async (seenUrl, attemptInit) => { + seen = { seenUrl, attemptInit } + return Response.json({}) + }, + url, + { method: 'POST', headers: { authorization: 'Bearer t' }, body: '{"v":1}' }, + { wait: async () => {} } + ) + assert.equal(seen.seenUrl, url) + assert.equal(seen.attemptInit.method, 'POST') + assert.deepEqual(seen.attemptInit.headers, { authorization: 'Bearer t' }) + assert.equal(seen.attemptInit.body, '{"v":1}') +}) diff --git a/cloud/dev/scripts/relay-evidence-code-provenance.mjs b/cloud/dev/scripts/relay-evidence-code-provenance.mjs new file mode 100644 index 00000000000..233a8139b85 --- /dev/null +++ b/cloud/dev/scripts/relay-evidence-code-provenance.mjs @@ -0,0 +1,94 @@ +import { spawnSync } from 'node:child_process' +import { fileURLToPath } from 'node:url' +import { + RELAY_REPOSITORY_ROOT, + relayTreePath, + relayWorkflowPath +} from './relay-repository.mjs' + +const SHA = /^[a-f0-9]{40}$/ + +// Every file that decides how relay evidence is produced, sealed, verified, and then spent against +// production; identical content across two commits is what makes the older commit's verdict binding. +export const TRUSTED_EVIDENCE_CODE_PATHS = [ + // Produces and seals the 15-minute dry-run evidence. + relayWorkflowPath('monitor-relay-production.yml'), + relayWorkflowPath('monitor-relay-production-job.yml'), + // Download it, verify its authority, and mutate production on it. + relayWorkflowPath('deploy-relay-production-same-cap.yml'), + relayWorkflowPath('deploy-relay-production-same-cap-job.yml'), + relayWorkflowPath('operate-relay-production-rehome.yml'), + relayWorkflowPath('operate-relay-production-rehome-job.yml'), + // Sealing, verification, the wave/canary authority, and the path constants below. + relayTreePath('dev/scripts/relay-evidence-code-provenance.mjs'), + relayTreePath('dev/scripts/relay-monitor-evidence.mjs'), + relayTreePath('dev/scripts/relay-production-same-cap-wave.mjs'), + relayTreePath('dev/scripts/relay-repository.mjs'), + // Every other script those jobs run against live production. + relayTreePath('dev/scripts/infra.mjs'), + relayTreePath('dev/scripts/operate-relay-regional-rehome.mjs'), + relayTreePath('dev/scripts/prepare-relay-production-capacity-canary.mjs'), + relayTreePath('dev/scripts/probe-relay-rehome-trust.mjs'), + relayTreePath('dev/scripts/validate-relay-capacity-plan.mjs'), + relayTreePath('dev/scripts/verify-relay-capacity-transition.mjs'), + // The monitor itself and the live preflight recheck, plus anything that changes their behaviour. + relayTreePath('apps/relay-ops'), + relayTreePath('package.json'), + relayTreePath('pnpm-lock.yaml'), + relayTreePath('pnpm-workspace.yaml'), + // The Cloud SQL rollout lease every mutation job takes and releases. + '.github/actions/cloud-sql-rollout-lease' +] + +function git(root, args) { + const result = spawnSync('git', ['-C', root, ...args], { encoding: 'utf8' }) + if (result.error) throw new Error('relay evidence provenance cannot run git') + return result +} + +/** + * Accepts evidence sealed at a different commit only when the current commit descends from it and + * every trusted path is byte-identical, so the verdict provably came from this exact code. Anything + * git cannot answer (no checkout, unknown commit, shallow clone) fails closed. + */ +export function requireSameEvidenceCode({ + sealedSha, + currentSha, + label, + repositoryRoot = fileURLToPath(RELAY_REPOSITORY_ROOT) +}) { + if (!SHA.test(sealedSha ?? '') || !SHA.test(currentSha ?? '')) { + throw new Error(`${label} commit is invalid`) + } + if (sealedSha === currentSha) return + if (git(repositoryRoot, ['rev-parse', '--git-dir']).status !== 0) { + throw new Error(`${label} commit cannot be compared without a git checkout`) + } + for (const sha of [sealedSha, currentSha]) { + if (git(repositoryRoot, ['rev-parse', '--verify', '--quiet', `${sha}^{commit}`]).status !== 0) { + throw new Error( + `${label} commit ${sha} is unknown to this checkout; check out with fetch-depth: 0` + ) + } + } + const ancestry = git(repositoryRoot, ['merge-base', '--is-ancestor', sealedSha, currentSha]) + if (ancestry.status === 1) { + throw new Error(`${label} commit ${sealedSha} is not an ancestor of ${currentSha}`) + } + if (ancestry.status !== 0) { + throw new Error(`${label} commit ancestry could not be determined`) + } + const diff = git(repositoryRoot, [ + 'diff', + '--name-only', + sealedSha, + currentSha, + '--', + ...TRUSTED_EVIDENCE_CODE_PATHS + ]) + if (diff.status !== 0) throw new Error(`${label} commit comparison failed`) + const changed = diff.stdout.split('\n').filter(Boolean) + if (changed.length > 0) { + throw new Error(`${label} code changed after it was sealed: ${changed.join(',')}`) + } +} diff --git a/cloud/dev/scripts/relay-monitor-evidence.mjs b/cloud/dev/scripts/relay-monitor-evidence.mjs index 7f387663f60..26eb37d0d4d 100644 --- a/cloud/dev/scripts/relay-monitor-evidence.mjs +++ b/cloud/dev/scripts/relay-monitor-evidence.mjs @@ -2,6 +2,7 @@ import { createHash } from 'node:crypto' import { chmod, readFile, readdir, stat, writeFile } from 'node:fs/promises' import { basename, join, resolve } from 'node:path' import { pathToFileURL } from 'node:url' +import { requireSameEvidenceCode } from './relay-evidence-code-provenance.mjs' const SAFE_ID = /^[A-Za-z0-9][A-Za-z0-9._-]{1,127}$/ const SHA = /^[a-f0-9]{40}$/ @@ -102,7 +103,7 @@ export async function createEvidenceManifest(argv) { return manifest } -async function readAndVerifyManifest(directory, expected) { +async function readAndVerifyManifest(directory, expected, sameCodeCommit) { const manifest = JSON.parse( await readFile(join(directory, 'evidence-manifest.json'), 'utf8') ) @@ -111,11 +112,23 @@ async function readAndVerifyManifest(directory, expected) { manifest.incidentId !== expected.incidentId || manifest.runId !== expected.runId || manifest.runAttempt !== expected.runAttempt || - manifest.commitSha !== expected.commitSha || - manifest.mode !== expected.mode + !SHA.test(manifest.commitSha ?? '') || + manifest.mode !== expected.mode || + (!sameCodeCommit && manifest.commitSha !== expected.commitSha) ) { throw new Error('relay monitor evidence provenance does not match') } + // Unrelated merges land on main every few minutes, so the deployer resolves a newer commit than + // the monitor it must trust; identical monitor and mutation code is the property the SHA stood in + // for. Restore and mutation keep the exact-SHA bind: both run at the commit that sealed them. + if (sameCodeCommit) { + requireSameEvidenceCode({ + sealedSha: manifest.commitSha, + currentSha: expected.commitSha, + label: 'relay monitor evidence', + ...sameCodeCommit + }) + } const names = Object.keys(manifest.files ?? {}) if (!names.includes(`${expected.incidentId}.state.json`)) { throw new Error('relay monitor evidence has no durable state') @@ -209,12 +222,12 @@ function validCompletedDryRunState(state, expected, nowMs, maxAgeMs) { ) } -export async function verifyDryRunAuthority(argv, now = Date.now) { +export async function verifyDryRunAuthority(argv, now = Date.now, repositoryRoot) { const values = argumentsByName(argv) const directory = resolve(values.directory ?? '') const expected = provenance(values) if (expected.mode !== 'dry-run') throw new Error('relay mutation requires dry-run evidence') - const manifest = await readAndVerifyManifest(directory, expected) + const manifest = await readAndVerifyManifest(directory, expected, { repositoryRoot }) const state = JSON.parse( await readFile(join(directory, `${expected.incidentId}.state.json`), 'utf8') ) diff --git a/cloud/dev/scripts/relay-monitor-evidence.test.mjs b/cloud/dev/scripts/relay-monitor-evidence.test.mjs index 45116761119..43d2ac02763 100644 --- a/cloud/dev/scripts/relay-monitor-evidence.test.mjs +++ b/cloud/dev/scripts/relay-monitor-evidence.test.mjs @@ -1,9 +1,15 @@ import assert from 'node:assert/strict' -import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises' +import { execFileSync } from 'node:child_process' +import { mkdir, mkdtemp, readFile, rm, stat, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' -import { join } from 'node:path' +import { dirname, join } from 'node:path' import test from 'node:test' -import { relayWorkflowPath, relayWorkflowUrl } from './relay-repository.mjs' +import { TRUSTED_EVIDENCE_CODE_PATHS } from './relay-evidence-code-provenance.mjs' +import { + RELAY_REPOSITORY_ROOT, + relayWorkflowPath, + relayWorkflowUrl +} from './relay-repository.mjs' import { createEvidenceManifest, verifyDryRunAuthority, @@ -12,7 +18,7 @@ import { } from './relay-monitor-evidence.mjs' const now = Date.parse('2026-07-28T12:00:00.000Z') -const provenance = [ +const provenanceFor = (commitSha) => [ '--incident-id', 'relay-123', '--run-id', @@ -20,10 +26,11 @@ const provenance = [ '--run-attempt', '1', '--commit-sha', - 'a'.repeat(40), + commitSha, '--mode', 'dry-run' ] +const provenance = provenanceFor('a'.repeat(40)) const selector = { generation: 2, membership: { @@ -513,3 +520,157 @@ test('monitor uses a reusable job so exact job_workflow_ref is present', async ( assert.match(job, /workflow_call:/) assert.match(job, /environment: production/) }) + +function gitIn(root, ...args) { + return execFileSync('git', ['-C', root, ...args], { encoding: 'utf8' }).trim() +} + +// A real repository shaped like main under unrelated merge traffic: one sealed commit, a +// descendant that only touched untrusted files, a descendant that touched the monitor, and a +// sibling that never descended from the seal. +async function trustedCodeRepository() { + const root = await mkdtemp(join(tmpdir(), 'relay-evidence-repository-')) + gitIn(root, 'init', '--quiet') + gitIn(root, 'config', 'user.email', 'relay@example.test') + gitIn(root, 'config', 'user.name', 'Relay Evidence Test') + gitIn(root, 'config', 'commit.gpgsign', 'false') + const commit = async (path, body, message) => { + await mkdir(dirname(join(root, path)), { recursive: true }) + await writeFile(join(root, path), body) + gitIn(root, 'add', '--all') + gitIn(root, 'commit', '--quiet', '--no-verify', '--message', message) + return gitIn(root, 'rev-parse', 'HEAD') + } + const base = await commit( + 'cloud/apps/relay-ops/src/incident-monitor.ts', + 'export const v = 1\n', + 'monitor' + ) + const sealed = await commit('README.md', 'base\n', 'base') + const sameCode = await commit('README.md', 'an unrelated merge\n', 'unrelated') + const changedCode = await commit( + 'cloud/apps/relay-ops/src/incident-monitor.ts', + 'export const v = 2\n', + 'monitor change' + ) + // Branches before the seal, so the seal is not in its history even though its code matches. + gitIn(root, 'checkout', '--quiet', '--detach', base) + const sibling = await commit('README.md', 'a divergent line\n', 'divergent') + return { root, sealed, sameCode, changedCode, sibling } +} + +const authorityAt = (directory, commitSha, repositoryRoot) => verifyDryRunAuthority( + [ + '--directory', + directory, + ...provenanceFor(commitSha), + '--required-migration-policy', + 'strict' + ], + () => now, + repositoryRoot +) + +test('accepts dry-run evidence sealed by identical code at an ancestor commit', async () => { + const repository = await trustedCodeRepository() + const directory = await evidenceDirectory() + try { + await createEvidenceManifest([ + '--directory', + directory, + ...provenanceFor(repository.sealed) + ]) + // An exact match never consults git: a root with no checkout at all still verifies. + await assert.doesNotReject(authorityAt(directory, repository.sealed, directory)) + await assert.doesNotReject(authorityAt(directory, repository.sameCode, repository.root)) + } finally { + await rm(repository.root, { recursive: true, force: true }) + await rm(directory, { recursive: true, force: true }) + } +}) + +test('rejects dry-run evidence whose monitor code or lineage differs', async () => { + const repository = await trustedCodeRepository() + const directory = await evidenceDirectory() + try { + await createEvidenceManifest([ + '--directory', + directory, + ...provenanceFor(repository.sealed) + ]) + await assert.rejects( + authorityAt(directory, repository.changedCode, repository.root), + /code changed after it was sealed: cloud\/apps\/relay-ops\/src\/incident-monitor\.ts/ + ) + await assert.rejects( + authorityAt(directory, repository.sibling, repository.root), + /is not an ancestor of/ + ) + // Fails closed: a shallow clone that never fetched the sealed commit proves nothing. + await assert.rejects( + authorityAt(directory, 'f'.repeat(40), repository.root), + /unknown to this checkout/ + ) + // Fails closed: no checkout to compare against. + await assert.rejects( + authorityAt(directory, repository.sameCode, directory), + /cannot be compared without a git checkout/ + ) + } finally { + await rm(repository.root, { recursive: true, force: true }) + await rm(directory, { recursive: true, force: true }) + } +}) + +test('keeps restore and mutation bound to the exact sealing commit', async () => { + const repository = await trustedCodeRepository() + const directory = await evidenceDirectory() + try { + await createEvidenceManifest([ + '--directory', + directory, + ...provenanceFor(repository.sealed) + ]) + await assert.rejects( + verifyRestoredEvidence([ + '--directory', + directory, + ...provenanceFor(repository.sameCode) + ]), + /provenance does not match/ + ) + await assert.rejects( + verifyMutationEvidence( + [ + '--directory', + directory, + ...provenanceFor(repository.sameCode), + '--mutation-mode', + 'execute', + '--source-cell-id', + 'c1', + '--director-origin', + 'https://relay.example' + ], + { ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }, + async () => Response.json({ selector }), + () => now + ), + /provenance does not match/ + ) + } finally { + await rm(repository.root, { recursive: true, force: true }) + await rm(directory, { recursive: true, force: true }) + } +}) + +// A trusted path that no longer exists silently stops being compared, so the same-code rule would +// pass over code it was written to pin. +test('every trusted provenance path exists in this checkout', async () => { + for (const path of TRUSTED_EVIDENCE_CODE_PATHS) { + await assert.doesNotReject( + stat(new URL(path, RELAY_REPOSITORY_ROOT)), + `${path} is missing` + ) + } +}) diff --git a/cloud/dev/scripts/relay-production-same-cap-wave.mjs b/cloud/dev/scripts/relay-production-same-cap-wave.mjs index e391c4c4381..6e84c1c9104 100644 --- a/cloud/dev/scripts/relay-production-same-cap-wave.mjs +++ b/cloud/dev/scripts/relay-production-same-cap-wave.mjs @@ -1,5 +1,6 @@ import { readFileSync } from 'node:fs' import { pathToFileURL } from 'node:url' +import { requireSameEvidenceCode } from './relay-evidence-code-provenance.mjs' export const SAME_CAP_CELLS = [ 'production-gce-c7', 'production-gce-c8', 'production-gce-c9', 'production-gce-c10', @@ -85,10 +86,10 @@ export function canaryAuthority(input) { } } -export function verifyCanaryAuthority(authority, expected) { +export function verifyCanaryAuthority(authority, expected, repositoryRoot) { if ( authority?.v !== 1 || - authority.commitSha !== expected.commitSha || + !/^[0-9a-f]{40}$/.test(authority.commitSha ?? '') || authority.runId !== expected.runId || authority.targetDigest !== expected.targetDigest || authority.rollbackDigest !== expected.rollbackDigest || @@ -96,6 +97,14 @@ export function verifyCanaryAuthority(authority, expected) { authority.rehomeGeneration !== Number(expected.rehomeGeneration) || !SAME_CAP_CELLS.includes(authority.cellId) ) throw new Error('canary authority does not match this batch') + // The batch dispatch resolves main after the canary sealed, so bind to the same code, not the + // same SHA; every field above still pins this batch to that exact canary. + requireSameEvidenceCode({ + sealedSha: authority.commitSha, + currentSha: expected.commitSha, + label: 'relay same-cap canary authority', + repositoryRoot + }) return authority } diff --git a/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs b/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs index 0b45ae85a99..d636c324b33 100644 --- a/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs +++ b/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs @@ -1,4 +1,8 @@ import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { dirname, join } from 'node:path' import { test } from 'node:test' import { canaryAuthority, @@ -104,3 +108,66 @@ test('seals and verifies canary authority for later batches', () => { rehomeGeneration: '4' }), /does not match/) }) + +function gitIn(root, ...args) { + return execFileSync('git', ['-C', root, ...args], { encoding: 'utf8' }).trim() +} + +async function canaryRepository() { + const root = await mkdtemp(join(tmpdir(), 'relay-same-cap-canary-')) + gitIn(root, 'init', '--quiet') + gitIn(root, 'config', 'user.email', 'relay@example.test') + gitIn(root, 'config', 'user.name', 'Relay Wave Test') + gitIn(root, 'config', 'commit.gpgsign', 'false') + const commit = async (path, body, message) => { + await mkdir(dirname(join(root, path)), { recursive: true }) + await writeFile(join(root, path), body) + gitIn(root, 'add', '--all') + gitIn(root, 'commit', '--quiet', '--no-verify', '--message', message) + return gitIn(root, 'rev-parse', 'HEAD') + } + const sealed = await commit( + 'cloud/dev/scripts/relay-production-same-cap-wave.mjs', + 'export const v = 1\n', + 'wave' + ) + const sameCode = await commit('README.md', 'an unrelated merge\n', 'unrelated') + const changedCode = await commit( + 'cloud/dev/scripts/relay-production-same-cap-wave.mjs', + 'export const v = 2\n', + 'wave change' + ) + return { root, sealed, sameCode, changedCode } +} + +test('a batch trusts a canary sealed by identical code at an ancestor commit', async () => { + const repository = await canaryRepository() + try { + const authority = canaryAuthority({ + cellIds: 'production-gce-c7', + targetDigest, + rollbackDigest, + confirmation: `ROLL_RELAY_SAME_CAP ${targetDigest} production-gce-c7`, + commitSha: repository.sealed, + runId: '42', + selectorGeneration: '11', + rehomeGeneration: '4' + }) + const verifyAt = (commitSha, repositoryRoot) => verifyCanaryAuthority(authority, { + commitSha, + runId: '42', + targetDigest, + rollbackDigest, + selectorGeneration: '13', + rehomeGeneration: '4' + }, repositoryRoot) + assert.equal(verifyAt(repository.sameCode, repository.root).cellId, 'production-gce-c7') + assert.throws( + () => verifyAt(repository.changedCode, repository.root), + /code changed after it was sealed/ + ) + assert.throws(() => verifyAt('f'.repeat(40), repository.root), /unknown to this checkout/) + } finally { + await rm(repository.root, { recursive: true, force: true }) + } +}) diff --git a/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs b/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs index e31403f6dd6..a77ab93cf15 100644 --- a/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs +++ b/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs @@ -73,7 +73,10 @@ test('same-cap wrapper is reusable, canary-bound, and sequential', () => { job, /--rollback-image "\$\{DESIRED_IMAGE\}" \\\n {16}--rehome-director-service-account "\$\{DIRECTOR_RUNTIME_SERVICE_ACCOUNT\}"/ ) - assert.match(job, /host-drain \\\n {14}\| jq -e '\.changes == 2' >\/dev\/null/) + assert.match( + job, + /host-drain \\\n {16}--regional-rehome-protocol "\$\{DESIRED_REHOME_PROTOCOL\}" \\\n {14}\| jq -e '\.changes == 2' >\/dev\/null/ + ) assert.match(job, /resume requires the isolated migration-only cell/) assert.match(job, /test "\$\{TARGET_INCARNATION\}" = "\$\{SOURCE_INCARNATION\}"/) assert.match(job, /\(.regionalRehomeProtocol \/\/ 0\) == \$protocol/) @@ -92,7 +95,11 @@ test('same-cap wrapper is reusable, canary-bound, and sequential', () => { // age checks must scale by wave or cell_2+ can never pass; the bound's // per-wave step is the cell job timeout, so the two must move together. assert.match(job, /--required-migration-policy strict \\\n --wave-index "\$\{WAVE_INDEX\}"/) - assert.match(job, /dry-run\.state\.json" \\\n --wave-index "\$\{WAVE_INDEX\}" "\$\{RETRY_ARGS\[@\]\}"/) + // Wave 0 must retry freshness-only failures too: one Cloud Monitoring publish + // lag at the sample instant is not health evidence, and single-shot wave 0 + // failed a whole batch on a series that was fresh again a minute later. + assert.match(job, /dry-run\.state\.json" \\\n --wave-index "\$\{WAVE_INDEX\}" --retry-freshness/) + assert.doesNotMatch(job, /RETRY_ARGS/) assert.match(job, /timeout-minutes: 75/) // Both age gates step by the cell job timeout above; the constant is // duplicated across the two languages, so pin each copy to it. diff --git a/cloud/dev/scripts/relay-repository.mjs b/cloud/dev/scripts/relay-repository.mjs index 7e8b01e4799..040acef5fe5 100644 --- a/cloud/dev/scripts/relay-repository.mjs +++ b/cloud/dev/scripts/relay-repository.mjs @@ -1,4 +1,6 @@ import { readFileSync } from 'node:fs' +import { relative } from 'node:path' +import { fileURLToPath } from 'node:url' // Single place naming the repository the Relay workflows live in and where their files sit. The // public-repo copy moves this tree under cloud/, prefixes every workflow filename, and changes the @@ -11,6 +13,19 @@ export const RELAY_WORKFLOW_FILE_PREFIX = 'cloud-' // this tree moves under cloud/, so the depth changes at the copy even though the layout does not. export const RELAY_WORKFLOW_DIRECTORY = new URL('../../../.github/workflows/', import.meta.url) +// Repository root, derived from the one directory above that already tracks the copy's depth. +export const RELAY_REPOSITORY_ROOT = new URL('../../', RELAY_WORKFLOW_DIRECTORY) + +// Repository-relative path for a file in this tree. The prefix is 'cloud/' here and empty where +// the tree is the repository root, so callers naming git paths never restate the layout. +export function relayTreePath(suffix) { + const prefix = relative( + fileURLToPath(RELAY_REPOSITORY_ROOT), + fileURLToPath(new URL('../../', import.meta.url)) + ).split(/[\\/]/).filter(Boolean) + return [...prefix, suffix].join('/') +} + export function relayWorkflowFile(name) { return `${RELAY_WORKFLOW_FILE_PREFIX}${name}` } diff --git a/cloud/dev/scripts/relay-same-cap-script-census.test.mjs b/cloud/dev/scripts/relay-same-cap-script-census.test.mjs new file mode 100644 index 00000000000..743aef7fc2d --- /dev/null +++ b/cloud/dev/scripts/relay-same-cap-script-census.test.mjs @@ -0,0 +1,217 @@ +import assert from 'node:assert/strict' +import { spawnSync } from 'node:child_process' +import { readFileSync } from 'node:fs' +import { describe, it } from 'node:test' +import { parseProductionCapacityCellArguments } from './prepare-relay-production-capacity-canary.mjs' +import { SAME_CAP_CELLS } from './relay-production-same-cap-wave.mjs' +import { readRelayWorkflow } from './relay-repository.mjs' +import { validateCapacityPlan } from './validate-relay-capacity-plan.mjs' + +const workflow = readRelayWorkflow('deploy-relay-production-same-cap-job.yml') +const capacityWorkflow = readRelayWorkflow('deploy-relay-production-capacity-job.yml') +const production = readFileSync( + new URL('../../infra/terraform/environments/production.tfvars', import.meta.url), + 'utf8' +) +const REHOME_SOURCE_CELLS = rehomeSourceCells() +const DIRECTOR_IDENTITY = 'relay-director@onorca-cloud.iam.gserviceaccount.com' +const AUDIENCE = 'https://relay.onorca.dev/v1/admin/host-drain' +const ROLLBACK_IMAGE = `us-central1-docker.pkg.dev/p/orca-cloud/relay@sha256:${'d'.repeat(64)}` +const TARGET_IMAGE = `us-central1-docker.pkg.dev/p/orca-cloud/relay@sha256:${'e'.repeat(64)}` + +// The startup template emits rehome trust only for cells in this list, so it is what decides +// whether a cell's plan may carry those lines at all. +function rehomeSourceCells() { + const start = production.indexOf('relay_region_rehome_source_cell_ids = [') + assert.notEqual(start, -1, 'production.tfvars has no rehome source cell list') + const end = production.indexOf(']', start) + assert.notEqual(end, -1, 'the rehome source cell list is unterminated') + return new Set( + [...production.slice(start, end).matchAll(/"([^"]+)"/g)].map(([, cell]) => cell) + ) +} + +function startupScript({ cap, image, trusted }) { + return [ + ` printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '${cap}'`, + ` printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '60'`, + ...(trusted ? [ + ` printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '${DIRECTOR_IDENTITY}'`, + ` printf 'ORCA_RELAY_REHOME_AUDIENCE=%s\\n' '${AUDIENCE}'` + ] : []), + `printf 'ORCA_RELAY_IMAGE_DIGEST=%s\\n' '${image.split('@')[1]}'`, + `docker pull '${image}'`, + 'docker run --detach \\', + ' --name orca-relay \\', + ` '${image}'` + ].join('\n') +} + +// The exact shape the apply step's plan has: template replaced, MIG rebound to it. +function rollPlan({ cellId, cap, protocol }) { + return { + configuration: { + root_module: { + resources: [{ + address: 'google_compute_instance_group_manager.relay_gce_cell', + expressions: { + version: [{ + instance_template: { + references: [ + 'google_compute_instance_template.relay_gce_cell', + 'each.key' + ] + }, + name: { constant_value: 'primary' } + }] + } + }] + } + }, + resource_changes: [ + { + address: `google_compute_instance_template.relay_gce_cell[${JSON.stringify(cellId)}]`, + change: { + actions: ['create', 'delete'], + before: { + metadata_startup_script: startupScript({ + cap, + image: ROLLBACK_IMAGE, + trusted: protocol === 1 + }) + }, + after: { + metadata_startup_script: startupScript({ + cap, + image: TARGET_IMAGE, + trusted: protocol === 1 + }), + self_link: null + }, + after_unknown: { self_link: true } + } + }, + { + address: `google_compute_instance_group_manager.relay_gce_cell[${JSON.stringify(cellId)}]`, + change: { + actions: ['update'], + before: { target_size: 1, version: [{ instance_template: 'old' }] }, + after: { target_size: 1, version: [{ instance_template: null }] }, + after_unknown: { version: [{ instance_template: true }] } + } + } + ] + } +} + +function hostname(cellId) { + return cellId.slice('production-gce-'.length) +} + +// The job resolves cap and region from the cell id before any admin call; run that block alone. +function resolveCellShape(cellId) { + const start = workflow.indexOf(' TARGET_HOSTNAME="${TARGET_CELL_ID#production-gce-}"') + assert.notEqual(start, -1, 'the same-cap cell shape block is missing') + const end = workflow.indexOf('\n esac\n', start) + assert.notEqual(end, -1, 'the same-cap cell shape block has no esac') + const script = workflow.slice(start, end + '\n esac'.length).replace(/^ {10}/gm, '') + return spawnSync('bash', [ + '-euo', + 'pipefail', + '-c', + `${script}\necho "\${EXPECTED_REGION} \${EXPECTED_HARD_CAP}"` + ], { env: { ...process.env, TARGET_CELL_ID: cellId }, encoding: 'utf8' }) +} + +describe('same-cap roll scripts accept every same-cap cell', () => { + it('parses every wave cell through the same-cap canary allowlist', () => { + for (const cellId of SAME_CAP_CELLS) { + for (const mode of ['isolate', 'drain', 'activate']) { + assert.deepEqual(parseProductionCapacityCellArguments([ + '--director-origin', 'https://relay.onorca.dev', + '--cell-origin', `https://${hostname(cellId)}.relay.onorca.dev`, + '--cell-id', cellId, + '--approved-cells', 'same-cap', + '--mode', mode + ]), { + directorOrigin: 'https://relay.onorca.dev', + cellOrigin: `https://${hostname(cellId)}.relay.onorca.dev`, + cellId, + mode + }) + } + } + }) + + it('resolves a cap and region for every wave cell and refuses anything else', () => { + for (const cellId of SAME_CAP_CELLS) { + const resolved = resolveCellShape(cellId) + assert.equal(resolved.status, 0, `${cellId}: ${resolved.stderr}`) + assert.match(resolved.stdout.trim(), /^(us-central1 1000|asia-east2 3000)$/) + } + assert.equal(resolveCellShape('production-gce-c17').status, 1) + assert.equal(resolveCellShape('production-gce-c30').status, 1) + }) + + it('passes the same-cap allowlist on every canary invocation the job runs', () => { + const invocations = workflow.split('prepare-relay-production-capacity-canary.mjs').slice(1) + assert.equal(invocations.length, 4) + for (const invocation of invocations) { + const lines = invocation.split('\n') + const end = lines.findIndex((line) => !line.endsWith('\\')) + const call = lines.slice(0, end + 1).join(' ') + assert.match(call, /--approved-cells same-cap/) + assert.match(call, /--mode (isolate|drain|activate)/) + } + }) + + it('passes this cell\'s rehome protocol on every plan validation the job runs', () => { + const invocations = workflow.split('validate-relay-capacity-plan.mjs').slice(1) + assert.equal(invocations.length, 2) + for (const invocation of invocations) { + const lines = invocation.split('\n') + const end = lines.findIndex((line) => !line.trimEnd().endsWith('\\')) + const call = lines.slice(0, end + 1).join(' ') + assert.match(call, /--mode same-cap-cell/) + assert.match(call, /--regional-rehome-protocol "\$\{DESIRED_REHOME_PROTOCOL\}"/) + } + }) + + it('validates a correct plan for every wave cell at that cell\'s rehome protocol', () => { + for (const cellId of SAME_CAP_CELLS) { + const [region, cap] = resolveCellShape(cellId).stdout.trim().split(' ') + const protocol = REHOME_SOURCE_CELLS.has(cellId) ? 1 : 0 + assert.equal(protocol, region === 'us-central1' ? 1 : 0, cellId) + const config = { + mode: 'same-cap-cell', + cellId, + hardCap: Number(cap), + unobservedBound: 60, + image: TARGET_IMAGE, + rollbackImage: ROLLBACK_IMAGE, + rehomeDirectorServiceAccount: DIRECTOR_IDENTITY, + rehomeAudience: AUDIENCE, + regionalRehomeProtocol: String(protocol) + } + const plan = rollPlan({ cellId, cap, protocol }) + assert.deepEqual( + validateCapacityPlan(plan, config), + { mode: 'same-cap-cell', changes: 2 }, + cellId + ) + // The other protocol must reject the same plan, or the flag decides nothing. + assert.throws( + () => validateCapacityPlan(plan, { + ...config, + regionalRehomeProtocol: String(1 - protocol) + }), + /reviewed image and capacity/, + cellId + ) + } + }) + + it('leaves the US-only capacity job on the default allowlist', () => { + assert.doesNotMatch(capacityWorkflow, /--approved-cells/) + }) +}) diff --git a/cloud/dev/scripts/validate-relay-capacity-plan.mjs b/cloud/dev/scripts/validate-relay-capacity-plan.mjs index 7307295d206..294e85ae31d 100644 --- a/cloud/dev/scripts/validate-relay-capacity-plan.mjs +++ b/cloud/dev/scripts/validate-relay-capacity-plan.mjs @@ -4,7 +4,18 @@ import { pathToFileURL } from 'node:url' const SERVICE_ACCOUNT_EMAIL = /^[a-z][a-z0-9-]{4,28}[a-z0-9]@[a-z0-9-]+\.iam\.gserviceaccount\.com$/ -function parseArguments(argv) { +const REHOME_CONFIG = + /^ printf 'ORCA_RELAY_REHOME_(?:DIRECTOR_SERVICE_ACCOUNT|AUDIENCE)=%s\\n' '[^'\n]+'$/ + +// Only cells listed as regional rehome sources get rehome trust lines in their startup script. +function rehomeProtocol({ regionalRehomeProtocol }) { + if (![0, 1, '0', '1'].includes(regionalRehomeProtocol)) { + throw new Error('same-cap Terraform plan has an invalid regional rehome protocol') + } + return Number(regionalRehomeProtocol) +} + +export function parseCapacityPlanArguments(argv) { const values = {} for (let index = 0; index < argv.length; index += 2) { const key = argv[index] @@ -31,8 +42,12 @@ function parseArguments(argv) { values.mode === 'same-cap-cell' && (!values['rollback-image'] || !values['rehome-director-service-account'] || - !values['rehome-audience']) + !values['rehome-audience'] || + !['0', '1'].includes(values['regional-rehome-protocol'])) ) throw new Error('same-cap validation requires rollback image and rehome trust config') + if (values.mode !== 'same-cap-cell' && values['regional-rehome-protocol'] !== undefined) { + throw new Error('--regional-rehome-protocol applies only to same-cap-cell validation') + } if (values.mode === 'same-cap-image' && !values['rollback-image']) { throw new Error('same-cap image validation requires a rollback image') } @@ -51,7 +66,8 @@ function parseArguments(argv) { capacityServiceAccount: values['capacity-service-account'], rollbackImage: values['rollback-image'], rehomeDirectorServiceAccount: values['rehome-director-service-account'], - rehomeAudience: values['rehome-audience'] + rehomeAudience: values['rehome-audience'], + regionalRehomeProtocol: values['regional-rehome-protocol'] } } @@ -175,15 +191,13 @@ function normalizedStartupScript( /^ printf 'ORCA_RELAY_CELL_CONNECTION_(?:HARD_CAP|UNOBSERVED_BOUND)=%s\\n' '[0-9]+'$/ const capacityIdentity = /^ printf 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT=%s\\n' '[a-z][a-z0-9-]{4,28}[a-z0-9]@[a-z0-9-]+\.iam\.gserviceaccount\.com'$/ - const rehomeConfig = - /^ printf 'ORCA_RELAY_REHOME_(?:DIRECTOR_SERVICE_ACCOUNT|AUDIENCE)=%s\\n' '[^'\n]+'$/ return script .split('\n') .filter( (line) => (preserveCapacity || !capacityAssignment.test(line)) && (!stripCapacityIdentity || !capacityIdentity.test(line)) && - (!stripRehomeConfig || !rehomeConfig.test(line)) + (!stripRehomeConfig || !REHOME_CONFIG.test(line)) ) .join('\n') .replaceAll(image, '') @@ -213,7 +227,8 @@ function requireDesiredStartupScript(script, config) { ` printf 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT=%s\\n' '${config.capacityServiceAccount}'` ]) } - if (config.mode === 'same-cap-cell') { + const rehomeTrusted = config.mode === 'same-cap-cell' && rehomeProtocol(config) === 1 + if (rehomeTrusted) { expected.push( [ /^ printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '[^'\n]+'$/, @@ -225,9 +240,15 @@ function requireDesiredStartupScript(script, config) { ] ) } + // A protocol-0 cell is not a rehome source, so gaining any rehome trust line is real drift. + const unexpectedRehome = + config.mode === 'same-cap-cell' && + !rehomeTrusted && + lines.some((line) => REHOME_CONFIG.test(line)) if ( typeof script !== 'string' || relayImage(script) !== config.image || + unexpectedRehome || expected.some(([pattern, line]) => !hasExactSingleAssignment(lines, pattern, line)) ) { throw new Error('cell plan does not contain the reviewed image and capacity') @@ -450,6 +471,9 @@ export function validateCapacityPlan(plan, config) { ) { throw new Error('capacity Terraform plan has an invalid service account') } + if (config.mode === 'same-cap-cell') { + rehomeProtocol(config) + } if ( config.mode === 'same-cap-cell' && (!SERVICE_ACCOUNT_EMAIL.test(config.rehomeDirectorServiceAccount ?? '') || @@ -504,7 +528,7 @@ export function validateCapacityPlan(plan, config) { } export function main(argv = process.argv.slice(2)) { - const config = parseArguments(argv) + const config = parseCapacityPlanArguments(argv) const plan = JSON.parse(readFileSync(0, 'utf8')) process.stdout.write(`${JSON.stringify({ event: 'relay_capacity_plan_verified', ...validateCapacityPlan(plan, config) })}\n`) } diff --git a/cloud/dev/scripts/validate-relay-capacity-plan.test.mjs b/cloud/dev/scripts/validate-relay-capacity-plan.test.mjs index 207285dc570..fb6ccb57e1c 100644 --- a/cloud/dev/scripts/validate-relay-capacity-plan.test.mjs +++ b/cloud/dev/scripts/validate-relay-capacity-plan.test.mjs @@ -1,6 +1,9 @@ import assert from 'node:assert/strict' import { test } from 'node:test' -import { validateCapacityPlan as validateCapacityPlanRaw } from './validate-relay-capacity-plan.mjs' +import { + parseCapacityPlanArguments, + validateCapacityPlan as validateCapacityPlanRaw +} from './validate-relay-capacity-plan.mjs' const config = { cellId: 'staging-gce-c3', @@ -466,7 +469,8 @@ test('same-cap mode preserves 1000/60 while adding only the reviewed trust confi image, rollbackImage, rehomeDirectorServiceAccount: directorIdentity, - rehomeAudience: audience + rehomeAudience: audience, + regionalRehomeProtocol: '1' } assert.deepEqual( validateCapacityPlan({ resource_changes: [template, manager] }, sameCapConfig), @@ -644,3 +648,134 @@ test('same-cap mode preserves 1000/60 while adding only the reviewed trust confi { mode: 'same-cap-image', changes: 1, changeKind: 'manager-convergence' } ) }) + +test('protocol-0 same-cap cells roll without rehome trust lines', () => { + const rollbackImage = `us-docker.pkg.dev/project/relay/image@sha256:${'d'.repeat(64)}` + const image = `us-docker.pkg.dev/project/relay/image@sha256:${'e'.repeat(64)}` + const directorIdentity = 'relay-director@project.iam.gserviceaccount.com' + const audience = 'https://relay.example.com/v1/admin/host-drain' + const startup = ({ selectedImage, trust = false }) => [ + ` printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '3000'`, + ` printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '60'`, + ` printf 'ORCA_RELAY_CELL_REGION=%s\\n' 'asia-east2'`, + ...(trust ? [ + ` printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '${directorIdentity}'`, + ` printf 'ORCA_RELAY_REHOME_AUDIENCE=%s\\n' '${audience}'` + ] : []), + `printf 'ORCA_RELAY_IMAGE_DIGEST=%s\\n' '${selectedImage.split('@')[1]}'`, + `docker pull '${selectedImage}'`, + 'docker run --detach \\', + ' --name orca-relay \\', + ` '${selectedImage}'` + ].join('\n') + const template = { + address: 'google_compute_instance_template.relay_gce_cell["production-gce-c27"]', + change: { + actions: ['create', 'delete'], + before: { metadata_startup_script: startup({ selectedImage: rollbackImage }) }, + after: { metadata_startup_script: startup({ selectedImage: image }), self_link: null }, + after_unknown: { self_link: true } + } + } + const manager = { + address: 'google_compute_instance_group_manager.relay_gce_cell["production-gce-c27"]', + change: { + actions: ['update'], + before: { target_size: 1, version: [{ instance_template: 'old' }] }, + after: { target_size: 1, version: [{ instance_template: null }] }, + after_unknown: { version: [{ instance_template: true }] } + } + } + const asiaConfig = { + cellId: 'production-gce-c27', + hardCap: 3_000, + unobservedBound: 60, + mode: 'same-cap-cell', + image, + rollbackImage, + rehomeDirectorServiceAccount: directorIdentity, + rehomeAudience: audience, + regionalRehomeProtocol: '0' + } + assert.deepEqual( + validateCapacityPlan({ resource_changes: [template, manager] }, asiaConfig), + { mode: 'same-cap-cell', changes: 2 } + ) + const gainsTrust = structuredClone(template) + gainsTrust.change.after.metadata_startup_script = startup({ + selectedImage: image, + trust: true + }) + assert.throws( + () => validateCapacityPlan({ resource_changes: [gainsTrust, manager] }, asiaConfig), + /reviewed image and capacity/ + ) + // Under protocol 1 that same script is the reviewed roll: trust is added, not drift. + assert.deepEqual( + validateCapacityPlan( + { resource_changes: [gainsTrust, manager] }, + { ...asiaConfig, regionalRehomeProtocol: '1' } + ), + { mode: 'same-cap-cell', changes: 2 } + ) + // A protocol-1 cell whose script has no rehome lines is the pre-existing failure, unchanged. + assert.throws( + () => validateCapacityPlan( + { resource_changes: [template, manager] }, + { ...asiaConfig, regionalRehomeProtocol: '1' } + ), + /reviewed image and capacity/ + ) + for (const protocol of [undefined, '', '2', 'yes']) { + assert.throws( + () => validateCapacityPlan( + { resource_changes: [template, manager] }, + { ...asiaConfig, regionalRehomeProtocol: protocol } + ), + /invalid regional rehome protocol/ + ) + } +}) + +test('the rehome protocol argument is required by same-cap-cell mode alone', () => { + const image = `us-docker.pkg.dev/project/relay/image@sha256:${'e'.repeat(64)}` + const rollbackImage = `us-docker.pkg.dev/project/relay/image@sha256:${'d'.repeat(64)}` + const sameCapArguments = (...extra) => [ + '--mode', 'same-cap-cell', + '--cell-id', 'production-gce-c27', + '--hard-cap', '3000', + '--unobserved-bound', '60', + '--image', image, + '--rollback-image', rollbackImage, + '--rehome-director-service-account', 'relay-director@project.iam.gserviceaccount.com', + '--rehome-audience', 'https://relay.onorca.dev/v1/admin/host-drain', + ...extra + ] + assert.equal( + parseCapacityPlanArguments(sameCapArguments('--regional-rehome-protocol', '0')) + .regionalRehomeProtocol, + '0' + ) + assert.throws( + () => parseCapacityPlanArguments(sameCapArguments()), + /requires rollback image and rehome trust config/ + ) + for (const protocol of ['', '2', 'true']) { + assert.throws( + () => parseCapacityPlanArguments(sameCapArguments('--regional-rehome-protocol', protocol)), + /requires rollback image and rehome trust config/ + ) + } + assert.throws( + () => parseCapacityPlanArguments([ + '--mode', 'bootstrap-cell', + '--cell-id', 'staging-gce-c3', + '--hard-cap', '1000', + '--unobserved-bound', '60', + '--image', image, + '--capacity-service-account', 'orca-cap@onorca-cloud.iam.gserviceaccount.com', + '--regional-rehome-protocol', '0' + ]), + /applies only to same-cap-cell validation/ + ) +}) diff --git a/cloud/dev/scripts/verify-relay-capacity-transition.mjs b/cloud/dev/scripts/verify-relay-capacity-transition.mjs index e5ebe77d45f..b81ea15afb3 100644 --- a/cloud/dev/scripts/verify-relay-capacity-transition.mjs +++ b/cloud/dev/scripts/verify-relay-capacity-transition.mjs @@ -1,4 +1,5 @@ import { pathToFileURL } from 'node:url' +import { fetchAdminOnceMore } from './relay-admin-transient-retry.mjs' const CAPACITY_PROTOCOL = 2 @@ -378,9 +379,12 @@ export async function verifyCapacityTransition(config, overrides = {}) { const token = overrides.token ?? process.env.ORCA_RELAY_ADMIN_ID_TOKEN if (!token || token.length > 8_192) throw new Error('admin identity token is unavailable') const health = await responseJson( - await fetchImpl(`${config.directorOrigin}/health`, { - signal: AbortSignal.timeout(15_000) - }), + await fetchAdminOnceMore( + fetchImpl, + `${config.directorOrigin}/health`, + {}, + { wait, timeoutMs: 15_000 } + ), 'director health' ) if (health.ok !== true || health.connectionCapacityProtocol !== CAPACITY_PROTOCOL) { @@ -394,12 +398,16 @@ export async function verifyCapacityTransition(config, overrides = {}) { lastObservation = { runtimeAvailable: runtime !== null } if ((runtime === null) === (config.runtime === 'unavailable')) { const result = await responseJson( - await fetchImpl(`${config.directorOrigin}/v1/admin/cell-status`, { - method: 'POST', - headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, - body: JSON.stringify({ v: 1, cellId: config.cellId }), - signal: AbortSignal.timeout(30_000) - }), + await fetchAdminOnceMore( + fetchImpl, + `${config.directorOrigin}/v1/admin/cell-status`, + { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, cellId: config.cellId }) + }, + { wait } + ), 'cell status' ) const status = result.status diff --git a/cloud/dev/scripts/verify-relay-capacity-transition.test.mjs b/cloud/dev/scripts/verify-relay-capacity-transition.test.mjs index fb865257c4a..e596ffbced7 100644 --- a/cloud/dev/scripts/verify-relay-capacity-transition.test.mjs +++ b/cloud/dev/scripts/verify-relay-capacity-transition.test.mjs @@ -1094,3 +1094,77 @@ test('does not retry a rejected cell admin token', async () => { ) assert.equal(waits, 0) }) + +test('retries a transient 503 on the director cell-status read', async () => { + const base = harness() + const statusCalls = [] + const result = await verifyCapacityTransition(config, { + token: 'masked-token', + wait: async () => {}, + fetch: async (url, options) => { + const path = new URL(url).pathname + if (path !== '/v1/admin/cell-status') return await base(url, options) + statusCalls.push(path) + if (statusCalls.length === 1) return new Response('warming up', { status: 503 }) + return await base(url, options) + } + }) + assert.equal(statusCalls.length, 2) + assert.equal(result.cellId, config.cellId) +}) + +test('fails when both director cell-status attempts return a transient 503', async () => { + const base = harness() + let statusCalls = 0 + await assert.rejects( + verifyCapacityTransition(config, { + token: 'masked-token', + wait: async () => {}, + fetch: async (url, options) => { + const path = new URL(url).pathname + if (path !== '/v1/admin/cell-status') return await base(url, options) + statusCalls += 1 + return new Response('warming up', { status: 503 }) + } + }), + /cell status returned 503/ + ) + assert.equal(statusCalls, 2) +}) + +test('retries a transient 503 on the director health preflight', async () => { + const base = harness() + let healthCalls = 0 + const result = await verifyCapacityTransition(config, { + token: 'masked-token', + wait: async () => {}, + fetch: async (url, options) => { + const path = new URL(url).pathname + if (path !== '/health') return await base(url, options) + healthCalls += 1 + if (healthCalls === 1) return new Response('warming up', { status: 503 }) + return await base(url, options) + } + }) + assert.equal(healthCalls, 2) + assert.equal(result.cellId, config.cellId) +}) + +test('fails when both director health attempts return a transient 503', async () => { + const base = harness() + let healthCalls = 0 + await assert.rejects( + verifyCapacityTransition(config, { + token: 'masked-token', + wait: async () => {}, + fetch: async (url, options) => { + const path = new URL(url).pathname + if (path !== '/health') return await base(url, options) + healthCalls += 1 + return new Response('warming up', { status: 503 }) + } + }), + /director health returned 503/ + ) + assert.equal(healthCalls, 2) +}) diff --git a/cloud/docs/relay-incident-monitor.md b/cloud/docs/relay-incident-monitor.md index 870c95dd413..8a8dfda1495 100644 --- a/cloud/docs/relay-incident-monitor.md +++ b/cloud/docs/relay-incident-monitor.md @@ -73,6 +73,13 @@ for a committed forward-recovery gate. Durable files default to gap resets the active window at the next fresh sample and preserves the prior window evidence. A threshold freeze never clears automatically. +A signal that reads missing or stale may miss up to two consecutive samples +without restarting the window. The sample still counts and is still checked +against every threshold it can read, and each tolerated gap is recorded in +`continuityEvents` with `tolerated: true`. A third consecutive miss of the same +signal, a failed collector, a runner gap, or any threshold breach restarts or +freezes as before. + A production candidate or multi-target mutation must download the exact dry-run artifact by workflow run ID and attempt. It verifies the artifact hashes and provenance, requires a green completed 15-minute state no older @@ -89,7 +96,8 @@ durably marked consumed before mutation and cannot authorize another run. | Signal | Freeze condition | | --- | ---: | | Active probe age | over 60 seconds | -| Cloud/log data age | over 180 seconds | +| Cloud Monitoring data age | over 330 seconds | +| Relay log and director admin data age | over 180 seconds | | Cell heartbeat age | over 45 seconds | | Endpoint latency | over 2,000 ms | | Cloud SQL CPU | over 80% | @@ -155,6 +163,25 @@ heartbeats, and matching live admission. separate it from today's baseline; the exhausted-retry bar (incident peak 467 vs bar 300), director concurrency, and the pool bars carry that role. Re-tighten after the fleet is on the 500 ms lock wait. +- Raised the Cloud Monitoring freshness bar from 180 s to 330 s and let a + freshness-only failure miss up to two consecutive samples without restarting + the window (2026-09-05). Basis: Google's metric list documents Cloud Run + `request_count`, `container/instance_count`, `container/cpu/utilizations`, + `container/memory/utilizations` and `container/max_request_concurrencies` as + "Sampled every 60 seconds. After sampling, data is not visible for up to 120 + seconds", and Cloud SQL `database/cpu/utilization`, + `database/memory/utilization`, `database/postgresql/num_backends`, + `database/postgresql/backends_in_wait` and `database/postgresql/deadlock_count` + as "up to 165 seconds", so the newest visible point is up to 180 s and 225 s + old respectively. Window-sum signals age further: `observedAt` is the newest + point in the 5-minute query window, so a label series that stops emitting + reads as 300 s old while its summed value is complete. The old bar sat under + all three. Production on 2026-09-04/05 restarted healthy 15-minute windows at + 181 s and 255 s (`auth.errors`, run 33928912676) and at 189 s + (`cloud_sql.lock_waits`, run 33944873727), and the last of those then blew the + 25-minute lineage cap at 1 500 004 ms, so a green fleet produced no verdict. + The director admin bar stays at 180 s and the nonzero lock-wait carry window + stays at 180 s; both publish on our own cadence. - Recalibrated the exhausted-PostgreSQL-retry freeze from 0 to 300 per five minutes (2026-09-04). Basis: #18521 cut the request-path cell-inventory lock wait from the 1 s pool `lock_timeout` to 500 ms, so contended waiters diff --git a/cloud/infra/terraform/relay-observability.tf b/cloud/infra/terraform/relay-observability.tf index 964fc1060e6..6bc938100c6 100644 --- a/cloud/infra/terraform/relay-observability.tf +++ b/cloud/infra/terraform/relay-observability.tf @@ -211,7 +211,8 @@ resource "google_logging_metric" "relay_snapshot" { label_extractors = { role = "EXTRACT(jsonPayload.role)" cell_id = "EXTRACT(jsonPayload.cellId)" - region = "EXTRACT(jsonPayload.region)" + # No region label: adding one replaces all 21 live metrics (label change = delete+create), + # which resets history and blanks the relay alert policies during the swap. } metric_descriptor { @@ -230,12 +231,6 @@ resource "google_logging_metric" "relay_snapshot" { value_type = "STRING" description = "Durable relay cell identifier." } - - labels { - key = "region" - value_type = "STRING" - description = "Coarse Relay region." - } } bucket_options { diff --git a/cloud/package.json b/cloud/package.json index c75d027d3d2..62dbadc7455 100644 --- a/cloud/package.json +++ b/cloud/package.json @@ -21,7 +21,7 @@ "load:relay:recovery-gate": "node dev/scripts/run-relay-recovery-wave-gate.mjs", "ops:relay": "pnpm --filter @orca-cloud/relay-ops dev", "pretest": "node --test dev/scripts/capture-terraform-plan-baseline.test.mjs dev/scripts/operate-relay-asia-admission.test.mjs dev/scripts/prepare-relay-asia-director-cells.test.mjs dev/scripts/prepare-relay-asia-topology-input.test.mjs dev/scripts/production-cloud-sql-rollout-lock.test.mjs dev/scripts/read-relay-serving-regional-placement-version.test.mjs dev/scripts/relay-asia-admission-workflow.test.mjs dev/scripts/relay-asia-rollout-evidence.test.mjs dev/scripts/relay-asia-topology-workflow.test.mjs dev/scripts/relay-cloud-sql-connection-budget.test.mjs dev/scripts/relay-load-reader-evidence.test.mjs dev/scripts/relay-staging-deploy-identity.test.mjs dev/scripts/sanitize-relay-asia-admission-result.test.mjs dev/scripts/terraform-root-partition.test.mjs dev/scripts/validate-relay-asia-topology-plan.test.mjs ../.github/actions/cloud-sql-rollout-lease/action-contract.test.mjs ../.github/actions/cloud-sql-rollout-lease/storage-lease.test.mjs", - "test": "pnpm -r test && node --test dev/scripts/classify-relay-production-capacity-director.test.mjs dev/scripts/classify-relay-staging-bootstrap.test.mjs dev/scripts/deploy-relay-blue-green.test.mjs dev/scripts/deploy-relay-gce-candidate.test.mjs dev/scripts/deploy-relay-gce-multi-target.test.mjs dev/scripts/github-smoke-token.test.mjs dev/scripts/infra.test.mjs dev/scripts/operate-relay-regional-rehome.test.mjs dev/scripts/power-staging-relay.test.mjs dev/scripts/prepare-relay-capacity-canary.test.mjs dev/scripts/prepare-relay-production-capacity-canary.test.mjs dev/scripts/probe-relay-legacy-admission.test.mjs dev/scripts/probe-relay-rehome-trust.test.mjs dev/scripts/production-cell-image-digest-consistency.test.mjs dev/scripts/read-relay-production-capacity-identity.test.mjs dev/scripts/relay-admission-selector.test.mjs dev/scripts/relay-gce-terraform-fence.test.mjs dev/scripts/relay-load-connection-failure.test.mjs dev/scripts/relay-load-control-peer.test.mjs dev/scripts/relay-load-director-capacity-gate.test.mjs dev/scripts/relay-load-model.test.mjs dev/scripts/relay-load-phase-barrier.test.mjs dev/scripts/relay-load-placement-boundary.test.mjs dev/scripts/relay-load-profile.test.mjs dev/scripts/relay-load-rebind-boundary.test.mjs dev/scripts/relay-load-region-behavior.test.mjs dev/scripts/relay-load-request-unit-boundary.test.mjs dev/scripts/relay-load-run-lifecycle.test.mjs dev/scripts/relay-monitor-evidence.test.mjs dev/scripts/relay-production-capacity-wave.test.mjs dev/scripts/relay-production-capacity-workflow.test.mjs dev/scripts/relay-production-identity-boundaries.test.mjs dev/scripts/relay-production-same-cap-wave.test.mjs dev/scripts/relay-public-workflow-contract.test.mjs dev/scripts/relay-recovery-wave-gate.test.mjs dev/scripts/relay-region-observation-evidence.test.mjs dev/scripts/relay-regional-rehome-workflow.test.mjs dev/scripts/relay-rehome-aggregate-evidence.test.mjs dev/scripts/relay-repository.test.mjs dev/scripts/relay-staging-c4-refresh-workflow.test.mjs dev/scripts/relay-staging-capacity-identity.test.mjs dev/scripts/staging-relay-apply-guard.test.mjs dev/scripts/validate-relay-capacity-plan.test.mjs dev/scripts/verify-relay-capacity-transition.test.mjs dev/scripts/verify-relay-legacy-bootstrap.test.mjs dev/scripts/workload-identity-attribute-conditions.test.mjs", + "test": "pnpm -r test && node --test dev/scripts/classify-relay-production-capacity-director.test.mjs dev/scripts/classify-relay-staging-bootstrap.test.mjs dev/scripts/deploy-relay-blue-green.test.mjs dev/scripts/deploy-relay-gce-candidate.test.mjs dev/scripts/deploy-relay-gce-multi-target.test.mjs dev/scripts/github-smoke-token.test.mjs dev/scripts/infra.test.mjs dev/scripts/operate-relay-regional-rehome.test.mjs dev/scripts/power-staging-relay.test.mjs dev/scripts/prepare-relay-capacity-canary.test.mjs dev/scripts/prepare-relay-production-capacity-canary.test.mjs dev/scripts/probe-relay-legacy-admission.test.mjs dev/scripts/probe-relay-rehome-trust.test.mjs dev/scripts/production-cell-image-digest-consistency.test.mjs dev/scripts/read-relay-production-capacity-identity.test.mjs dev/scripts/relay-admin-endpoint-retry-workflow.test.mjs dev/scripts/relay-admin-transient-retry.test.mjs dev/scripts/relay-admission-selector.test.mjs dev/scripts/relay-gce-terraform-fence.test.mjs dev/scripts/relay-load-connection-failure.test.mjs dev/scripts/relay-load-control-peer.test.mjs dev/scripts/relay-load-director-capacity-gate.test.mjs dev/scripts/relay-load-model.test.mjs dev/scripts/relay-load-phase-barrier.test.mjs dev/scripts/relay-load-placement-boundary.test.mjs dev/scripts/relay-load-profile.test.mjs dev/scripts/relay-load-rebind-boundary.test.mjs dev/scripts/relay-load-region-behavior.test.mjs dev/scripts/relay-load-request-unit-boundary.test.mjs dev/scripts/relay-load-run-lifecycle.test.mjs dev/scripts/relay-monitor-evidence.test.mjs dev/scripts/relay-production-capacity-wave.test.mjs dev/scripts/relay-production-capacity-workflow.test.mjs dev/scripts/relay-production-identity-boundaries.test.mjs dev/scripts/relay-production-same-cap-wave.test.mjs dev/scripts/relay-public-workflow-contract.test.mjs dev/scripts/relay-recovery-wave-gate.test.mjs dev/scripts/relay-region-observation-evidence.test.mjs dev/scripts/relay-regional-rehome-workflow.test.mjs dev/scripts/relay-rehome-aggregate-evidence.test.mjs dev/scripts/relay-repository.test.mjs dev/scripts/relay-same-cap-script-census.test.mjs dev/scripts/relay-staging-c4-refresh-workflow.test.mjs dev/scripts/relay-staging-capacity-identity.test.mjs dev/scripts/staging-relay-apply-guard.test.mjs dev/scripts/validate-relay-capacity-plan.test.mjs dev/scripts/verify-relay-capacity-transition.test.mjs dev/scripts/verify-relay-legacy-bootstrap.test.mjs dev/scripts/workload-identity-attribute-conditions.test.mjs", "typecheck": "pnpm -r typecheck" }, "devDependencies": { diff --git a/config/docker/cli-launch-contract/Dockerfile b/config/docker/cli-launch-contract/Dockerfile index f6a618a8ece..c90cbcd979c 100644 --- a/config/docker/cli-launch-contract/Dockerfile +++ b/config/docker/cli-launch-contract/Dockerfile @@ -6,8 +6,13 @@ ARG LIBASOUND_PACKAGE=libasound2t64 ENV DEBIAN_FRONTEND=noninteractive # Install Electron's link-time libraries without adding a display server or FUSE. -RUN apt-get update \ - && apt-get install -y --no-install-recommends \ +# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts. +RUN for attempt in 1 2 3 4 5; do \ + apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \ + if [ "$attempt" = 5 ]; then exit 100; fi; \ + rm -rf /var/lib/apt/lists/*; sleep 20; \ + done \ + && apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \ bash \ ca-certificates \ coreutils \ diff --git a/config/docker/headless-pairing/Dockerfile b/config/docker/headless-pairing/Dockerfile index 03664f68b0d..e4b4cafeefc 100644 --- a/config/docker/headless-pairing/Dockerfile +++ b/config/docker/headless-pairing/Dockerfile @@ -5,8 +5,13 @@ ARG LIBASOUND_PACKAGE=libasound2t64 ENV DEBIAN_FRONTEND=noninteractive -RUN apt-get update \ - && apt-get install -y --no-install-recommends \ +# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts. +RUN for attempt in 1 2 3 4 5; do \ + apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \ + if [ "$attempt" = 5 ]; then exit 100; fi; \ + rm -rf /var/lib/apt/lists/*; sleep 20; \ + done \ + && apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \ bash \ ca-certificates \ dbus-x11 \ diff --git a/config/docker/headless-serve-shutdown/Dockerfile b/config/docker/headless-serve-shutdown/Dockerfile index 13b1ed2b69f..8ee7b942499 100644 --- a/config/docker/headless-serve-shutdown/Dockerfile +++ b/config/docker/headless-serve-shutdown/Dockerfile @@ -2,8 +2,13 @@ FROM ubuntu@sha256:678c6550cc43645e08669028bc177f50be4e7c5b8cca677067b1914d4afc7 ENV DEBIAN_FRONTEND=noninteractive -RUN apt-get update \ - && apt-get install -y --no-install-recommends \ +# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts. +RUN for attempt in 1 2 3 4 5; do \ + apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \ + if [ "$attempt" = 5 ]; then exit 100; fi; \ + rm -rf /var/lib/apt/lists/*; sleep 20; \ + done \ + && apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \ bash \ ca-certificates \ dbus-x11 \ diff --git a/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs b/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs index dd26784ee46..1e908754dc6 100644 --- a/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs +++ b/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs @@ -12,14 +12,15 @@ const { join, resolve } = require('node:path') * `_inSocket`, and it wraps a real Windows named-pipe handle from `fs.openSync(term.conin, 'w')`. * Every terminal leaks one File handle for the life of the host process. * - * The obvious fix -- and the one the desktop patch ships -- releases it at the TOP of the branch, - * before `_getConsoleProcessList()` forks and before the native kill. That is measurably worse than - * leaving the leak alone: teardown aborts partway, the forked console-list agent is never reaped, - * and both pipe handles stay alive instead of one. This asset releases it at the END of the branch - * instead, after the fork and the kill have already happened. + * The obvious fix -- and the placement `config/patches/node-pty@1.1.0.patch` uses -- releases it at + * the TOP of the branch, before `_getConsoleProcessList()` forks and before the native kill. That is + * measurably worse than leaving the leak alone: teardown aborts partway, the forked console-list + * agent is never reaped, and both pipe handles stay alive instead of one. This asset releases it at + * the END of the branch instead, after the fork and the kill have already happened. * * Measured on a Windows SSH host, 20 spawn/kill cycles, handles bucketed by NT object type - * (identical numbers standalone and through a real relay): + * (identical numbers standalone and through a real relay). Every row is the NON-DLL branch, which + * is the branch a relay runs -- see the divergence note below for why that matters: * * published node-pty File +1/terminal, Process flat * desktop patch placement File +2/terminal, Process +1/terminal <-- 3x WORSE @@ -34,18 +35,64 @@ const { join, resolve } = require('node:path') * Why this ships as a relay asset rather than only in config/patches/node-pty@1.1.0.patch: pnpm * patches do not cross the SSH boundary -- a relay host runs the tree `npm install` put there. * - * DELIBERATE DIVERGENCE FROM THE DESKTOP: the desktop patch has the early placement and therefore - * the +2 File / +1 Process regression, measured against its exact installed tree. Correcting it - * there is a separate change with its own verification, so the two trees differ on this one hunk on - * purpose, and the test pins that so a future "sync the patches" does not copy the bug back. + * DELIBERATE DIVERGENCE FROM THE DESKTOP, AND WHY IT IS NOT A DESKTOP-TERMINAL BUG: the two hosts + * do not run the same branch of `kill()`. node-pty defaults `_useConptyDll` to false + * (`windowsPtyAgent.js`). Every desktop site that opens a terminal pane sets it true -- + * `local-pty-utils.ts` (two) and `native-pty-spawn.ts` -- as does the `windows-conpty-warmup.ts` + * warm-up, so all of those take the `else` branch, where UPSTREAM ALREADY destroys the input + * socket. The relay passes no such option (`src/relay/pty-handler.ts`), so it takes the + * `!useConptyDll` branch -- the one this asset and the desktop patch both edit. * - * NOT ADDRESSED, AND A SEPARATE DEFECT THAT IS STILL OPEN: a terminal that exits on its own is - * still torn down through `kill()` -- both hosts call `destroy()` on natural exit and - * `WindowsTerminal.destroy()` is `kill()` -- but the shell is already gone by then, and the - * ordering this patch relies on does not hold. Measured over 20 self-exit cycles with that - * `destroy()` issued: published +3 File/+1 Process per terminal, desktop-patched +2/+1, this tree - * +2/+1. So this patch does not close it and the desktop patch does not either. It is reachable - * for every Windows user, local and relay, on every terminal closed by typing `exit`. + * THE DESKTOP IS NOT ENTIRELY OFF THAT BRANCH. Two desktop sites omit the option and so run it + * too: the hidden rate-limit probes in `src/main/rate-limits/claude-pty.ts` and + * `codex-pty-rate-limit-probe.ts`. Both recur -- their fetchers poll -- and both tear down through + * `kill()`, so this hunk is live on the desktop, just never for a pane a user can see. Do not + * restate this as "the desktop never executes that branch": that sentence stood here for two + * revisions and is false. + * + * What the numbers above therefore do NOT cover: they were measured on relay-style spawn/kill + * cycles. Whether the early placement costs the same +2 File / +1 Process across a probe's + * lifecycle is UNMEASURED -- plausible, not established, and worth measuring before anyone quotes + * a desktop figure. What IS settled is the claim this comment replaced: that the desktop patch made + * every Windows user worse off ON EVERY TERMINAL. Terminals take the DLL branch, and the harness + * that produced that claim defaulted into the branch it was not trying to measure. + * + * The divergence is therefore about which branch each host runs for the workload that matters, not + * about a regression in the terminals users open. The test still pins it, because a future "sync + * the patches" would put the early placement onto the relay's branch, where it does cost +2 File + * and +1 Process per terminal. + * + * If you extend this enumeration, grep for `node-pty` rather than for a static import: those two + * probes were missed three times because they use `await import('node-pty')`. + * + * THE SELF-EXIT LEAK: FIXED FOR THE DESKTOP BY #18635, STILL LIVE ON A RELAY. A terminal that exits + * on its own is also torn down through `kill()` -- both hosts call `destroy()` on natural exit and + * `WindowsTerminal.destroy()` is `kill()` -- but the shell is already gone by then, so the ordering + * this asset relies on does not hold. Measured over 20 self-exit cycles on the NON-DLL branch: + * published +3 File/+1 Process per terminal, desktop patch placement +2/+1, this tree +2/+1. This + * asset does not close it. + * + * #18635 does, in `config/patches/node-pty@1.1.0.patch`: the baton outlives the shell so `PtyKill` + * still reaches `ClosePseudoConsole`, plus an unconditional conout dispose on the DLL branch. That + * fix does not reach a Windows relay, and no hunk in THIS file can carry it, because it is mostly + * NATIVE (`src/win/conpty.cc`) and this asset only rewrites `lib/*.js`. Three delivery paths exist + * and none currently covers Windows: + * + * - the pnpm patch does not cross the SSH boundary -- the remote `npm install` yields upstream's + * unpatched node-pty; + * - the orcad prebuild matrix has no win32 entry (`MATRIX_SLOTS`, + * `config/scripts/build-orcad-prebuilds.mjs`), so no Windows binary is ever compiled from + * patched source to ship; + * - a relay asset CAN patch native source and rebuild on the host -- that is exactly what + * `node-pty-1.1.0-master-cloexec-patch.cjs` does -- but it returns + * `skipped:unsupported-platform` for anything but linux/darwin. Extending it to win32 means + * requiring an MSVC toolchain on the relay host, a far heavier precondition than on Linux, + * where node-gyp already runs at install time. + * + * So a Windows SSH relay still leaks a pseudoconsole per self-exiting terminal, and closing it is a + * DELIVERY problem, not another hunk here. Do not read #18635's flat self-exit relay numbers as + * covering deployed relays: they were measured against a locally rebuilt binary, so they describe + * the relay CODE PATH on a patched tree, not the tree a relay host actually installs. */ const EXPECTED_NODE_PTY_VERSION = '1.1.0' diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc index bb169fb013d..1fed7b8315d 100644 --- a/config/reliability-gates.jsonc +++ b/config/reliability-gates.jsonc @@ -2951,7 +2951,7 @@ "https://github.com/stablyai/orca/pull/13876" ], "invariant": "Opening one HTML preview from a paired client renders the workspace document in exactly one client-local browser tab, located by that document and served over the orca-preview scheme. The client gains exactly that one browser workspace and it is the document one — blank where a URL page carries a URL, named by the document, with the chip naming the file — while the host gains no browser page at all, neither in its own page registry nor in the tab snapshot its clients publish into. The preview occupies its own split without taking focus from the source editor; an explicit click activates it, and closing it removes only the preview. Following a document from a file link is the other half of that switch and does move the reader to it, tab group included, whether the preview is new or already open, because opening a file is a request to look at it. A document tab quit with the client comes back as the same row on a grant the relaunched client mints afresh. A preview is named by the browser page it is open in, not by a namespace of its own, and the page registry has two halves: a workspace-document guest is registered in its own map and is absent from the browsing one entirely. That absence is the fence. Page, session and profile management, agent tab enumeration and command targeting, download routing and certificate attribution all read the browsing map directly, in more places than a per-channel guard could be remembered in, so none of them can name a document page and none of them carries a guard. Browser tools the reader drives (element grab, hover describe, selection capture, the annotation viewport bridge) are the one operation that legitimately spans the halves, and they go through the single authority that reads both, keyed by the page and its hosting renderer. The halves are disjoint in both directions: browsing registration refuses a page the document half already holds, and minting a grant refuses a page the browsing half already holds, so one id can never name a surface in both. The headless backend acts on that refusal by destroying the window it had already opened rather than leaving a policy-less page behind an id nothing can drive, keeping nothing under that id for its own shutdown to hand back. Registration refuses on the same terms when the guest it was asked about is already gone. The exit door is guarded in both its halves: a preview withdraws by revoking its grant and never through the unregister channel, so a page the document half holds arriving there is refused before either the registration teardown or the grab-state disposal beside it, which would otherwise drop the intent an in-flight preview grab compares by identity and leave that grab answering ok without ever arming its guest. A bridge request whose guest does not resolve is refused without tearing down the page it named, so a misaddressed request cannot cancel a healthy page's in-flight downloads and grabs. The annotation viewport bridge resolves its guest when its serialized op actually runs rather than when the request arrived, so a cross-process navigation while it waited cannot leave the bridge installed in a retired guest while the reader looks at a new one. State main keys by a preview's page is disposed when that page's grant is revoked, which is the only signal a preview's surface is gone. A tool asking for a page whose guest has not attached yet waits for that registration and arms when it arrives, rather than answering not-ready at the reader; that wait resolves only the request already naming this page, never the worktree-wide or any-tab waits the CLI and agents use to ask for a browser tab to drive. Handing the previewed document to the reader's own machine routes on the owners its grant was minted against — the file's own connection owner and the worktree's own runtime owner, neither read from the tab's stored fields. Only a document proven to live on this machine reaches the client OS; one with a resolved remote owner is downloaded first; and one whose owner cannot be resolved at all, workspace root included, is refused with a message naming that, because the download route would otherwise read the same absolute path on the client and hand back a same-named local file under the remote document's name. A runtime-owned path that falls outside its worktree root is refused by that route itself and surfaces as a failure toast rather than a download. Nothing the document does writes a file to this machine either: the preview partition denies downloads outright instead of routing them through the browser download flow, which has no page to attribute a preview's bytes to and would otherwise reserve a name in this desktop's Downloads folder and write them there unprompted. That refusal is visible to the reader and invisible to the document: the preview's shell carries a fixed sentence saying downloads are off, published at most once per preview per interval so a document asking in a loop cannot fill Orca's chrome, while the page itself gets back exactly what it got before, which is nothing. The sentence names no file, because the document chooses the name it offers; and a refusal never takes the document away the way an entry document's own failure does, whatever it names. A preview is a browser tab, not an editor tab in a preview mode: it is named the way a browser tab is named — by the document it shows when that document declares a title, and by the file it shows when it does not — while the chip goes on naming the file and the host whatever the document calls itself. A title is refused on the same terms the url is: a document that declares none has Chromium report the grant URL as its title, and that title is stored, mirrored onto the tab and written to disk, so anything carrying the scheme falls back to the file instead. It is created by the preview action as a page located by its document, it carries the workspace-relative path copy the editor's path header owned, and closing it revokes the grant that made the document readable while a URL tab closing beside it revokes nothing. Chrome persisted by builds that made previews editor tabs is dropped on restore rather than coming back naming a surface no restore can produce, and the ordinary editor tab for the same document is left alone. A document tab is held back at the mobile publish boundary — no client holds its grant, and the wire has no tab kind for it — while an ordinary browser tab beside it still publishes. It is held back from the group projection that publishes tab order, recency and group activity as well as from the tab list itself, so no published group names a tab the phone is never sent. A browser page can be located by a workspace document instead of a URL, and the document is the whole of its stored identity. The grant and the orca-preview URL that document is served over are minted when the page mounts and replaced by a hard reload, so neither is ever written to the page's url, mirrored onto its tab, persisted or published: such a page's url is the blank URL from creation through restore, including when a session written elsewhere carries a grant URL in, and what the session carries is the worktree and path a restored page mints afresh against today's owners. Every door onto a page's url holds that line — creation, the title update, and the navigation commit alike — so a report about a document page cannot give it a URL it never had, and the title fallback and the loading affordance follow the url each door actually wrote. The mirror carries the document too, so a tab entry cannot go on naming a document its active page has left. Every guest in the app is policy-attached through one door: a workspace document takes a restricted profile there rather than a separate installer beside it, so the attachment bookkeeping that door owns — what registration refuses, and what teardown frees — covers a preview on the same terms as a browsing page, and a preview takes none of the browsing machinery that door installs. That authority answers from the moment the embedder hands the guest over rather than only after a later navigation: the guest binds to the grant it is already showing, so the tools reach the document the reader opened and not just one they navigated to. A read the host reports as truncated or over-cap is refused rather than served partially, and a document outside the paired worktree is refused with a message naming that boundary instead of a bare read failure. The rendered document reaches nothing off-machine on its own: every served response carries a self-only content security policy, the preview session cancels any request that is not in-document, subframes cannot navigate outside the grant, a guest no document has yet bound to a grant may not navigate at all, the guest gathers no ICE candidates, and an SSH path that canonicalizes outside the grant root is refused before it is read. The one route out is a link the reader presses: a trusted click on an anchor, reported by the preview's own preload from a guest still bound to a live grant, leaves as an Orca browser tab rather than a native window or a dead click — and only after the reader confirms the exact destination URL, so a document cannot spend a single stray press exfiltrating what it can read into a link it authored. The preview hands its guest that focus itself whenever it is the surface the reader is in — a browsing page gets it from the chrome around it, and a preview has no chrome to get it from — and it does so only then, so a preview mounted behind a terminal or an editor never takes the keyboard from what the reader is actually in. It offers again when the window itself takes focus back and nothing in the embedder has claimed that focus, because another app coming to the front lands focus on the embedder rather than the guest and the route out would otherwise stay shut until something remounted the pane — while the same window focus also arrives when the reader presses a tab, that being the guest's own blur returning, and taking focus back from there would fight the reader for their own click. Nothing else does. A navigation or popup the document starts by itself is swallowed whatever else is happening, including immediately after a genuine press elsewhere in the document, so a page that can read its grant cannot hand it to a browser tab; a middle click opens nothing; and a fragment link is answered inside the document. A preview attach carries the preview preload and no renderer-supplied one, and no other attach path can acquire it. A subresource the workspace will not send degrades the document to a notice naming that file, never to a failure panel over a page that rendered. A grant outlives neither the tab that owns it nor the renderer document that minted it, and only the trusted renderer can mint or revoke one. For the browser creations this gate still owns, owner-pinned creation returns the canonical host page identity before navigation readiness; delayed navigation cannot turn a created page into an unidentifiable failure or a duplicate retry. Capability rejection before host mutation must preserve the original error, issue no RPC, surface a failure toast, and remove only a caller-declared newly-created empty split. Post-create reconciliation failure requires exact rollback; ambiguous rollback rejects without local fallback.", - "oracle": "In paired Electron, write an HTML fixture that declares its own title on the host, invoke the Explorer preview action on the client, and require the document text to be readable out of the orca-preview guest before judging any absence. With that presence established, require the client to hold exactly one browser workspace more than its baseline and that workspace to be the document one: page and tab url blank, the document path mirrored onto both, the tab named by the document's title, the chip naming the file, no editor row of the retired preview species anywhere, and the guest URL carrying the orca-preview scheme. Ask the host through its own page registry as well as through the tab snapshot, and in the same run open an ordinary URL browser tab from the same client and require that one to arrive in both — the presence precondition without which “the host gained nothing” is satisfied just as well by an oracle that cannot see browser pages at all. Require the preview to sit in a group other than the source editor's while the active group and tab remain the source editor's. Then click away to the terminal, click the preview tab, require it to reactivate and still render, close it with its own X, and require the document tab to be gone while the host still holds only the URL tab and the source group, source editor and terminal survive. Quit the client with a document tab open and relaunch it on the same profile: require the same workspace row to come back, blank and named by the document, rendering the document again over a grant URL that differs from the one that was quit, with no preview-scheme or document-named page anywhere in what the host holds. Prove the halves are live by flipping one product property at a time and requiring the run to fail: publish document workspaces to the host like ordinary ones, and stop mirroring the document onto the workspace row. Have the fixture document attempt its own egress on every load — an unattended window.open and location.href to an off-machine URL, plus an inline ICE gathering probe — and require the same baseline counts and a candidate count of zero, so the document's own attempts are measured rather than assumed. Then, as a separate phase after the close oracle has already run, bring the client window to the front, press the document's heading with a real mouse event, and require the document to report that the same press drove it to attempt a second window.open and location.href while both browser counts stay at that phase's baseline and nothing routes — the case a recent-input gate cannot distinguish from the press's own effect. Only then press the target=_blank link with a real mouse event and require both a recorded routing call that returned success and a browser count above that baseline, with the preview tab still open. Drive the preload's click policy as a unit oracle over a real document: a dispatched click, a trusted press on an external anchor, an anchor reached through what it wraps, an SVG animated href, a sibling preview link, fragment and percent-encoded fragment targets, a bare hash, and a middle click. Create a browser page located by a workspace document, handing creation a live grant URL, and require its stored url, its mirrored tab url and the written session payload all to be blank with no orca-preview string anywhere in what was written, while an ordinary page created the same way keeps the URL it was given and asks for the address bar the document page never does. Parse the written page and tab through the session schema and require the document to survive both halves. Hydrate them back and require the document page to return blank and still named — including when its page row was salvaged away and only the tab's own copy remains, and when a foreign session carried a grant URL into both rows. Drive the mirror across a page switch out of the document and back, and across a repair in which the document is the only mirrored field that differs. Name a document page from its document and require the tab to take that name, name it with an empty title and require the file, name it with a live grant URL and require the file again with no preview scheme anywhere in the written session, and require an ordinary blank browser tab beside it to still be called New Tab. Dispatch a title update out of a rendered preview's own guest and require it to reach the page state while the identity chip still reads the document's workspace-relative path. Attach a browsing guest and a workspace-document guest through the same method in one run and require the browsing one to take clicked-link routing, popup handling and anti-detection while the document guest takes none of them, stays inside the grant it is showing, denies every window it asks for, and is dropped from the page-keyed document registry by the same teardown that frees its id for a later attach. Register a browsing guest and attach a workspace-document guest in one run against the real manager, require the one door to answer each page with the guest of its own half, and require the document page to be absent from the browsing map and from its enumeration. Drive both browsing registration entry points with a page the document half already holds and require them to register nothing, and drive the mint channel with a page the browsing half already holds and require it to refuse; and drive the offscreen one with a guest that is missing and with one already destroyed, requiring the same refusal. Arm a grab on a live preview target, drive the unregister channel at that same target in the window before the queued operation runs, and require the grab to reach the guest anyway — then drive the same sequence for an ordinary browser page and require its grab state to be disposed after all. Hold one viewport-bridge op open, queue a second behind it, swap the page's guest while that second op waits, and require the injection to land in the guest the page has then. Revoke a grant after a tool has run against its page and require that page's grab state to be cancelled and disposed. Ask a tool for a document page whose guest has not attached, require the request to park in the registration wait, attach the guest, and require the same request to arm on it; require a page nothing ever renders to answer not-ready once that wait elapses. With a document open, ask a tool for a browsing page id and require it to be answered by the browsing half or not at all, with the same channel reaching the document guest under the page it really renders. Drive open-externally for a document whose per-file owner is remote while the workspace-scoped owner is unresolved, for a runtime-owned worktree whose preview tab carries no runtime id of its own, and for a worktree that resolves no runtime owner while the tab still carries one, requiring the download route in each; and for an owner that cannot be resolved at all, and for an unknown workspace root while nothing names another host, requiring a refusal that neither opens nor downloads. Drive the headless backend with a page the document half already holds and require it to reject, destroy the window, and unregister nothing — then shut the backend down and require it still to have unregistered nothing. Navigate a bound preview guest at a second grant through both latch events and require it to stay on the grant it bound to. Mount a preview while a renderer drag is already in flight and require its guest to be click-through at the moment it is appended, not a turn later. Render the editor panel shell in each remaining tab mode and require the path header exactly where the surface does not already name itself. Drive the preview action and require a browser tab located by the document rather than an editor tab, require a second open of the same document to activate the tab it is already in, and require closing that tab to revoke its grant while a URL tab closed beside it revokes none. Hydrate a session carrying preview chrome from a build that made previews editor tabs and require it dropped while the ordinary editor tab for the same document survives. Publish a worktree holding a document tab and a URL tab and require only the URL tab to reach the mobile snapshot. Install the shared partition policies for a preview partition and for an ordinary browsing partition in the same run, fire each one's own will-download listener, and require the preview's to cancel while the browsing one still reaches the download router — then require the preview protocol installer to be what asks for that deny. In the same run, require the cancelled download to raise a reader-facing notice and the routed one to raise none. Drive that notice directly for a guest bound to a live grant, for repeated attempts inside and outside its interval, for two previews at once, and for a contents no preview is bound to; require the guest registry to name the bound grant for a live preview guest and nothing for a contents that is not one, has committed no document, or is gone. Drive the shell with a refusal and require one fixed sentence, still one row after three more refusals, standing beside an asset failure rather than being counted with it, gone behind the failure panel, and ignored when it names another preview's grant. Drive the main-side report gate directly for a sender that is no preview guest, a guest with no bound or a revoked grant, a genuine press Electron's webview focus flag misreports as unfocused, and non-web URLs; drive the reader-facing confirmation for accept, cancel, and a confirmed tab the browser refuses; and drive will-attach-webview in both preload directions. Run the per-owner reader, grant-containment, scheme-admission, guest-policy, and plan-routing contracts as unit oracles, including a host-reported truncation, an over-cap binary, and an out-of-worktree paired path. Drive the reader-facing component with the payloads the reader can actually produce — the entry document fails only as truncated or unreadable, a subresource additionally as a refused format — and require the asset case to leave the guest mounted. Drive the closed-tab cleanup hook, the window installer, and the grant IPC handlers directly, requiring the grant to be released when the preview tab closes, cleared at window creation and on a cross-document main-frame navigation, and refused to any sender that is not the trusted renderer. For the browser creations this gate still owns, run the unchanged contract oracle for direct create and side-preview callers with absent status, unknown capabilities, and a mixed-version host, requiring the original unsupported error or visible toast, zero RPCs, and no retained new split; hold a real navigation response beyond the 15-second client deadline after host creation and require the first RPC to return the exact host inventory page ID, one host page, and no retry; repeat reconciliation faults against headless serve and retain the separate exact-page reconciliation rollback oracle.", + "oracle": "In paired Electron, write an HTML fixture that declares its own title on the host, invoke the Explorer preview action on the client, and require the document text to be readable out of the orca-preview guest before judging any absence. With that presence established, require the client to hold exactly one browser workspace more than its baseline and that workspace to be the document one: page and tab url blank, the document path mirrored onto both, the tab named by the document's title, the chip naming the file, no editor row of the retired preview species anywhere, and the guest URL carrying the orca-preview scheme. Ask the host through its own page registry as well as through the tab snapshot, and in the same run open an ordinary URL browser tab from the same client and require that one to arrive in both — the presence precondition without which “the host gained nothing” is satisfied just as well by an oracle that cannot see browser pages at all. Require the preview to sit in a group other than the source editor's while the active group and tab remain the source editor's. Then click away to the terminal, click the preview tab, require it to reactivate and still render, close it with its own X, and require the document tab to be gone while the host still holds only the URL tab and the source group, source editor and terminal survive. Quit the client with a document tab open and relaunch it on the same profile: require the same workspace row to come back, blank and named by the document, rendering the document again over a grant URL that differs from the one that was quit, with no preview-scheme or document-named page anywhere in what the host holds. Prove the halves are live by flipping one product property at a time and requiring the run to fail: publish document workspaces to the host like ordinary ones, and stop mirroring the document onto the workspace row. Have the fixture document attempt its own egress on every load — an unattended window.open and location.href to an off-machine URL, plus an inline ICE gathering probe — and require the same baseline counts and a candidate count of zero, so the document's own attempts are measured rather than assumed. Then, as a separate phase after the close oracle has already run, bring the client window to the front, press the document's heading with a real mouse event, and require the document to report that the same press drove it to attempt a second window.open and location.href while both browser counts stay at that phase's baseline and nothing routes — the case a recent-input gate cannot distinguish from the press's own effect. Only then press the target=_blank link with a real mouse event and require both a recorded routing call that returned success and a browser count above that baseline, with the preview tab still open. Drive the preload's click policy as a unit oracle over a real document: a dispatched click, a trusted press on an external anchor, an anchor reached through what it wraps, an SVG animated href, a sibling preview link, fragment and percent-encoded fragment targets, a bare hash, and a middle click. Create a browser page located by a workspace document, handing creation a live grant URL, and require its stored url, its mirrored tab url and the written session payload all to be blank with no orca-preview string anywhere in what was written, while an ordinary page created the same way keeps the URL it was given and asks for the address bar the document page never does. Parse the written page and tab through the session schema and require the document to survive both halves. Hydrate them back and require the document page to return blank and still named — including when its page row was salvaged away and only the tab's own copy remains, and when a foreign session carried a grant URL into both rows. Drive the mirror across a page switch out of the document and back, and across a repair in which the document is the only mirrored field that differs. Name a document page from its document and require the tab to take that name, name it with an empty title and require the file, name it with a live grant URL and require the file again with no preview scheme anywhere in the written session, and require an ordinary blank browser tab beside it to still be called New Tab. Dispatch a title update out of a rendered preview's own guest and require it to reach the page state while the identity chip still reads the document's workspace-relative path. Attach a browsing guest and a workspace-document guest through the same method in one run and require the browsing one to take clicked-link routing, popup handling and auth-identity detach tracking while the document guest takes none of them, stays inside the grant it is showing, denies every window it asks for, and is dropped from the page-keyed document registry by the same teardown that frees its id for a later attach. Register a browsing guest and attach a workspace-document guest in one run against the real manager, require the one door to answer each page with the guest of its own half, and require the document page to be absent from the browsing map and from its enumeration. Drive both browsing registration entry points with a page the document half already holds and require them to register nothing, and drive the mint channel with a page the browsing half already holds and require it to refuse; and drive the offscreen one with a guest that is missing and with one already destroyed, requiring the same refusal. Arm a grab on a live preview target, drive the unregister channel at that same target in the window before the queued operation runs, and require the grab to reach the guest anyway — then drive the same sequence for an ordinary browser page and require its grab state to be disposed after all. Hold one viewport-bridge op open, queue a second behind it, swap the page's guest while that second op waits, and require the injection to land in the guest the page has then. Revoke a grant after a tool has run against its page and require that page's grab state to be cancelled and disposed. Ask a tool for a document page whose guest has not attached, require the request to park in the registration wait, attach the guest, and require the same request to arm on it; require a page nothing ever renders to answer not-ready once that wait elapses. With a document open, ask a tool for a browsing page id and require it to be answered by the browsing half or not at all, with the same channel reaching the document guest under the page it really renders. Drive open-externally for a document whose per-file owner is remote while the workspace-scoped owner is unresolved, for a runtime-owned worktree whose preview tab carries no runtime id of its own, and for a worktree that resolves no runtime owner while the tab still carries one, requiring the download route in each; and for an owner that cannot be resolved at all, and for an unknown workspace root while nothing names another host, requiring a refusal that neither opens nor downloads. Drive the headless backend with a page the document half already holds and require it to reject, destroy the window, and unregister nothing — then shut the backend down and require it still to have unregistered nothing. Navigate a bound preview guest at a second grant through both latch events and require it to stay on the grant it bound to. Mount a preview while a renderer drag is already in flight and require its guest to be click-through at the moment it is appended, not a turn later. Render the editor panel shell in each remaining tab mode and require the path header exactly where the surface does not already name itself. Drive the preview action and require a browser tab located by the document rather than an editor tab, require a second open of the same document to activate the tab it is already in, and require closing that tab to revoke its grant while a URL tab closed beside it revokes none. Hydrate a session carrying preview chrome from a build that made previews editor tabs and require it dropped while the ordinary editor tab for the same document survives. Publish a worktree holding a document tab and a URL tab and require only the URL tab to reach the mobile snapshot. Install the shared partition policies for a preview partition and for an ordinary browsing partition in the same run, fire each one's own will-download listener, and require the preview's to cancel while the browsing one still reaches the download router — then require the preview protocol installer to be what asks for that deny. In the same run, require the cancelled download to raise a reader-facing notice and the routed one to raise none. Drive that notice directly for a guest bound to a live grant, for repeated attempts inside and outside its interval, for two previews at once, and for a contents no preview is bound to; require the guest registry to name the bound grant for a live preview guest and nothing for a contents that is not one, has committed no document, or is gone. Drive the shell with a refusal and require one fixed sentence, still one row after three more refusals, standing beside an asset failure rather than being counted with it, gone behind the failure panel, and ignored when it names another preview's grant. Drive the main-side report gate directly for a sender that is no preview guest, a guest with no bound or a revoked grant, a genuine press Electron's webview focus flag misreports as unfocused, and non-web URLs; drive the reader-facing confirmation for accept, cancel, and a confirmed tab the browser refuses; and drive will-attach-webview in both preload directions. Run the per-owner reader, grant-containment, scheme-admission, guest-policy, and plan-routing contracts as unit oracles, including a host-reported truncation, an over-cap binary, and an out-of-worktree paired path. Drive the reader-facing component with the payloads the reader can actually produce — the entry document fails only as truncated or unreadable, a subresource additionally as a refused format — and require the asset case to leave the guest mounted. Drive the closed-tab cleanup hook, the window installer, and the grant IPC handlers directly, requiring the grant to be released when the preview tab closes, cleared at window creation and on a cross-document main-frame navigation, and refused to any sender that is not the trusted renderer. For the browser creations this gate still owns, run the unchanged contract oracle for direct create and side-preview callers with absent status, unknown capabilities, and a mixed-version host, requiring the original unsupported error or visible toast, zero RPCs, and no retained new split; hold a real navigation response beyond the 15-second client deadline after host creation and require the first RPC to return the exact host inventory page ID, one host page, and no retry; repeat reconciliation faults against headless serve and retain the separate exact-page reconciliation rollback oracle.", "commands": [ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-browser.test.ts src/main/runtime/rpc/methods/browser.test.ts src/renderer/src/lib/file-preview.test.ts src/renderer/src/runtime/web-session-browser-placement.test.ts src/renderer/src/runtime/web-runtime-session.test.ts src/renderer/src/runtime/web-session-tabs-sync.test.ts src/renderer/src/runtime/remote-server-parity.test.ts", "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/runtime/web-runtime-browser-materialization.test.ts", diff --git a/config/scripts/agent-inspection-cadence-batching-benchmark.mjs b/config/scripts/agent-inspection-cadence-batching-benchmark.mjs new file mode 100644 index 00000000000..128627676c4 --- /dev/null +++ b/config/scripts/agent-inspection-cadence-batching-benchmark.mjs @@ -0,0 +1,139 @@ +#!/usr/bin/env node +// Counts how many whole-host process-table captures the agent-completion cadence costs. +// +// Local panes all resolve out of one TTL-deduped snapshot, and the inspection queue collapses +// every shared-observation task enqueued in the same tick onto a single capture. So the capture +// count is the number of DISTINCT wake instants across panes, not the number of pane wakes. +// +// This drives the production interval picker (`nextCadenceInspectionDelayMs`) against a baseline +// that reproduces the pre-change ±10% jitter, over a simulated wall-clock window. +import { spawnSync } from 'node:child_process' +import fs from 'node:fs' +import nodeModule from 'node:module' +import path from 'node:path' +import process from 'node:process' +import { fileURLToPath } from 'node:url' + +if (!process.execArgv.includes('--experimental-transform-types')) { + const result = spawnSync( + process.execPath, + ['--experimental-transform-types', '--no-warnings', import.meta.filename], + { stdio: 'inherit' } + ) + process.exit(result.status ?? 1) +} + +nodeModule.registerHooks({ + resolve(specifier, context, nextResolve) { + if (specifier.startsWith('.') && !/\.[cm]?[jt]s$/.test(specifier) && context.parentURL) { + const candidate = new URL(`${specifier}.ts`, context.parentURL) + if (fs.existsSync(fileURLToPath(candidate))) { + return { url: candidate.href, shortCircuit: true } + } + } + return nextResolve(specifier, context) + } +}) + +const ROOT = path.resolve(import.meta.dirname, '../..') +const WINDOW_MS = Number(process.env.ORCA_INSPECTION_BENCH_WINDOW_MS ?? '60000') +const PANE_COUNTS = (process.env.ORCA_INSPECTION_BENCH_PANES ?? '1,2,4,8') + .split(',') + .map((value) => Number(value.trim())) + +if (!Number.isSafeInteger(WINDOW_MS) || WINDOW_MS <= 0) { + throw new Error(`ORCA_INSPECTION_BENCH_WINDOW_MS must be a positive integer, got ${WINDOW_MS}`) +} +for (const paneCount of PANE_COUNTS) { + if (!Number.isSafeInteger(paneCount) || paneCount <= 0) { + throw new Error(`ORCA_INSPECTION_BENCH_PANES entries must be positive, got ${paneCount}`) + } +} + +const { nextCadenceInspectionDelayMs } = await import( + path.join(ROOT, 'src/renderer/src/components/terminal-pane/agent-completion-poll-interval.ts') +) +const { POLL_TIER_INTERVAL_MS } = await import( + path.join(ROOT, 'src/renderer/src/components/terminal-pane/agent-completion-poll-cadence.ts') +) +const { PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS } = await import( + path.join(ROOT, 'src/shared/process-table-snapshot-reader.ts') +) + +// Pre-change: independent ±10% jitter per pane, re-rolled on every reschedule. +function baselineDelayMs(baseMs) { + return Math.round(baseMs * (1 + (Math.random() * 0.2 - 0.1))) +} + +function simulate(paneCount, baseMs, pickDelay) { + const startedAt = 1_700_000_000_000 + const wakes = [] + for (let pane = 0; pane < paneCount; pane += 1) { + // Panes mount at arbitrary moments, which is what spreads them apart in the first place. + let clock = startedAt + Math.floor(Math.random() * baseMs) + while ((clock += pickDelay(baseMs, clock)) < startedAt + WINDOW_MS) { + wakes.push(clock) + } + } + // A wake is served from the snapshot the previous capture produced until that snapshot's TTL + // lapses, so the TTL window starts at the capture, not on an epoch grid. + let captures = 0 + let snapshotExpiresAt = -Infinity + for (const wakeAt of wakes.sort((left, right) => left - right)) { + if (wakeAt >= snapshotExpiresAt) { + captures += 1 + snapshotExpiresAt = wakeAt + PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS + } + } + return captures +} + +function medianOf(rounds, run) { + const samples = Array.from({ length: rounds }, run).sort((left, right) => left - right) + return samples[Math.floor(samples.length / 2)] +} + +const baseMs = POLL_TIER_INTERVAL_MS.idle +console.log( + `Agent-completion cadence — whole-host \`ps\` captures over ${WINDOW_MS / 1000}s at the idle tier (${baseMs}ms)\n` +) +console.log('| visible panes | before | after | reduction |') +console.log('| --- | --- | --- | --- |') +for (const paneCount of PANE_COUNTS) { + const before = medianOf(21, () => simulate(paneCount, baseMs, baselineDelayMs)) + const after = medianOf(21, () => + simulate(paneCount, baseMs, (base, now) => + nextCadenceInspectionDelayMs({ + baseMs: base, + hasConsecutiveErrors: false, + alignToSharedGrid: true, + now + }) + ) + ) + // A window shorter than one cadence tier can leave the baseline at zero; reporting a + // percentage off that divides by zero and prints a meaningless reduction. + const reduction = before > 0 ? `${(((before - after) / before) * 100).toFixed(0)}%` : 'n/a' + console.log(`| ${paneCount} | ${before} | ${after} | ${reduction} |`) +} + +// Detection latency must not regress: the grid deadline is always within one interval. +let worstDelay = 0 +for (let sample = 0; sample < 100_000; sample += 1) { + const now = 1_700_000_000_000 + sample * 7 + worstDelay = Math.max( + worstDelay, + nextCadenceInspectionDelayMs({ + baseMs, + hasConsecutiveErrors: false, + alignToSharedGrid: true, + now + }) + ) +} +if (worstDelay > baseMs) { + throw new Error(`grid alignment delayed a poll to ${worstDelay}ms, above the ${baseMs}ms tier`) +} +console.log( + `\nWorst observed wait: ${worstDelay}ms (tier interval ${baseMs}ms) — no inspection is ever delayed.` +) diff --git a/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs b/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs index 64fb1b056b8..ba3e64bfa2f 100644 --- a/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs +++ b/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs @@ -1,11 +1,18 @@ -// The relay's copy of the ConPTY teardown release, and the guard that keeps it in lockstep with the -// desktop's own node-pty patch. pnpm patches do not cross the SSH boundary, so a relay runs the tree -// `npm install` put there; the desktop had this fix and the relay did not, and every terminal on a -// Windows SSH host leaked one File handle for the life of the relay process. +// The relay's copy of the ConPTY teardown release, and the guard that keeps it in lockstep with +// `config/patches/node-pty@1.1.0.patch`. pnpm patches do not cross the SSH boundary, so a relay runs +// the tree `npm install` put there, and every terminal on a Windows SSH host leaked one File handle +// for the life of the relay process. // -// The ORDER of the conin release is the fix. Releasing it at the top of the branch -- what the -// desktop patch does -- was measured at 3x WORSE than shipping nothing (File +2/terminal and a new -// Process +1/terminal); releasing it after the console-list fork and the native kill is flat. +// The ORDER of the conin release is the fix. Releasing it at the top of the branch -- the placement +// the desktop patch uses -- was measured at 3x WORSE than shipping nothing (File +2/terminal and a +// new Process +1/terminal); releasing it after the console-list fork and the native kill is flat. +// +// Those numbers are the `!useConptyDll` branch, which is the branch a RELAY runs. Every desktop +// site that opens a terminal pane sets `useConptyDll: true` and takes the other branch, where +// upstream already destroys the input socket. Two hidden rate-limit probes +// (`src/main/rate-limits/claude-pty.ts`, `codex-pty-rate-limit-probe.ts`) do omit the option and so +// do run this hunk, but no user-visible pane does. The divergence pinned below is about which +// branch each host runs for terminals -- not about a regression in the panes users open. import { createRequire } from 'node:module' import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' import { join, resolve } from 'node:path' @@ -90,9 +97,11 @@ describe('Windows SSH relay node-pty ConPTY teardown patch', () => { ) }) - // The one hunk that must NOT match the desktop, and the reason is measured, not stylistic: - // releasing conin before `_getConsoleProcessList()` forks aborts teardown partway. - it('releases conin after the console-list fork, not before it like the desktop patch', () => { + // The one hunk that must NOT match the desktop patch, and the reason is measured, not stylistic: + // on the branch a relay runs, releasing conin before `_getConsoleProcessList()` forks aborts + // teardown partway. Desktop terminal panes take the other branch, so no pane is affected either + // way; what this guards is a patch sync putting the early placement onto the relay's branch. + it('releases conin after the console-list fork, unlike the desktop patch placement', () => { const fixture = writeNodePtyFixture('1.1.0') patchNodePtyWindowsTeardown(fixture.root) const patched = readFileSync(join(fixture.libDir, 'windowsPtyAgent.js'), 'utf8') @@ -108,7 +117,8 @@ describe('Windows SSH relay node-pty ConPTY teardown patch', () => { expect(branch.indexOf('this._inSocket.destroy();')).toBeGreaterThan( branch.indexOf('this._getConsoleProcessList()') ) - // Pinned so a future "sync the relay asset to config/patches" cannot copy the regression back. + // Pinned so a future "sync the relay asset to config/patches" cannot copy the early placement + // onto the relay's branch, where it costs +2 File and +1 Process per terminal. expect(patched).not.toBe(readFileSync(desktopPath('windowsPtyAgent.js'), 'utf8')) }) diff --git a/config/scripts/pr-code-change-scope.mjs b/config/scripts/pr-code-change-scope.mjs index f5a73f6239f..15ded915c67 100644 --- a/config/scripts/pr-code-change-scope.mjs +++ b/config/scripts/pr-code-change-scope.mjs @@ -219,6 +219,7 @@ const WINDOWS_PACKAGE_TESTS = [ 'src/main/agent-hooks/windows-hook-payload-delivery.test.ts', 'src/main/windows/windows-pty-job.win32.test.ts', 'src/main/windows/windows-host-job.win32.test.ts', + 'src/main/windows-live-tree-kill.win32.test.ts', 'src/main/wsl/wsl-runner.test.ts', 'src/main/wsl/wsl-guest-environment.test.ts', 'src/main/wsl/wsl-invocation-boundary.test.ts', diff --git a/config/scripts/renderer-quadratic-scan-benchmark.mjs b/config/scripts/renderer-quadratic-scan-benchmark.mjs new file mode 100644 index 00000000000..681b6cf8550 --- /dev/null +++ b/config/scripts/renderer-quadratic-scan-benchmark.mjs @@ -0,0 +1,364 @@ +#!/usr/bin/env node +// Benchmarks four renderer projections that scaled worse than linearly with user data, each on a +// path that reruns per keystroke or per store write. +// +// Scenarios 1, 3 and 4 time the production export against a hand-written reproduction of the +// pre-change shape and assert both agree first. Scenario 2 is MODELLED on both sides: the +// projection lives inside the `useTabGroupItemProjections` React hook and cannot be imported +// without a renderer, so it reproduces the before/after loops rather than driving production. +import { spawnSync } from 'node:child_process' +import { transformSync } from 'esbuild' +import { performance } from 'node:perf_hooks' +import fs from 'node:fs' +import nodeModule from 'node:module' +import path from 'node:path' +import process from 'node:process' +import { fileURLToPath, pathToFileURL } from 'node:url' + +if (!process.execArgv.includes('--experimental-transform-types')) { + const result = spawnSync( + process.execPath, + ['--experimental-transform-types', '--no-warnings', import.meta.filename], + { stdio: 'inherit' } + ) + process.exit(result.status ?? 1) +} + +const ROOT = path.resolve(import.meta.dirname, '../..') +const RENDERER = path.join(ROOT, 'src/renderer/src') + +nodeModule.registerHooks({ + resolve(specifier, context, nextResolve) { + if (!context.parentURL) { + return nextResolve(specifier, context) + } + const candidates = specifier.startsWith('@/') + ? ['.ts', '.tsx', '/index.ts', '/index.tsx', ''].map( + (suffix) => path.join(RENDERER, specifier.slice(2)) + suffix + ) + : specifier.startsWith('.') && !/\.[cm]?[jt]sx?$/.test(specifier) + ? ['.ts', '.tsx'].map((suffix) => + fileURLToPath(new URL(specifier + suffix, context.parentURL)) + ) + : [] + const resolved = candidates.find((file) => fs.existsSync(file) && fs.statSync(file).isFile()) + return resolved + ? { url: pathToFileURL(resolved).href, shortCircuit: true } + : nextResolve(specifier, context) + }, + // Node strips types from .ts but not .tsx; the sidebar row model transitively imports icons. + load(url, context, nextLoad) { + if (url.endsWith('.tsx')) { + const source = fs.readFileSync(fileURLToPath(url), 'utf8') + const { code } = transformSync(source, { loader: 'tsx', format: 'esm', jsx: 'automatic' }) + return { format: 'module', source: code, shortCircuit: true } + } + if (url.endsWith('.json') && !url.includes('/node_modules/')) { + const source = fs.readFileSync(fileURLToPath(url), 'utf8') + return { format: 'module', source: `export default ${source}`, shortCircuit: true } + } + return nextLoad(url, context) + } +}) + +const importRenderer = (relativePath) => + import(pathToFileURL(path.join(RENDERER, relativePath)).href) + +function envInt(name, fallback) { + const value = Number(process.env[name] ?? fallback) + if (!Number.isSafeInteger(value) || value <= 0) { + throw new Error(`${name} must be a positive integer, got ${value}`) + } + return value +} + +const KEYSTROKES = envInt('ORCA_QUADRATIC_BENCH_KEYSTROKES', 12) +const WORKTREES = envInt('ORCA_QUADRATIC_BENCH_WORKTREES', 300) +const TABS = envInt('ORCA_QUADRATIC_BENCH_TABS', 60) +const OPEN_FILES = envInt('ORCA_QUADRATIC_BENCH_OPEN_FILES', 120) +const CHANGED_FILES = envInt('ORCA_QUADRATIC_BENCH_CHANGED_FILES', 5000) +const SIDEBAR_ROWS = envInt('ORCA_QUADRATIC_BENCH_SIDEBAR_ROWS', 600) +const SIDEBAR_REPOS = envInt('ORCA_QUADRATIC_BENCH_SIDEBAR_REPOS', 80) +if (SIDEBAR_REPOS > SIDEBAR_ROWS) { + throw new Error( + 'ORCA_QUADRATIC_BENCH_SIDEBAR_REPOS must not exceed ORCA_QUADRATIC_BENCH_SIDEBAR_ROWS' + ) +} + +function timeRounds(run, rounds = 7) { + run() + const samples = Array.from({ length: rounds }, () => { + const start = performance.now() + run() + return performance.now() - start + }).sort((left, right) => left - right) + return samples[Math.floor(rounds / 2)] +} + +function repeat(times, run) { + return () => { + let last + for (let round = 0; round < times; round += 1) { + last = run() + } + return last + } +} + +const results = [] +function compare({ label, scale, drives, before, after }) { + if (JSON.stringify(before()) !== JSON.stringify(after())) { + throw new Error(`${label}: baseline disagreed with the indexed shape`) + } + results.push({ label, scale, drives, beforeMs: timeRounds(before), afterMs: timeRounds(after) }) +} + +// ------------------------------------------------- 1. workspace board search index + +const { buildWorkspaceBoardPaletteDocuments, matchWorkspaceBoardWorktrees } = await importRenderer( + 'components/sidebar/workspace-kanban-search.ts' +) + +const repoMap = new Map([ + ['repo-1', { id: 'repo-1', name: 'orca', path: '/tmp/orca', branch: 'main' }] +]) +const boardWorktrees = Array.from({ length: WORKTREES }, (_, index) => ({ + id: `repo-1::/tmp/worktree-${index}`, + repoId: 'repo-1', + path: `/tmp/worktree-${index}`, + branch: `feature/search-target-${index}`, + title: `Workspace ${index} search target`, + isMain: false +})) +const queries = Array.from({ length: KEYSTROKES }, (_, index) => 'search'.slice(0, (index % 6) + 1)) +const matchAll = (documents) => + queries.map((query) => [ + ...matchWorkspaceBoardWorktrees({ worktrees: boardWorktrees, query, repoMap, documents }) + ]) + +compare({ + label: 'workspace board filter (per keystroke burst)', + scale: `${WORKTREES} worktrees x ${KEYSTROKES} keystrokes`, + drives: 'production', + // Omitting `documents` is the pre-change shape: the index is rebuilt inside every match. + before: () => matchAll(undefined), + // The hook memoizes the index on [worktrees, repoMap]; only the match reruns per keystroke. + after: () => matchAll(buildWorkspaceBoardPaletteDocuments({ worktrees: boardWorktrees, repoMap })) +}) + +// ------------------------------------------------- 2. tab-group projections (modelled) + +const groupTabs = Array.from({ length: TABS }, (_, index) => ({ + id: `tab-${index}`, + entityId: `entity-${index}`, + contentType: index % 3 === 0 ? 'editor' : 'terminal' +})) +const openFiles = Array.from({ length: OPEN_FILES }, (_, index) => ({ + id: `entity-${index}`, + path: `/tmp/file-${index}.ts` +})) +const tabOrder = groupTabs.map((tab) => tab.id) +// Production memoizes each index on its own source list, so a unified-tab write reuses it. +const openFileById = new Map(openFiles.map((item) => [item.id, item])) +const groupTabById = new Map(groupTabs.map((item) => [item.id, item])) + +function tabProjections(findOpenFile, findGroupTab) { + const editorItems = groupTabs + .filter((item) => item.contentType === 'editor') + .map((item) => findOpenFile(item.entityId)) + .filter((file) => file !== undefined) + const order = tabOrder.map((itemId) => findGroupTab(itemId)?.entityId ?? itemId) + return [editorItems, order] +} + +compare({ + label: 'tab-group projections (per unified-tab write)', + scale: `${TABS} tabs x ${OPEN_FILES} open files`, + drives: 'modelled', + before: repeat(200, () => + tabProjections( + (id) => openFiles.find((candidate) => candidate.id === id), + (id) => groupTabs.find((candidate) => candidate.id === id) + ) + ), + after: repeat(200, () => + tabProjections( + (id) => openFileById.get(id), + (id) => groupTabById.get(id) + ) + ) +}) + +// ------------------------------------------------- 3. source-control tree build + +const { buildSourceControlTree } = await importRenderer( + 'components/right-sidebar/source-control-tree.ts' +) +const { normalizeRelativePath } = await importRenderer('lib/path.ts') +const { splitPathSegments } = await importRenderer('components/right-sidebar/path-tree.ts') +const { compareFileNames } = await import( + pathToFileURL(path.join(ROOT, 'src/shared/file-name-sort.ts')).href +) + +const changedEntries = Array.from({ length: CHANGED_FILES }, (_, index) => ({ + path: `src/area-${index % 20}/module-${index % 60}/nested/deep/part-${index % 7}/file-${index}.ts` +})) + +// Pre-change `buildSourceControlTree`: identical except each ancestor path is re-joined. +function buildSourceControlTreeBefore(area, entries) { + const makeDirectory = (dirPath, name, depth) => ({ + type: 'directory', + key: `dir::${area}::${dirPath}`, + name, + path: dirPath, + area, + depth, + fileCount: 0, + children: [], + directoryChildren: new Map() + }) + const root = makeDirectory('', '', -1) + for (const entry of entries) { + const normalizedPath = normalizeRelativePath(entry.path) + const segments = splitPathSegments(normalizedPath) + if (segments.length === 0) { + continue + } + let parent = root + for (let index = 0; index < segments.length - 1; index += 1) { + const name = segments[index] + const dirPath = segments.slice(0, index + 1).join('/') + let dir = parent.directoryChildren.get(name) + if (!dir) { + dir = makeDirectory(dirPath, name, index) + parent.directoryChildren.set(name, dir) + parent.children.push(dir) + } + parent = dir + } + parent.children.push({ + type: 'file', + key: `${area}::${entry.path}`, + name: segments.at(-1), + path: normalizedPath, + entry, + area, + depth: segments.length - 1 + }) + } + const finalize = (node) => { + const directories = node.children.filter((child) => child.type === 'directory').map(finalize) + const files = node.children.filter((child) => child.type === 'file') + directories.sort((a, b) => compareFileNames(a.name, b.name)) + files.sort((a, b) => compareFileNames(a.entry.path, b.entry.path)) + const { directoryChildren: _, ...rest } = node + return { + ...rest, + fileCount: files.length + directories.reduce((count, dir) => count + dir.fileCount, 0), + children: [...directories, ...files] + } + } + return finalize(root).children +} + +compare({ + label: 'source-control tree build (per filter keystroke)', + scale: `${CHANGED_FILES} changed files`, + drives: 'production', + before: () => buildSourceControlTreeBefore('unstaged', changedEntries), + after: () => buildSourceControlTree('unstaged', changedEntries) +}) + +// ------------------------------------------------- 4. sidebar header boundaries + +const { getRepoHeaderSectionEndByRepoId } = await importRenderer( + 'components/sidebar/worktree-header-section-boundaries.ts' +) +const { estimateRenderRowSize } = await importRenderer( + 'components/sidebar/worktree-list/viewport/virtual-rows.ts' +) + +const headerRowIndexes = new Set( + Array.from({ length: SIDEBAR_REPOS }, (_, repo) => + Math.floor((repo * SIDEBAR_ROWS) / SIDEBAR_REPOS) + ) +) +const sidebarRows = Array.from({ length: SIDEBAR_ROWS }, (_, index) => + headerRowIndexes.has(index) + ? { + type: 'header', + key: `repo:${index}`, + label: '', + count: 0, + tone: '', + repo: { id: `repo-${index}` } + } + : { type: 'item', rowKey: `wt:${index}`, sectionKey: '', depth: 0, groupDepth: 0 } +) +const headerRepoIds = sidebarRows.filter((row) => row.type === 'header').map((row) => row.repo.id) +const boundaryArgs = { + rows: sidebarRows, + firstHeaderIndex: 0, + // What `getSidebarOrderedRepoHeaderIdsByBucket` yields for repos outside any project group. + sidebarRepoHeaderIdsByBucket: new Map([['ungrouped', headerRepoIds]]), + repoHeaderBucketByRepoId: new Map(headerRepoIds.map((id) => [id, 'ungrouped'])) +} + +// Pre-change `getRepoHeaderSectionEndByRepoId`: a findIndex and an indexOf per header row. +function getRepoHeaderSectionEndByRepoIdBefore(args) { + const rowStarts = [] + let offset = 0 + for (let index = 0; index < args.rows.length; index += 1) { + rowStarts[index] = offset + offset += estimateRenderRowSize(args.rows, index, args.firstHeaderIndex, null) + } + rowStarts[args.rows.length] = offset + const sectionEndByRepoId = new Map() + for (let index = 0; index < args.rows.length; index += 1) { + const row = args.rows[index] + const repoId = row?.type === 'header' ? row.repo?.id : undefined + if (!repoId) { + continue + } + const bucketKey = args.repoHeaderBucketByRepoId.get(repoId) + const bucketRepoIds = bucketKey ? args.sidebarRepoHeaderIdsByBucket.get(bucketKey) : undefined + const bucketIndex = bucketRepoIds?.indexOf(repoId) ?? -1 + const nextRepoId = bucketIndex >= 0 ? bucketRepoIds?.[bucketIndex + 1] : undefined + let endIndex = -1 + if (nextRepoId) { + endIndex = args.rows.findIndex((r) => r.type === 'header' && r.repo?.id === nextRepoId) + } else { + endIndex = args.rows.length + for (let next = index + 1; next < args.rows.length; next += 1) { + if (args.rows[next]?.type === 'header' || args.rows[next]?.type === 'host-header') { + endIndex = next + break + } + } + } + sectionEndByRepoId.set( + repoId, + rowStarts[endIndex >= 0 ? endIndex : args.rows.length] ?? rowStarts[args.rows.length] ?? 0 + ) + } + return sectionEndByRepoId +} + +compare({ + label: 'sidebar header boundaries (per row-model rebuild)', + scale: `${SIDEBAR_REPOS} repos x ${SIDEBAR_ROWS} rows`, + drives: 'production', + before: repeat(50, () => [...getRepoHeaderSectionEndByRepoIdBefore(boundaryArgs)]), + after: repeat(50, () => [...getRepoHeaderSectionEndByRepoId(boundaryArgs)]) +}) + +// ------------------------------------------------- + +console.log('Renderer quadratic-scan removals\n') +console.log('| projection | drives | scale | before | after | |') +console.log('| --- | --- | --- | --- | --- | --- |') +for (const row of results) { + console.log( + `| ${row.label} | ${row.drives} | ${row.scale} | ${row.beforeMs.toFixed(2)} ms | ${row.afterMs.toFixed(2)} ms | ${(row.beforeMs / row.afterMs).toFixed(1)}x |` + ) +} diff --git a/config/scripts/run-headless-linux-pairing-docker.mjs b/config/scripts/run-headless-linux-pairing-docker.mjs index 635c66348cc..799b8d73ab1 100644 --- a/config/scripts/run-headless-linux-pairing-docker.mjs +++ b/config/scripts/run-headless-linux-pairing-docker.mjs @@ -70,7 +70,7 @@ function valueAfter(flag) { function buildImage(image) { console.log(`Building ${image.name} fixture...`) - docker([ + const buildArgs = [ 'build', '--build-arg', `BASE_IMAGE=${image.base}`, @@ -81,7 +81,16 @@ function buildImage(image) { '-t', image.tag, '.' - ]) + ] + // Why: apt fetches from archive.ubuntu.com stall or fail mid-sync; a second build usually lands on a healthy index. + try { + docker(buildArgs) + } catch (error) { + console.error( + `${error instanceof Error ? error.message : String(error)}\nRetrying docker build once...` + ) + docker(buildArgs) + } } function extractAppImage(image) { diff --git a/config/scripts/run-headless-serve-shutdown-docker.mjs b/config/scripts/run-headless-serve-shutdown-docker.mjs index 184713c41a0..d8dcdd345ad 100755 --- a/config/scripts/run-headless-serve-shutdown-docker.mjs +++ b/config/scripts/run-headless-serve-shutdown-docker.mjs @@ -43,7 +43,7 @@ const artifactVolume = `orca-headless-serve-shutdown-${suffix}` const sha256 = createHash('sha256').update(readFileSync(appImage)).digest('hex') try { - docker([ + const buildArgs = [ 'build', '--platform', platform, @@ -52,7 +52,15 @@ try { '-t', image, shutdownDockerDirectory - ]) + ] + // Why: apt fetches from archive.ubuntu.com stall or fail mid-sync; a second build usually lands on a healthy index. + const firstBuild = docker(buildArgs, { allowFailure: true }) + if (firstBuild.status !== 0) { + process.stderr.write( + `${firstBuild.stdout}${firstBuild.stderr}\ndocker build failed with status ${firstBuild.status}; retrying once...\n` + ) + docker(buildArgs) + } docker(['volume', 'create', artifactVolume]) runDesktopStartupOracle({ image, appImage, platform }) docker([ diff --git a/config/scripts/run-linux-cli-launch-contract-docker.mjs b/config/scripts/run-linux-cli-launch-contract-docker.mjs index 901e0877e85..bd414947026 100755 --- a/config/scripts/run-linux-cli-launch-contract-docker.mjs +++ b/config/scripts/run-linux-cli-launch-contract-docker.mjs @@ -173,20 +173,26 @@ function runCase(caseName) { function buildImage() { console.log(`Building ${tag}…`) - docker( - [ - 'build', - ...dockerPlatformArgs, - '--build-arg', - `BASE_IMAGE=${base}`, - '-f', - 'config/docker/cli-launch-contract/Dockerfile', - '-t', - tag, - 'config/docker/cli-launch-contract' - ], - { timeoutMs: BUILD_TIMEOUT_MS } - ) + const buildArgs = [ + 'build', + ...dockerPlatformArgs, + '--build-arg', + `BASE_IMAGE=${base}`, + '-f', + 'config/docker/cli-launch-contract/Dockerfile', + '-t', + tag, + 'config/docker/cli-launch-contract' + ] + // Why: apt fetches from archive.ubuntu.com stall or fail mid-sync; a second build usually lands on a healthy index. + try { + docker(buildArgs, { timeoutMs: BUILD_TIMEOUT_MS }) + } catch (error) { + console.error( + `${error instanceof Error ? error.message : String(error)}\nRetrying docker build once…` + ) + docker(buildArgs, { timeoutMs: BUILD_TIMEOUT_MS }) + } } // Extract unprivileged so chrome-sandbox is not root-owned setuid. diff --git a/config/scripts/session-write-hot-path-benchmark.mjs b/config/scripts/session-write-hot-path-benchmark.mjs new file mode 100644 index 00000000000..361e7eb7098 --- /dev/null +++ b/config/scripts/session-write-hot-path-benchmark.mjs @@ -0,0 +1,221 @@ +#!/usr/bin/env node +// Benchmarks two CPU costs `setLocalWorkspaceSession` pays on every session write — the write +// that fires on something as ordinary as clicking between two terminal split panes. +// +// 1. capTerminalScrollbackSessionBuffer — UTF-8 budget scan per retained scrollback buffer +// 2. remapPaneKeys — pane-key map rebuild that steady state throws away +// +// The snapshot disk rewrite on the same path is measured separately (#18764). +// +// Each scenario runs the production export against a baseline that reproduces the pre-change +// shape, so the reported speedup cannot drift away from what production actually does. +import { spawnSync } from 'node:child_process' +import { performance } from 'node:perf_hooks' +import fs from 'node:fs' +import nodeModule from 'node:module' +import path from 'node:path' +import process from 'node:process' +import { fileURLToPath } from 'node:url' + +if (!process.execArgv.includes('--experimental-transform-types')) { + const result = spawnSync( + process.execPath, + ['--experimental-transform-types', '--no-warnings', import.meta.filename], + { stdio: 'inherit' } + ) + process.exit(result.status ?? 1) +} + +// The app's TS sources import siblings without an extension; Node's ESM resolver needs it. +nodeModule.registerHooks({ + resolve(specifier, context, nextResolve) { + if (specifier.startsWith('.') && !/\.[cm]?[jt]s$/.test(specifier) && context.parentURL) { + const candidate = new URL(`${specifier}.ts`, context.parentURL) + if (fs.existsSync(fileURLToPath(candidate))) { + return { url: candidate.href, shortCircuit: true } + } + } + return nextResolve(specifier, context) + } +}) + +const ROOT = path.resolve(import.meta.dirname, '../..') +const ROUNDS = Number(process.env.ORCA_SESSION_WRITE_BENCH_ROUNDS ?? '9') +const LEAVES = Number(process.env.ORCA_SESSION_WRITE_BENCH_LEAVES ?? '8') +const PANE_KEYS = Number(process.env.ORCA_SESSION_WRITE_BENCH_PANE_KEYS ?? '2000') + +for (const [name, value] of [ + ['ORCA_SESSION_WRITE_BENCH_ROUNDS', ROUNDS], + ['ORCA_SESSION_WRITE_BENCH_LEAVES', LEAVES], + ['ORCA_SESSION_WRITE_BENCH_PANE_KEYS', PANE_KEYS] +]) { + if (!Number.isSafeInteger(value) || value <= 0) { + throw new Error(`${name} must be a positive integer, got ${value}`) + } +} + +const { capTerminalScrollbackSessionBuffer } = await import( + path.join(ROOT, 'src/shared/workspace-session-terminal-buffers.ts') +) +const { TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT } = await import( + path.join(ROOT, 'src/shared/terminal-scrollback-limits.ts') +) +const { remapAcknowledgedAgentPaneKeys } = await import( + path.join(ROOT, 'src/main/persistence/restoring-sessions/pane-key-remapping.ts') +) +const { clampUtf8TextTail, measureUtf8ByteLength } = await import( + path.join(ROOT, 'src/shared/utf8-byte-limits.ts') +) +const { isTerminalLeafId, makePaneKey, parsePaneKey } = await import( + path.join(ROOT, 'src/shared/stable-pane-id.ts') +) + +function median(samples) { + const sorted = [...samples].sort((left, right) => left - right) + return sorted[Math.floor(sorted.length / 2)] +} + +function timeRounds(run) { + const samples = [] + run() + for (let round = 0; round < ROUNDS; round += 1) { + const start = performance.now() + run() + samples.push(performance.now() - start) + } + return median(samples) +} + +function report(label, baselineMs, currentMs, extra = '') { + const speedup = baselineMs / currentMs + console.log( + `${label}\n before ${baselineMs.toFixed(3)} ms → after ${currentMs.toFixed(3)} ms (${speedup.toFixed(1)}x)${extra}` + ) + return speedup +} + +// ---------------------------------------------------------------- scenario 1 + +// Verbatim pre-change capTerminalScrollbackSessionBuffer; measureUtf8ByteLength itself is unchanged. +function baselineCapScrollbackBuffer(buffer) { + if ( + buffer.length <= TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT && + !measureUtf8ByteLength(buffer, { + stopAfterBytes: TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT + }).exceededLimit + ) { + return buffer + } + return clampUtf8TextTail(buffer, TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT).text +} + +// A terminal that has been running a while sits at the cap, which is the case that scanned in full. +const scrollbackLine = `${''}build output line with a path /Users/dev/project/src/index.ts and a status ok\n` +let atCapBuffer = '' +while (atCapBuffer.length < TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT) { + atCapBuffer += scrollbackLine +} +atCapBuffer = atCapBuffer.slice(0, TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT) + +if (capTerminalScrollbackSessionBuffer(atCapBuffer) !== baselineCapScrollbackBuffer(atCapBuffer)) { + throw new Error('scrollback cap disagreed with the baseline implementation') +} + +// The session write runs the prune twice, once per retained leaf. +const CAP_CALLS_PER_WRITE = LEAVES * 2 +const capBaselineMs = timeRounds(() => { + for (let call = 0; call < CAP_CALLS_PER_WRITE; call += 1) { + baselineCapScrollbackBuffer(atCapBuffer) + } +}) +const capCurrentMs = timeRounds(() => { + for (let call = 0; call < CAP_CALLS_PER_WRITE; call += 1) { + capTerminalScrollbackSessionBuffer(atCapBuffer) + } +}) + +console.log( + `Session-write hot path — ${LEAVES} retained scrollback leaves, ${PANE_KEYS} accumulated pane keys\n` +) +report( + `1. scrollback UTF-8 budget scan (${CAP_CALLS_PER_WRITE} calls/write @ ${(atCapBuffer.length / 1024).toFixed(0)} KB)`, + capBaselineMs, + capCurrentMs +) + +// ---------------------------------------------------------------- scenario 2 + +const paneKeys = {} +const leafIdByInputLeafIdByTabId = new Map() +for (let index = 0; index < PANE_KEYS; index += 1) { + const tabId = `tab-${index % 64}` + const leafId = `${(index % 64).toString(16).padStart(8, '0')}-0000-4000-8000-${index.toString(16).padStart(12, '0')}` + paneKeys[makePaneKey(tabId, leafId)] = index + let leaves = leafIdByInputLeafIdByTabId.get(tabId) + if (!leaves) { + leaves = new Map() + leafIdByInputLeafIdByTabId.set(tabId, leaves) + } + // Steady state: a stable UUID leaf maps to itself. + leaves.set(leafId, leafId) +} + +// Verbatim pre-change remapPaneKeys: parses every key, then rebuilds the object regardless. +function baselineRemapPaneKeys(values, remap) { + if (!values || Object.keys(values).length === 0) { + return { values, changed: false } + } + let changed = false + const next = {} + const setValue = (paneKey, value) => { + const existing = next[paneKey] + next[paneKey] = existing === undefined ? value : Math.max(existing, value) + } + for (const [paneKey, value] of Object.entries(values)) { + if (parsePaneKey(paneKey)) { + setValue(paneKey, value) + continue + } + const delimiter = paneKey.indexOf(':') + if (delimiter <= 0 || delimiter === paneKey.length - 1) { + setValue(paneKey, value) + continue + } + const tabId = paneKey.slice(0, delimiter) + const remappedLeafId = remap.get(tabId)?.get(paneKey.slice(delimiter + 1)) + if (!remappedLeafId || !isTerminalLeafId(remappedLeafId)) { + setValue(paneKey, value) + continue + } + try { + setValue(makePaneKey(tabId, remappedLeafId), value) + changed = true + } catch { + setValue(paneKey, value) + } + } + return { values: next, changed } +} + +// The write remaps three of these maps: acknowledgements, activity cutoffs, manual unread. +const REMAP_CALLS_PER_WRITE = 3 +const remapBaselineMs = timeRounds(() => { + for (let call = 0; call < REMAP_CALLS_PER_WRITE; call += 1) { + baselineRemapPaneKeys(paneKeys, leafIdByInputLeafIdByTabId) + } +}) +const remapCurrentMs = timeRounds(() => { + for (let call = 0; call < REMAP_CALLS_PER_WRITE; call += 1) { + remapAcknowledgedAgentPaneKeys(paneKeys, leafIdByInputLeafIdByTabId) + } +}) +const remapResult = remapAcknowledgedAgentPaneKeys(paneKeys, leafIdByInputLeafIdByTabId) +if (remapResult.changed || remapResult.acknowledgements !== paneKeys) { + throw new Error('steady-state remap should return the input map untouched') +} +report( + `2. pane-key remap (${REMAP_CALLS_PER_WRITE} maps/write @ ${PANE_KEYS} keys)`, + remapBaselineMs, + remapCurrentMs, + ' — and 3 discarded objects/write become 0' +) diff --git a/config/scripts/terminal-partial-escape-tail-benchmark.mjs b/config/scripts/terminal-partial-escape-tail-benchmark.mjs new file mode 100644 index 00000000000..0337daa4e9d --- /dev/null +++ b/config/scripts/terminal-partial-escape-tail-benchmark.mjs @@ -0,0 +1,88 @@ +#!/usr/bin/env node +// Times the partial-escape-tail fold that runs once per PTY chunk for every terminal against a +// baseline with the pre-change shape (unconditional concat + per-code-unit walk). Equivalence is +// proven over a corpus first, so the reported speedup cannot come from the gate changing the answer. +import { performance } from 'node:perf_hooks' +import { + advancePartialEscapeTail, + extractPartialEscapeTail, + MAX_PARTIAL_ESCAPE_TAIL_LENGTH +} from '../../src/shared/terminal-partial-escape-tail.ts' + +const CHUNK_BYTES = 16 * 1024 +const CHUNKS = 640 +const ROUNDS = 7 + +function baselineAdvance(pendingTail, chunk) { + const tail = extractPartialEscapeTail(pendingTail + chunk) + return tail.length > MAX_PARTIAL_ESCAPE_TAIL_LENGTH ? '' : tail +} + +const chunkOf = (line) => line.repeat(Math.ceil(CHUNK_BYTES / line.length)).slice(0, CHUNK_BYTES) +const escFreeChunk = chunkOf('[build] compiled src/renderer/src/components/thing.tsx in 12ms\n') +const colouredChunk = chunkOf( + '\x1b[32m[build]\x1b[0m compiled src/renderer/src/components/thing.tsx in 12ms\n' +) + +// Every state the scanner can be left in, plus the boundaries the gate must not swallow. +const PIECES = [ + '', + 'plain output\n', + '\x1b[32mgreen\x1b[0m', + '\x1b[3', + '\x1b]0;title\x07', + '\x1b]0;partial', + '\x1bP dcs payload', + '\x1b', + '\x18', + '\x1a', + '\x1b]8;;https://example.com\x1b\\', + '\x1b]8;;https://example.com\x1b', + '\x1b(B', + '\x1b(', + '\x1b[1;2;3', + escFreeChunk +] +let checked = 0 +for (const pending of PIECES.map((piece) => extractPartialEscapeTail(piece))) { + for (const chunk of PIECES) { + const expected = baselineAdvance(pending, chunk) + const actual = advancePartialEscapeTail(pending, chunk) + if (expected !== actual) { + throw new Error( + `gate changed the tracked tail: ${JSON.stringify({ pending, chunk, expected, actual })}` + ) + } + checked += 1 + } +} + +function medianMs(advance, chunk) { + // First sample is the warm-up and is discarded. + const samples = Array.from({ length: ROUNDS + 1 }, () => { + const start = performance.now() + let tail = '' + for (let index = 0; index < CHUNKS; index += 1) { + tail = advance(tail, chunk) + } + return performance.now() - start + }) + return samples.slice(1).sort((left, right) => left - right)[Math.floor(ROUNDS / 2)] +} + +const megabytes = ((CHUNK_BYTES * CHUNKS) / 1024 / 1024).toFixed(1) +console.log( + `Partial-escape-tail fold: ${CHUNKS} x ${CHUNK_BYTES / 1024} KB chunks (${megabytes} MB), ${checked} equivalence cases verified\n` +) +console.log('| stream shape | before | after | |') +console.log('| --- | --- | --- | --- |') +for (const [label, chunk] of [ + ['ESC-free (build logs, `cat`, piped output)', escFreeChunk], + ['SGR-coloured output (gate does not apply)', colouredChunk] +]) { + const before = medianMs(baselineAdvance, chunk) + const after = medianMs(advancePartialEscapeTail, chunk) + console.log( + `| ${label} | ${before.toFixed(2)} ms | ${after.toFixed(2)} ms | ${(before / after).toFixed(1)}x |` + ) +} diff --git a/config/scripts/verify-localization-catalog.mjs b/config/scripts/verify-localization-catalog.mjs index 002a84360f6..a73e9d5e3cc 100644 --- a/config/scripts/verify-localization-catalog.mjs +++ b/config/scripts/verify-localization-catalog.mjs @@ -11,7 +11,12 @@ import { repairTranslatedValue } from './locale-translation-policy.mjs' const SOURCE_EXTENSIONS = new Set(['.ts', '.tsx', '.js', '.jsx', '.mts', '.cts']) const SKIP_PATH_PARTS = new Set(['.git', 'dist', 'node_modules', 'out', '__snapshots__', 'assets']) -const LOCALIZATION_FUNCTION_NAMES = new Set(['t', 'translate', 'translateMain', 'translateSearchKeyword']) +const LOCALIZATION_FUNCTION_NAMES = new Set([ + 't', + 'translate', + 'translateMain', + 'translateSearchKeyword' +]) const PLACEHOLDER_RE = /\{\{[^}]+\}\}/g const LOCALES_RELATIVE_DIR = path.join('src', 'renderer', 'src', 'i18n', 'locales') export const LOCALIZATION_SOURCE_ROOTS = [ diff --git a/docs/assets/readme-downloads.svg b/docs/assets/readme-downloads.svg index c240c965fee..fe660c42295 100644 --- a/docs/assets/readme-downloads.svg +++ b/docs/assets/readme-downloads.svg @@ -1,5 +1,5 @@ - - downloads: 39m + + downloads: 40m @@ -15,7 +15,7 @@ downloads downloads - 39m - 39m + 40m + 40m diff --git a/docs/site/content/docs/browser/profiles.mdx b/docs/site/content/docs/browser/profiles.mdx index 97286a6cb0c..102dd1e443d 100644 --- a/docs/site/content/docs/browser/profiles.mdx +++ b/docs/site/content/docs/browser/profiles.mdx @@ -9,9 +9,7 @@ Browser-use profiles let you run the Orca browser with a specific identity — a 1. Open [Settings → Browser → Profiles](/docs/settings). 1. Click **Add profile**, give it a name. 1. Optionally seed it with cookies, a user-agent, and a viewport size. -1. For sites that reject Orca's default Chrome-shaped UA (some Google sign-in flows), create a profile that keeps the **native Electron user agent** instead of spoofing. Default profiles still use the cleaned Chrome UA for broader Cloudflare compatibility. - -You can also create a no-spoof profile from the CLI with `orca tab profile create --no-ua-spoof` when you script browser setup. +1. Every profile presents Electron's own user agent. Orca no longer rewrites it to look like Chrome, because Cloudflare Turnstile rejects a Chrome-shaped UA that sends no client hints and accepts a declared Electron client. The only exception is Google's sign-in hosts, where Orca presents a Firefox identity so Google issues cookies bound to the embedded browser. A **native user agent** profile (`orca tab profile create --no-ua-spoof`) also skips that Google exception. ## Cookie import and Google sign-in diff --git a/mobile/app.json b/mobile/app.json index d5a420cc74b..131e3899396 100644 --- a/mobile/app.json +++ b/mobile/app.json @@ -2,7 +2,7 @@ "expo": { "name": "Orca", "slug": "orca-mobile", - "version": "0.0.47", + "version": "0.0.48", "orientation": "default", "icon": "./assets/icon.png", "userInterfaceStyle": "automatic", diff --git a/mobile/src/components/MobileMarkdown.tsx b/mobile/src/components/MobileMarkdown.tsx index cc88c01e564..2f5b52cfe18 100644 --- a/mobile/src/components/MobileMarkdown.tsx +++ b/mobile/src/components/MobileMarkdown.tsx @@ -192,6 +192,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile return ( {renderInline(block.text, onOpenFile)} @@ -201,7 +202,9 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile if (block.type === 'quote') { return ( - {renderInline(block.text, onOpenFile)} + + {renderInline(block.text, onOpenFile)} + ) } @@ -223,7 +226,9 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile return ( {block.language ? {block.language} : null} - {block.text} + + {block.text} + ) } @@ -251,7 +256,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile {visibleHeaders.map((header, cellIndex) => ( - + {renderInline(header, onOpenFile)} ))} @@ -259,7 +264,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile {visibleRows.map((row, rowIndex) => ( {visibleHeaders.map((_, cellIndex) => ( - + {renderInline(row[cellIndex] ?? '', onOpenFile)} ))} @@ -290,7 +295,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile ? '[x]' : '[ ]'} - + {renderInline(item.text, onOpenFile)} diff --git a/mobile/src/session/MobileNativeChatMessage.test.ts b/mobile/src/session/MobileNativeChatMessage.test.ts index 10b96cc3d20..677b76d240c 100644 --- a/mobile/src/session/MobileNativeChatMessage.test.ts +++ b/mobile/src/session/MobileNativeChatMessage.test.ts @@ -6,11 +6,21 @@ import type { NativeChatMessage } from '../../../src/shared/native-chat-types' vi.mock('react-native', async () => { const React = await import('react') + const Text = ({ children, ...props }: { children?: unknown }): unknown => + React.createElement('Text', props, children) return { + Animated: { + Text, + Value: class { + setValue(): void {} + }, + loop: (animation: unknown) => animation, + sequence: () => ({ start: vi.fn(), stop: vi.fn() }), + timing: () => ({ start: vi.fn(), stop: vi.fn() }) + }, Image: 'Image', Pressable: 'Pressable', - Text: ({ children, ...props }: { children?: unknown }) => - React.createElement('Text', props, children), + Text, View: ({ children, ...props }: { children?: unknown }) => React.createElement('View', props, children), StyleSheet: { create: (styles: unknown) => styles, hairlineWidth: 1 } @@ -21,7 +31,10 @@ vi.mock('lucide-react-native', () => ({ ArrowUp: 'ArrowUp', ChevronDown: 'ChevronDown', Copy: 'Copy', - SquareChevronRight: 'SquareChevronRight' + SquareChevronRight: 'SquareChevronRight', + SquareTerminal: 'SquareTerminal', + Wrench: 'Wrench', + ChevronRight: 'ChevronRight' })) vi.mock('../components/MobileMarkdown', () => ({ MobileMarkdown: 'MobileMarkdown' })) @@ -45,7 +58,18 @@ describe('MobileNativeChatMessage', () => { function render( message: NativeChatMessage, - props: { toolsExpanded?: boolean } = {} + props: { + toolsExpanded?: boolean + structuredActivityUi?: boolean + activeTurnIsWorking?: boolean + turnExpanded?: boolean + turnStatus?: { + startedAt: number | null + thinking: boolean + workedSeconds: number | null + } | null + onToggleTurn?: () => void + } = {} ): ReactTestRenderer { act(() => { renderer = create(createElement(MobileNativeChatMessage, { message, ...props })) @@ -152,4 +176,96 @@ describe('MobileNativeChatMessage', () => { expect(tree.root.findAllByType('ChevronDown' as never)).toHaveLength(1) expect(tree.root.findAllByType('SquareChevronRight' as never)).toHaveLength(1) }) + + describe('structured activity UI', () => { + const runningCall = { + type: 'tool-call' as const, + name: 'Bash', + input: { command: 'npm test' }, + state: 'running' as const + } + const settledCall = { + type: 'tool-call' as const, + name: 'Read', + input: { file_path: 'a/b.ts' }, + state: 'completed' as const + } + + it('shows the live tool label with a terminal glyph while a command runs', () => { + const tree = render(toolMessage([runningCall]), { + structuredActivityUi: true, + activeTurnIsWorking: true + }) + expect(textIn(tree.root)).toContain('Running npm test') + expect(tree.root.findAllByType('SquareTerminal' as never)).toHaveLength(1) + expect(tree.root.findAllByType('Wrench' as never)).toHaveLength(0) + }) + + it('uses the wrench glyph for a non-command tool', () => { + const tree = render( + toolMessage([ + { type: 'tool-call', name: 'Read', input: { file_path: 'a/b.ts' }, state: 'running' } + ]), + { structuredActivityUi: true, activeTurnIsWorking: true } + ) + expect(textIn(tree.root)).toContain('Running Read a/b.ts') + expect(tree.root.findAllByType('Wrench' as never)).toHaveLength(1) + }) + + it('falls back to the collapsed count row once the run settles', () => { + const tree = render(toolMessage([settledCall]), { + structuredActivityUi: true, + activeTurnIsWorking: true + }) + expect(textIn(tree.root)).not.toContain('Running Read a/b.ts') + expect(textIn(tree.root)).toContain('1×') + }) + + it("hides a completed turn's activity until the turn caret discloses it", () => { + const collapsed = render(toolMessage([settledCall]), { + structuredActivityUi: true, + activeTurnIsWorking: false + }) + expect(textIn(collapsed.root)).not.toContain('1×') + act(() => collapsed.unmount()) + + const disclosed = render(toolMessage([settledCall]), { + structuredActivityUi: true, + activeTurnIsWorking: false, + turnExpanded: true + }) + expect(textIn(disclosed.root)).toContain('1×') + }) + + it('lets the global Tools toggle reveal a hidden settled run', () => { + // Otherwise the composer's Tools control is a no-op on every settled turn. + const tree = render(toolMessage([settledCall]), { + structuredActivityUi: true, + activeTurnIsWorking: false, + toolsExpanded: true + }) + expect(textIn(tree.root)).toContain('1\u00d7') + }) + + it('keeps the bridge lane on its always-visible tool run', () => { + const tree = render(toolMessage([settledCall]), { activeTurnIsWorking: false }) + expect(textIn(tree.root)).toContain('1×') + expect(tree.root.findAllByType('Wrench' as never)).toHaveLength(0) + }) + + it('renders the turn status row under a user message', () => { + const tree = render(userMessage([{ type: 'text', text: 'go' }]), { + structuredActivityUi: true, + turnStatus: { startedAt: Date.now(), thinking: true, workedSeconds: null } + }) + expect(textIn(tree.root)).toContain('Thinking') + }) + + it('does not render a turn status row without one', () => { + const tree = render(userMessage([{ type: 'text', text: 'go' }]), { + structuredActivityUi: true + }) + expect(textIn(tree.root)).toEqual(['go']) + }) + }) }) diff --git a/mobile/src/session/MobileNativeChatMessage.tsx b/mobile/src/session/MobileNativeChatMessage.tsx index 0a676061e5b..cc6c086b1f3 100644 --- a/mobile/src/session/MobileNativeChatMessage.tsx +++ b/mobile/src/session/MobileNativeChatMessage.tsx @@ -1,139 +1,20 @@ import { memo, useEffect, useRef, useState } from 'react' import { Image, Pressable, Text, View } from 'react-native' import * as Clipboard from 'expo-clipboard' -import { ArrowUp, ChevronDown, Copy, SquareChevronRight } from 'lucide-react-native' -import { diffFromText, diffFromToolCall } from '../../../src/shared/native-chat-diff' -import type { NativeChatDiffLine as DiffLine } from '../../../src/shared/native-chat-diff' -import { pairToolBlocks, splitNativeChatBlocks } from '../../../src/shared/native-chat-tool-fold' -import type { NativeChatToolPair as ToolPair } from '../../../src/shared/native-chat-tool-fold' -import { - createToolInputDisplay, - summarizeToolRun, - truncateToolDetail -} from '../../../src/shared/native-chat-tool-summary' +import { ArrowUp, Copy } from 'lucide-react-native' +import { splitNativeChatBlocks } from '../../../src/shared/native-chat-tool-fold' +import { selectActiveToolCall } from '../../../src/shared/native-chat-tool-activity' import { isImageRefBlock, isTextBlock } from '../../../src/shared/native-chat-types' import type { NativeChatBlock, NativeChatMessage } from '../../../src/shared/native-chat-types' import { MobileMarkdown } from '../components/MobileMarkdown' +import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus' +import { ToolRun } from './MobileNativeChatToolRun' +import type { NativeChatTurnStatus } from './use-mobile-native-chat-turn-status' import { colors } from '../theme/mobile-theme' import { isRenderableImageUri } from './mobile-native-chat-image-preview' import { styles, TEXT_SIZE } from './mobile-native-chat-message-styles' import { nativeChatMessageText } from './mobile-native-chat-message-text' -const MAX_VISIBLE_TOOL_PAIRS = 6 -const MAX_TOOL_RUN_DIFF_ROWS = 240 - -function DiffView({ lines }: { lines: DiffLine[] }): React.JSX.Element { - return ( - - {lines.map((line, i) => ( - - {line.kind === 'add' ? '+' : line.kind === 'del' ? '-' : ' '} - {line.text} - - ))} - - ) -} - -/** A single inline tool line — `▸ ToolName preview` — that expands in place to - * show the call's diff/input or the result's body. Mirrors the reference design - * where tool calls read as flat lines in the conversation, not boxed blocks. */ -function ResultBody({ - output, - isError, - diff -}: { - output: string - isError?: boolean - diff: DiffLine[] | null -}): React.JSX.Element { - if (diff) { - return - } - return ( - - {truncateToolDetail(output)} - - ) -} - -/** One request: a tool call and its result rendered together as a single - * expandable line. `defaultExpanded` lets the group toggle open every line. */ -function ToolLine({ - pair, - defaultExpanded, - diffLineLimit, - onOpenFile -}: { - pair: ToolPair - defaultExpanded: boolean - diffLineLimit: number - onOpenFile?: (relativePath: string) => void -}): React.JSX.Element { - const [expanded, setExpanded] = useState(defaultExpanded) - const { call, result } = pair - const name = call ? call.name : 'Result' - const inputDisplay = call ? createToolInputDisplay(call.input) : null - const preview = inputDisplay?.label ?? result?.output.split('\n')[0]?.slice(0, 80) ?? '' - // Why: collapsed tool rows are the common path; defer bounded diff parsing - // and detail formatting until the user asks to reveal the detail. - const callDiff = expanded && call ? diffFromToolCall(call.name, call.input, diffLineLimit) : null - const resultDiff = expanded && result ? diffFromText(result.output, diffLineLimit) : null - const callDetail = expanded && inputDisplay && !callDiff ? inputDisplay.formatDetail() : undefined - const hasDetail = callDiff !== null || result !== undefined || inputDisplay?.hasDetail === true - // The group toggle opens every line at once, bypassing the tap guard, so the - // panel has to consult it too — else a detail-less row echoes its own label - // under itself and no tap can dismiss it. - const showDetail = hasDetail && expanded - // A tool that targets a file (Read/Edit/Write…) renders its preview as a - // tappable link that opens the file, independent of the line's expand tap. - const filePath = inputDisplay?.filePath ?? null - const openable = filePath !== null && onOpenFile !== undefined - return ( - - hasDetail && setExpanded((v) => !v)} - hitSlop={6} - > - {showDetail ? ( - - ) : ( - - )} - {name} - {preview ? ( - onOpenFile!(filePath!) : undefined} - suppressHighlighting={!openable} - > - {preview} - - ) : null} - - {showDetail ? ( - - {callDiff ? : null} - {callDetail ? {callDetail} : null} - {result ? ( - - ) : null} - - ) : null} - - ) -} - function Prose({ block, invert, @@ -150,7 +31,9 @@ function Prose({ // markdown renderer's light-on-dark palette. if (invert) { return ( - {block.text} + + {block.text} + ) } return ( @@ -180,67 +63,6 @@ function Prose({ return null } -/** A run of a message's tool calls/results, collapsed to a one-line summary that - * expands to the individual inline tool lines. `defaultExpanded` lets the global - * toolbar toggle drive every run at once while still allowing per-run override. */ -function ToolRun({ - blocks, - defaultExpanded, - trailing, - onOpenFile -}: { - blocks: NativeChatBlock[] - defaultExpanded: boolean - trailing?: React.ReactNode - onOpenFile?: (relativePath: string) => void -}): React.JSX.Element { - const [open, setOpen] = useState(defaultExpanded) - const pairs = pairToolBlocks(blocks, MAX_VISIBLE_TOOL_PAIRS) - const diffLineLimit = Math.max(1, Math.floor(MAX_TOOL_RUN_DIFF_ROWS / (pairs.length * 2 || 1))) - let callCount = 0 - for (const block of blocks) { - if (block.type === 'tool-call') { - callCount++ - } - } - callCount ||= pairs.length - const summary = summarizeToolRun(blocks) - return ( - - - setOpen((v) => !v)} hitSlop={6}> - {open ? ( - - ) : ( - - )} - {callCount}× - - {summary || `${callCount} tool ${callCount === 1 ? 'call' : 'calls'}`} - - - {trailing} - - {open ? ( - - {pairs.map((pair, i) => ( - - ))} - {callCount > pairs.length ? ( - … {callCount - pairs.length} more tool calls - ) : null} - - ) : null} - - ) -} - /** Subtle top-right controls for an agent message: copy its prose, or scroll so * this message's top aligns to the top of the viewport. */ function AgentControls({ @@ -280,7 +102,13 @@ function MobileNativeChatMessageImpl({ fontScale = 1, messageIndex, onScrollToMessage, - onOpenFile + onOpenFile, + turnStatus, + turnExpanded, + turnKey, + onToggleTurn, + activeTurnIsWorking, + structuredActivityUi = false }: { message: NativeChatMessage toolsExpanded?: boolean @@ -291,6 +119,18 @@ function MobileNativeChatMessageImpl({ /** Ask the list to align this message's top to the top of the viewport. */ onScrollToMessage?: (index: number) => void onOpenFile?: (relativePath: string) => void + /** This turn's status row, rendered under a user message (desktop parity). */ + turnStatus?: NativeChatTurnStatus | null + /** Whether the turn caret has disclosed this turn's activity. */ + turnExpanded?: boolean + /** Set only when this row's turn has settled and can disclose its activity. */ + turnKey?: string + /** Stable across renders; the row supplies its own key when tapped. */ + onToggleTurn?: (turnKey: string) => void + /** Session-level working state for this message's turn; gates the live tool row. */ + activeTurnIsWorking?: boolean + /** Structured lane only: live tool progress plus the turn-status disclosure. */ + structuredActivityUi?: boolean }): React.JSX.Element { const isUser = message.role === 'user' const isReasoning = message.role === 'reasoning' @@ -310,6 +150,20 @@ function MobileNativeChatMessageImpl({ // tool calls fold into a collapsible run beneath. The user's own messages get // an inverted (filled accent) bubble so they stand apart from agent prose. const { prose, tools } = splitNativeChatBlocks(message.blocks) + const activeCall = structuredActivityUi + ? selectActiveToolCall(tools, { activeTurnIsWorking }) + : null + // A completed turn's activity belongs behind the turn-status caret. Leaving the + // grouped row visible made a failed child command read as a failed response. + // The composer's global Tools toggle still overrides this, or it would silently + // do nothing on every settled turn. + const settledToolsHidden = + structuredActivityUi && + activeCall == null && + activeTurnIsWorking === false && + !turnExpanded && + !toolsExpanded + const showToolRun = tools.length > 0 && !settledToolsHidden const handleCopy = (): void => { const text = nativeChatMessageText(message.blocks) @@ -338,39 +192,52 @@ function MobileNativeChatMessageImpl({ ) : null return ( - - - {prose.map((block, index) => ( - - ))} - {tools.length > 0 ? ( - - ) : controls ? ( - {controls} - ) : null} + <> + + + {prose.map((block, index) => ( + + ))} + {showToolRun ? ( + + ) : controls ? ( + {controls} + ) : null} + - + {turnStatus ? ( + onToggleTurn(turnKey) : undefined} + /> + ) : null} + ) } diff --git a/mobile/src/session/MobileNativeChatOverlay.tsx b/mobile/src/session/MobileNativeChatOverlay.tsx index 23724300ddf..357a089466e 100644 --- a/mobile/src/session/MobileNativeChatOverlay.tsx +++ b/mobile/src/session/MobileNativeChatOverlay.tsx @@ -71,6 +71,7 @@ export function MobileNativeChatOverlay({ error={session.error} agent={controller.nativeChatAgent} agentWorking={controller.nativeChatAgentWorking} + structuredActivityUi={controller.nativeChatStructured} streaming={streaming} onStop={controller.handleNativeChatStop} ask={controller.nativeChatAsk} diff --git a/mobile/src/session/MobileNativeChatPromptCard.tsx b/mobile/src/session/MobileNativeChatPromptCard.tsx new file mode 100644 index 00000000000..470ba2ee8b6 --- /dev/null +++ b/mobile/src/session/MobileNativeChatPromptCard.tsx @@ -0,0 +1,74 @@ +import type { AskAnswerSelection, AskPrompt } from '../../../src/shared/native-chat-ask' +import { MobileNativeChatAsk } from './MobileNativeChatAsk' +import { MobileNativeChatPermission } from './MobileNativeChatPermission' +import type { MobileChatPermission } from './mobile-native-chat-permission' +import { MobileNativeChatQuestion } from './MobileNativeChatQuestion' +import { mobileChatQuestionKey, type MobileChatQuestion } from './mobile-native-chat-question' + +/** The one pending agent prompt shown above the composer: a structured + * AskUserQuestion wins, then a heuristic permission, then a heuristic question. + * The controller owns dismissal (it must survive this subtree unmounting on a + * view toggle); `ask` arrives already nulled while dismissed. */ +export function MobileNativeChatPromptCard({ + ask, + askKey, + onDismissAsk, + onAnswerAsk, + onCancelAsk, + permission, + onRespondPermission, + question, + onAnswerQuestion +}: { + ask?: AskPrompt | null + askKey?: string | null + onDismissAsk?: () => void + onAnswerAsk?: (prompt: AskPrompt, selections: AskAnswerSelection[]) => Promise + onCancelAsk?: () => Promise + permission?: MobileChatPermission | null + onRespondPermission?: (send: string) => Promise + question?: MobileChatQuestion | null + onAnswerQuestion?: (text: string) => Promise +}): React.JSX.Element | null { + if (ask) { + return ( + { + const accepted = (await onAnswerAsk?.(ask, selections)) ?? false + if (accepted) { + onDismissAsk?.() + } + return accepted + }} + onCancel={async () => { + const accepted = (await onCancelAsk?.()) ?? false + if (accepted) { + onDismissAsk?.() + } + return accepted + }} + /> + ) + } + if (permission) { + return ( + (await onRespondPermission?.(send)) ?? false} + /> + ) + } + if (question) { + return ( + (await onAnswerQuestion?.(text)) ?? false} + /> + ) + } + return null +} diff --git a/mobile/src/session/MobileNativeChatSessionOptionPickers.test.ts b/mobile/src/session/MobileNativeChatSessionOptionPickers.test.ts index 4430c60115a..5a959b870bf 100644 --- a/mobile/src/session/MobileNativeChatSessionOptionPickers.test.ts +++ b/mobile/src/session/MobileNativeChatSessionOptionPickers.test.ts @@ -41,6 +41,7 @@ const MODEL_DESCRIPTOR: SessionOptionDescriptor = { ] }, valueSource: 'reported', + transport: 'catalog', settable: true } @@ -57,6 +58,7 @@ const EFFORT_DESCRIPTOR: SessionOptionDescriptor = { ] }, valueSource: 'dispatched', + transport: 'catalog', settable: true } @@ -66,6 +68,7 @@ const FAST_MODE_DESCRIPTOR: SessionOptionDescriptor = { category: 'mode', kind: { type: 'boolean', currentValue: false }, valueSource: 'reported', + transport: 'catalog', settable: true } @@ -227,6 +230,7 @@ describe('MobileNativeChatSessionOptionPickers', () => { ...MODEL_DESCRIPTOR, kind: { type: 'select', choices: [] }, valueSource: 'unknown', + transport: 'catalog', action: { type: 'agent-picker' } } ]) @@ -236,6 +240,46 @@ describe('MobileNativeChatSessionOptionPickers', () => { expect(invokeAction).toHaveBeenCalledWith('model') }) + // The terminal transport can only learn the outcome by parsing the screen back, + // so the sheet admits the value is unconfirmed; the structured transport reports + // it every turn, which makes the same caption noise there. + it.each([ + { transport: 'catalog' as const, caption: true }, + { transport: 'agent-session' as const, caption: false } + ])('captions a dispatched value only on the terminal transport', async (scenario) => { + mount([ + MODEL_DESCRIPTOR, + { ...EFFORT_DESCRIPTOR, valueSource: 'dispatched', transport: scenario.transport } + ]) + await act(async () => pill('Model').props.onPress()) + await act(async () => rowByText('Effort').props.onPress()) + const captions = renderer!.root + .findAll((node) => node.type === 'Text') + .filter( + (node) => + (node.props as { children?: unknown }).children === 'Sent to the agent — not confirmed' + ) + expect(captions.length > 0).toBe(scenario.caption) + }) + + it.each(['catalog', 'agent-session'] as const)( + 'does not caption a reported value on the %s transport', + async (transport) => { + mount([MODEL_DESCRIPTOR, { ...EFFORT_DESCRIPTOR, valueSource: 'reported', transport }]) + await act(async () => pill('Model').props.onPress()) + await act(async () => rowByText('Effort').props.onPress()) + expect( + renderer!.root + .findAll((node) => node.type === 'Text') + .some( + (node) => + (node.props as { children?: unknown }).children === + 'Sent to the agent — not confirmed' + ) + ).toBe(false) + } + ) + it('locks the pills while the agent is working', () => { mount([MODEL_DESCRIPTOR, EFFORT_DESCRIPTOR], true) expect(pill('Model').props).toMatchObject({ disabled: true }) diff --git a/mobile/src/session/MobileNativeChatSessionOptionPickers.tsx b/mobile/src/session/MobileNativeChatSessionOptionPickers.tsx index bfa5244a398..c9d641f74ec 100644 --- a/mobile/src/session/MobileNativeChatSessionOptionPickers.tsx +++ b/mobile/src/session/MobileNativeChatSessionOptionPickers.tsx @@ -3,9 +3,10 @@ import { ActivityIndicator, Keyboard, Pressable, StyleSheet, Text, View } from ' import { ChevronLeft, X } from 'lucide-react-native' import { BottomDrawer } from '../components/BottomDrawer' import { colors, radii, spacing, typography } from '../theme/mobile-theme' -import type { - SessionOptionDescriptor, - SessionOptionValue +import { + sessionOptionDispatchUnconfirmed, + type SessionOptionDescriptor, + type SessionOptionValue } from '../../../src/shared/native-chat-session-options' import { mobileModelPillLabel, @@ -119,7 +120,7 @@ export function MobileNativeChatSessionOptionPickers({ ) : null} - {activeDescriptor.valueSource === 'dispatched' ? ( + {sessionOptionDispatchUnconfirmed(activeDescriptor) ? ( Sent to the agent — not confirmed ) : null} {reason ? {reason} : null} diff --git a/mobile/src/session/MobileNativeChatToolRun.tsx b/mobile/src/session/MobileNativeChatToolRun.tsx new file mode 100644 index 00000000000..db3ddbea063 --- /dev/null +++ b/mobile/src/session/MobileNativeChatToolRun.tsx @@ -0,0 +1,250 @@ +import { useEffect, useRef, useState } from 'react' +import { Animated, Pressable, Text, View } from 'react-native' +import { ChevronDown, SquareChevronRight, SquareTerminal, Wrench } from 'lucide-react-native' +import { diffFromText, diffFromToolCall } from '../../../src/shared/native-chat-diff' +import type { NativeChatDiffLine as DiffLine } from '../../../src/shared/native-chat-diff' +import { pairToolBlocks } from '../../../src/shared/native-chat-tool-fold' +import type { NativeChatToolPair as ToolPair } from '../../../src/shared/native-chat-tool-fold' +import { + createToolInputDisplay, + summarizeToolRun, + truncateToolDetail +} from '../../../src/shared/native-chat-tool-summary' +import { + describeActiveToolCall, + formatActiveToolLabel, + formatToolCallCount, + isCommandToolName, + selectActiveToolCall +} from '../../../src/shared/native-chat-tool-activity' +import type { NativeChatBlock } from '../../../src/shared/native-chat-types' +import { colors } from '../theme/mobile-theme' +import { styles } from './mobile-native-chat-message-styles' + +const MAX_VISIBLE_TOOL_PAIRS = 6 +const MAX_TOOL_RUN_DIFF_ROWS = 240 + +function DiffView({ lines }: { lines: DiffLine[] }): React.JSX.Element { + return ( + + {lines.map((line, i) => ( + + {line.kind === 'add' ? '+' : line.kind === 'del' ? '-' : ' '} + {line.text} + + ))} + + ) +} + +/** A single inline tool line — `▸ ToolName preview` — that expands in place to + * show the call's diff/input or the result's body. Mirrors the reference design + * where tool calls read as flat lines in the conversation, not boxed blocks. */ +function ResultBody({ + output, + isError, + diff +}: { + output: string + isError?: boolean + diff: DiffLine[] | null +}): React.JSX.Element { + if (diff) { + return + } + return ( + + {truncateToolDetail(output)} + + ) +} + +/** One request: a tool call and its result rendered together as a single + * expandable line. `defaultExpanded` lets the group toggle open every line. */ +function ToolLine({ + pair, + defaultExpanded, + diffLineLimit, + onOpenFile +}: { + pair: ToolPair + defaultExpanded: boolean + diffLineLimit: number + onOpenFile?: (relativePath: string) => void +}): React.JSX.Element { + const [expanded, setExpanded] = useState(defaultExpanded) + const { call, result } = pair + const name = call ? call.name : 'Result' + const inputDisplay = call ? createToolInputDisplay(call.input) : null + const preview = inputDisplay?.label ?? result?.output.split('\n')[0]?.slice(0, 80) ?? '' + // Why: collapsed tool rows are the common path; defer bounded diff parsing + // and detail formatting until the user asks to reveal the detail. + const callDiff = expanded && call ? diffFromToolCall(call.name, call.input, diffLineLimit) : null + const resultDiff = expanded && result ? diffFromText(result.output, diffLineLimit) : null + const callDetail = expanded && inputDisplay && !callDiff ? inputDisplay.formatDetail() : undefined + const hasDetail = callDiff !== null || result !== undefined || inputDisplay?.hasDetail === true + // The group toggle opens every line at once, bypassing the tap guard, so the + // panel has to consult it too — else a detail-less row echoes its own label + // under itself and no tap can dismiss it. + const showDetail = hasDetail && expanded + // A tool that targets a file (Read/Edit/Write…) renders its preview as a + // tappable link that opens the file, independent of the line's expand tap. + const filePath = inputDisplay?.filePath ?? null + const openable = filePath !== null && onOpenFile !== undefined + return ( + + hasDetail && setExpanded((v) => !v)} + hitSlop={6} + > + {showDetail ? ( + + ) : ( + + )} + {name} + {preview ? ( + onOpenFile!(filePath!) : undefined} + suppressHighlighting={!openable} + > + {preview} + + ) : null} + + {showDetail ? ( + + {callDiff ? : null} + {callDetail ? {callDetail} : null} + {result ? ( + + ) : null} + + ) : null} + + ) +} + +/** Breathing label for a still-running tool, matching desktop's `animate-pulse`. */ +function PulsingText({ + style, + numberOfLines, + children +}: { + style?: React.ComponentProps['style'] + numberOfLines?: number + children: React.ReactNode +}): React.JSX.Element { + const pulse = useRef(new Animated.Value(1)).current + useEffect(() => { + const animation = Animated.loop( + Animated.sequence([ + Animated.timing(pulse, { toValue: 0.45, duration: 700, useNativeDriver: true }), + Animated.timing(pulse, { toValue: 1, duration: 700, useNativeDriver: true }) + ]) + ) + animation.start() + return () => animation.stop() + }, [pulse]) + return ( + + {children} + + ) +} + +/** A run of a message's tool calls/results, collapsed to a one-line summary that + * expands to the individual inline tool lines. `defaultExpanded` lets the global + * toolbar toggle drive every run at once while still allowing per-run override. */ +export function ToolRun({ + blocks, + defaultExpanded, + expandChildren, + activeCall, + trailing, + onOpenFile +}: { + blocks: NativeChatBlock[] + defaultExpanded: boolean + /** Child tool lines stay collapsed when the turn caret drove the run open. */ + expandChildren: boolean + /** The still-running call, when the turn is live (desktop parity). */ + activeCall: ReturnType + trailing?: React.ReactNode + onOpenFile?: (relativePath: string) => void +}): React.JSX.Element { + const [open, setOpen] = useState(defaultExpanded) + const pairs = pairToolBlocks(blocks, MAX_VISIBLE_TOOL_PAIRS) + const diffLineLimit = Math.max(1, Math.floor(MAX_TOOL_RUN_DIFF_ROWS / (pairs.length * 2 || 1))) + let callCount = 0 + for (const block of blocks) { + if (block.type === 'tool-call') { + callCount++ + } + } + callCount ||= pairs.length + const summary = summarizeToolRun(blocks) + const ActiveToolIcon = activeCall && isCommandToolName(activeCall.name) ? SquareTerminal : Wrench + return ( + + + {activeCall ? ( + setOpen((v) => !v)} + hitSlop={6} + accessibilityRole="button" + accessibilityState={{ expanded: open }} + accessibilityLiveRegion="polite" + > + + + {formatActiveToolLabel(describeActiveToolCall(activeCall))} + + {open ? : null} + + ) : ( + setOpen((v) => !v)} hitSlop={6}> + {open ? ( + + ) : ( + + )} + {callCount}× + + {summary || formatToolCallCount(callCount)} + + + )} + {trailing} + + {open ? ( + + {pairs.map((pair, i) => ( + + ))} + {callCount > pairs.length ? ( + … {callCount - pairs.length} more tool calls + ) : null} + + ) : null} + + ) +} diff --git a/mobile/src/session/MobileNativeChatTurnStatus.test.ts b/mobile/src/session/MobileNativeChatTurnStatus.test.ts new file mode 100644 index 00000000000..78ac01e0d37 --- /dev/null +++ b/mobile/src/session/MobileNativeChatTurnStatus.test.ts @@ -0,0 +1,112 @@ +import { createElement } from 'react' +import { act, create, type ReactTestInstance, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +vi.mock('react-native', async () => { + const React = await import('react') + const Text = ({ children, ...props }: { children?: unknown }): unknown => + React.createElement('Text', props, children) + return { + Animated: { + Text, + Value: class { + constructor(private value: number) {} + setValue(next: number): void { + this.value = next + } + }, + loop: (animation: unknown) => animation, + sequence: () => ({ start: vi.fn(), stop: vi.fn() }), + timing: () => ({ start: vi.fn(), stop: vi.fn() }) + }, + Pressable: ({ children, ...props }: { children?: unknown }) => + React.createElement('Pressable', props, children), + Text, + View: ({ children, ...props }: { children?: unknown }) => + React.createElement('View', props, children), + StyleSheet: { create: (styles: unknown) => styles, hairlineWidth: 1 } + } +}) +vi.mock('lucide-react-native', () => ({ ChevronRight: 'ChevronRight' })) + +import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus' + +describe('MobileNativeChatTurnStatus', () => { + let renderer: ReactTestRenderer | null = null + + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(new Date('2026-09-04T00:00:00Z')) + }) + + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + vi.useRealTimers() + }) + + function render(props: { + startedAt: number | null + thinking: boolean + workedSeconds?: number | null + expanded?: boolean + onToggleExpanded?: () => void + }): ReactTestRenderer { + act(() => { + renderer = create(createElement(MobileNativeChatTurnStatus, props)) + }) + return renderer! + } + + const labels = (node: ReactTestInstance): string[] => + node.findAllByType('Text' as never).map((text) => String(text.children.join(''))) + + it('reads "Thinking" before the turn produces output', () => { + const tree = render({ startedAt: Date.now(), thinking: true }) + expect(labels(tree.root)).toEqual(['Thinking']) + }) + + it('counts up once the turn is producing output', () => { + const startedAt = Date.now() + const tree = render({ startedAt, thinking: false }) + expect(labels(tree.root)).toEqual(['Working for 0s']) + act(() => { + vi.advanceTimersByTime(12_000) + }) + expect(labels(tree.root)).toEqual(['Working for 12s']) + }) + + it('settles to a tappable "Worked for" row that toggles the turn', () => { + const onToggleExpanded = vi.fn() + const tree = render({ + startedAt: Date.now(), + thinking: false, + workedSeconds: 184, + onToggleExpanded + }) + expect(labels(tree.root)).toEqual(['Worked for 3m 4s']) + const button = tree.root.findByType('Pressable' as never) + expect(button.props.accessibilityLabel).toBe('Toggle turn details') + expect(button.props.accessibilityState).toEqual({ expanded: false }) + act(() => button.props.onPress()) + expect(onToggleExpanded).toHaveBeenCalledOnce() + }) + + it('stays a plain row when the settled turn has nothing to disclose', () => { + const tree = render({ startedAt: Date.now(), thinking: false, workedSeconds: 5 }) + expect(tree.root.findAllByType('Pressable' as never)).toHaveLength(0) + expect(labels(tree.root)).toEqual(['Worked for 5s']) + }) + + it('holds no interval once the turn has settled', () => { + render({ startedAt: Date.now(), thinking: false, workedSeconds: 5 }) + expect(vi.getTimerCount()).toBe(0) + }) + + it('announces the live row to assistive tech', () => { + const tree = render({ startedAt: Date.now(), thinking: true }) + const row = tree.root.findByType('View' as never) + expect(row.props.accessibilityLiveRegion).toBe('polite') + expect(row.props.accessibilityLabel).toBe('Agent is responding') + }) +}) diff --git a/mobile/src/session/MobileNativeChatTurnStatus.tsx b/mobile/src/session/MobileNativeChatTurnStatus.tsx new file mode 100644 index 00000000000..4ce73cdcd38 --- /dev/null +++ b/mobile/src/session/MobileNativeChatTurnStatus.tsx @@ -0,0 +1,117 @@ +import { useEffect, useRef, useState } from 'react' +import { Animated, Pressable, StyleSheet, Text, View } from 'react-native' +import { ChevronRight } from 'lucide-react-native' +import { + formatNativeChatTurnStatusLabel, + NATIVE_CHAT_TURN_STATUS_COPY, + nativeChatElapsedSeconds +} from '../../../src/shared/native-chat-turn-status' +import { colors, spacing, typography } from '../theme/mobile-theme' + +/** Seconds tick only while a turn is actually counting, so a settled transcript + * holds no timers. */ +function useElapsedSeconds(startedAt: number | null, counting: boolean): number { + // Preserves the pre-stamp epoch for the frame before the turn's startedAt lands. + const [mountedAt] = useState(() => Date.now()) + const [now, setNow] = useState(() => Date.now()) + useEffect(() => { + if (!counting) { + return + } + setNow(Date.now()) + const timer = setInterval(() => setNow(Date.now()), 1_000) + return () => clearInterval(timer) + }, [counting]) + return counting ? nativeChatElapsedSeconds(startedAt, mountedAt, now) : 0 +} + +/** The per-turn status row — "Thinking", then "Working for 12s" while the turn + * runs, settling to a tappable "Worked for 3m 4s" that discloses the turn's + * tool activity. Desktop parity: `NativeChatWorkingStatus`. */ +export function MobileNativeChatTurnStatus({ + startedAt, + thinking, + workedSeconds, + expanded = false, + onToggleExpanded +}: { + startedAt: number | null + thinking: boolean + workedSeconds?: number | null + expanded?: boolean + onToggleExpanded?: () => void +}): React.JSX.Element { + const counting = !thinking && workedSeconds == null + const elapsedSeconds = useElapsedSeconds(startedAt, counting) + const label = formatNativeChatTurnStatusLabel({ thinking, workedSeconds, elapsedSeconds }) + + const pulse = useRef(new Animated.Value(1)).current + useEffect(() => { + if (!thinking) { + pulse.setValue(1) + return + } + const animation = Animated.loop( + Animated.sequence([ + Animated.timing(pulse, { toValue: 0.45, duration: 700, useNativeDriver: true }), + Animated.timing(pulse, { toValue: 1, duration: 700, useNativeDriver: true }) + ]) + ) + animation.start() + return () => animation.stop() + }, [pulse, thinking]) + + const rowStyle = [styles.row, thinking ? null : styles.rowSettled] + + if (workedSeconds != null && onToggleExpanded) { + return ( + [...rowStyle, pressed && styles.pressed]} + onPress={onToggleExpanded} + hitSlop={6} + accessibilityRole="button" + accessibilityState={{ expanded }} + accessibilityLabel={NATIVE_CHAT_TURN_STATUS_COPY.toggleDetails} + > + {label} + + + + + ) + } + + return ( + + {label} + + ) +} + +const styles = StyleSheet.create({ + row: { + flexDirection: 'row', + alignItems: 'center', + gap: spacing.xs, + minHeight: 28, + paddingHorizontal: spacing.md + }, + rowSettled: { + borderBottomWidth: StyleSheet.hairlineWidth, + borderBottomColor: colors.borderSubtle + }, + pressed: { + opacity: 0.6 + }, + label: { + color: colors.textMuted, + fontSize: typography.bodySize + }, + caretOpen: { + transform: [{ rotate: '90deg' }] + } +}) diff --git a/mobile/src/session/MobileNativeChatView.test.ts b/mobile/src/session/MobileNativeChatView.test.ts index 9d171e3ac8a..d101c3f0ef6 100644 --- a/mobile/src/session/MobileNativeChatView.test.ts +++ b/mobile/src/session/MobileNativeChatView.test.ts @@ -72,6 +72,9 @@ type Overrides = { inputLockReason?: 'disconnected' | 'waiting' | null onSend?: (text: string) => Promise pending?: Parameters[0]['pending'] + structuredActivityUi?: boolean + agentWorking?: boolean + sendSurfaceId?: string } function assistantTurn(id: string, text: string): NativeChatMessage { @@ -243,4 +246,111 @@ describe('MobileNativeChatView', () => { vi.useRealTimers() } }) + + describe('structured turn status wiring', () => { + const userTurn = (id: string, text: string): NativeChatMessage => ({ + id, + role: 'user', + blocks: [{ type: 'text', text }], + timestamp: 0, + source: 'transcript' + }) + + function rowProps(id: string): Record { + return (renderedRow(id) as { props: Record }).props + } + + function workingIndicators(): ReactTestInstance[] { + return renderer!.root.findAll((node) => node.type === 'WorkingIndicator') + } + + it('gives the live user turn a status row and drops the three-dot indicator', async () => { + const folded = [userTurn('u1', 'go')] + await render({ messages: folded, folded, structuredActivityUi: true, agentWorking: true }) + const props = rowProps('u1') + expect(props.structuredActivityUi).toBe(true) + expect(props.turnStatus).toMatchObject({ thinking: true, workedSeconds: null }) + expect(props.activeTurnIsWorking).toBe(true) + expect(workingIndicators()).toHaveLength(0) + }) + + it('keeps the bridge lane on the three-dot indicator with no turn status', async () => { + const folded = [userTurn('u1', 'go')] + await render({ messages: folded, folded, agentWorking: true }) + const props = rowProps('u1') + expect(props.structuredActivityUi).toBe(false) + expect(props.turnStatus).toBeNull() + expect(props.activeTurnIsWorking).toBe(false) + expect(workingIndicators()).toHaveLength(1) + }) + + it('settles the finished turn to a tappable duration', async () => { + const folded = [userTurn('u1', 'go'), assistantTurn('a1', 'done')] + await render({ messages: folded, folded, structuredActivityUi: true, agentWorking: true }) + expect(rowProps('u1').turnStatus).toMatchObject({ thinking: false, workedSeconds: null }) + await update({ messages: folded, folded, structuredActivityUi: true, agentWorking: false }) + const settled = rowProps('u1') + expect(settled.turnStatus).toMatchObject({ thinking: false }) + expect((settled.turnStatus as { workedSeconds: number | null }).workedSeconds).toBeTypeOf( + 'number' + ) + expect(settled.onToggleTurn).toBeTypeOf('function') + expect(settled.activeTurnIsWorking).toBe(false) + }) + + it('hangs no status row on an assistant row', async () => { + const folded = [userTurn('u1', 'go'), assistantTurn('a1', 'done')] + await render({ messages: folded, folded, structuredActivityUi: true, agentWorking: true }) + expect(rowProps('a1').turnStatus).toBeNull() + // The assistant row still belongs to the live turn, so its tool row stays visible. + expect(rowProps('a1').activeTurnIsWorking).toBe(true) + }) + + it('does not carry a running turn clock across chat surfaces', async () => { + vi.useFakeTimers() + try { + vi.setSystemTime(1_000) + const firstTab = [userTurn('u1', 'first')] + await render({ + messages: firstTab, + folded: firstTab, + structuredActivityUi: true, + agentWorking: true, + sendSurfaceId: 'host\0worktree\0tab-a' + }) + expect(rowProps('u1').turnStatus).toMatchObject({ startedAt: 1_000 }) + + vi.setSystemTime(12_000) + const secondTab = [userTurn('u2', 'second')] + await update({ + messages: secondTab, + folded: secondTab, + structuredActivityUi: true, + agentWorking: true, + sendSurfaceId: 'host\0worktree\0tab-b' + }) + + expect(rowProps('u2').turnStatus).toMatchObject({ startedAt: 12_000 }) + } finally { + vi.useRealTimers() + } + }) + + it('does not treat pre-user history as part of the live turn', async () => { + const history = [ + assistantTurn('a0', 'before the first prompt'), + userTurn('u1', 'go'), + assistantTurn('a1', 'working') + ] + await render({ + messages: history, + folded: history, + structuredActivityUi: true, + agentWorking: true + }) + + expect(rowProps('a0').activeTurnIsWorking).toBe(false) + expect(rowProps('a1').activeTurnIsWorking).toBe(true) + }) + }) }) diff --git a/mobile/src/session/MobileNativeChatView.tsx b/mobile/src/session/MobileNativeChatView.tsx index 4db4e437b43..59c024435b6 100644 --- a/mobile/src/session/MobileNativeChatView.tsx +++ b/mobile/src/session/MobileNativeChatView.tsx @@ -21,16 +21,16 @@ import { type MobileNativeChatPendingItem } from './mobile-native-chat-render-data' import { useMobileNativeChatPinchGesture } from './use-mobile-native-chat-pinch-gesture' +import { useMobileNativeChatTurnDisclosure } from './use-mobile-native-chat-turn-disclosure' +import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus' import { MobileAgentWorkingIndicator } from './MobileAgentWorkingIndicator' import type { PendingNativeChatImage } from './mobile-native-chat-image-attachment' import { MobileNativeChatComposer } from './MobileNativeChatComposer' +import { MobileNativeChatPromptCard } from './MobileNativeChatPromptCard' +import type { MobileChatPermission } from './mobile-native-chat-permission' +import type { MobileChatQuestion } from './mobile-native-chat-question' import type { MobileNativeChatSessionOptionPickersProps } from './MobileNativeChatSessionOptionPickers' import { MobileNativeChatMessage } from './MobileNativeChatMessage' -import { MobileNativeChatAsk } from './MobileNativeChatAsk' -import { MobileNativeChatPermission } from './MobileNativeChatPermission' -import type { MobileChatPermission } from './mobile-native-chat-permission' -import { MobileNativeChatQuestion } from './MobileNativeChatQuestion' -import { mobileChatQuestionKey, type MobileChatQuestion } from './mobile-native-chat-question' import type { MobileNativeChatStatus } from './use-mobile-native-chat-session' const INPUT_LOCK_SETTLE_MS = 600 @@ -49,6 +49,9 @@ type Props = { /** Resolved agent for this chat; names the empty-state copy (desktop parity). */ agent?: string | null agentWorking?: boolean + /** Structured lane: per-turn "Working for N" status plus live tool progress, + * replacing the bridge lane's static three-dot working row (desktop parity). */ + structuredActivityUi?: boolean /** Interrupt the agent mid-turn (shown as a Stop button on the working bar). */ onStop?: () => void /** Live partial assistant text to show as an in-progress bubble, already gated @@ -126,6 +129,7 @@ export function MobileNativeChatView({ error, agent, agentWorking, + structuredActivityUi = false, onStop, streaming, hasMore, @@ -252,6 +256,15 @@ export function MobileNativeChatView({ listRef.current?.scrollToIndex({ index, viewPosition: 0, animated: true }) }, []) + // Per-turn "Thinking / Working for N / Worked for N" rows. The structured lane + // owns them; the bridge lane keeps its three-dot indicator. + const turns = useMobileNativeChatTurnDisclosure({ + messages: data, + enabled: structuredActivityUi, + isWorking: agentWorking === true, + scopeKey: sendSurfaceId + }) + const renderItem = useCallback( ({ item, index }: { item: NativeChatMessage; index: number }) => ( ), - [toolsExpanded, fontScale, onScrollToMessage, onOpenFile] + [toolsExpanded, fontScale, onScrollToMessage, onOpenFile, structuredActivityUi, turns] ) const emptyState = mobileNativeChatEmptyState(status, agent ?? null, error) @@ -337,6 +353,15 @@ export function MobileNativeChatView({ ) : null } + ListFooterComponent={ + turns.activeTurnIsUnanchored && turns.active ? ( + + ) : null + } ListEmptyComponent={ emptyState ? ( @@ -360,47 +385,22 @@ export function MobileNativeChatView({ ) : null} )} - {/* Pending agent prompt: a structured AskUserQuestion wins, then a - heuristic permission, then a heuristic question. The controller owns - dismissal (it must survive this subtree unmounting on a view toggle); - `ask` arrives already nulled while dismissed. */} - {ask ? ( - { - const accepted = (await onAnswerAsk?.(ask, selections)) ?? false - if (accepted) { - onDismissAsk?.() - } - return accepted - }} - onCancel={async () => { - const accepted = (await onCancelAsk?.()) ?? false - if (accepted) { - onDismissAsk?.() - } - return accepted - }} - /> - ) : permission ? ( - (await onRespondPermission?.(send)) ?? false} - /> - ) : question ? ( - (await onAnswerQuestion?.(text)) ?? false} - /> - ) : null} + {/* Chrome row above the composer: the working indicator and the global tool-calls expand/collapse toggle on the left, Stop in the far corner. */} - {agentWorking ? : null} + {agentWorking && !structuredActivityUi ? : null} [styles.chromeToggle, pressed && styles.pressed]} onPress={() => setToolsExpanded((v) => !v)} diff --git a/mobile/src/session/mobile-native-chat-controller-contract.ts b/mobile/src/session/mobile-native-chat-controller-contract.ts index 890a3a1562e..53187e0d6db 100644 --- a/mobile/src/session/mobile-native-chat-controller-contract.ts +++ b/mobile/src/session/mobile-native-chat-controller-contract.ts @@ -25,6 +25,8 @@ export type MobileNativeChatController = { chatPending: MobileNativeChatPendingMessage[] chatImagePreviewsByMessageId: Record nativeChatSession: ReturnType + /** Structured lane: drives the per-turn status row and live tool progress. */ + nativeChatStructured: boolean nativeChatAgentWorking: boolean nativeChatStreamingText?: string /** Agent mid-turn, regardless of whether chat is the visible view. */ diff --git a/mobile/src/session/mobile-native-chat-message-styles.ts b/mobile/src/session/mobile-native-chat-message-styles.ts index ad7cf4b4009..7ae1128445a 100644 --- a/mobile/src/session/mobile-native-chat-message-styles.ts +++ b/mobile/src/session/mobile-native-chat-message-styles.ts @@ -80,6 +80,18 @@ export const styles = StyleSheet.create({ fontFamily: typography.monoFamily, fontSize: MONO_SIZE }, + toolRunActive: { + flex: 1, + flexDirection: 'row', + alignItems: 'center', + gap: spacing.sm, + paddingVertical: 3 + }, + toolRunActiveLabel: { + flex: 1, + color: colors.textSecondary, + fontSize: typography.bodySize + }, toolRunBody: { paddingLeft: spacing.sm, borderLeftWidth: 2, diff --git a/mobile/src/session/mobile-structured-agent-session-launch.test.ts b/mobile/src/session/mobile-structured-agent-session-launch.test.ts index 54f9b5cbe88..f575d5ac6d4 100644 --- a/mobile/src/session/mobile-structured-agent-session-launch.test.ts +++ b/mobile/src/session/mobile-structured-agent-session-launch.test.ts @@ -139,4 +139,76 @@ describe('mobile structured Codex launch', () => { kind: 'unknown' }) }) + + it.each(['structured_agent_session_unsupported', 'method_not_found'])( + 'treats a top-level %s as a definitive refusal', + async (code) => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { ok: false, error: { code, message: 'structured create unavailable' } } + ) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({ + kind: 'failed', + message: 'structured create unavailable' + }) + } + ) + + it.each(['agent_session_operation_unknown', 'runtime_error', 'future_unknown_code'])( + 'keeps a top-level %s outcome unknown', + async (code) => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { ok: false, error: { code, message: 'create outcome ambiguous' } } + ) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({ + kind: 'unknown', + message: 'create outcome ambiguous' + }) + } + ) + + it('treats an envelope unsupported refusal as definitive', async () => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { + ok: true, + result: { + ok: false, + refusal: { + code: 'structured_agent_session_unsupported', + message: 'structured create unavailable' + } + } + } + ) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({ + kind: 'failed', + message: 'structured create unavailable' + }) + }) + + it.each(['agent_session_operation_unknown', 'agent_session_ownership_unknown', 'future_code'])( + 'keeps an envelope %s refusal unknown', + async (code) => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { + ok: true, + result: { + ok: false, + refusal: { code, message: 'create outcome ambiguous' } + } + } + ) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({ + kind: 'unknown', + message: 'create outcome ambiguous' + }) + } + ) }) diff --git a/mobile/src/session/mobile-structured-agent-session-launch.ts b/mobile/src/session/mobile-structured-agent-session-launch.ts index ecad0410dfd..b7eb8289e84 100644 --- a/mobile/src/session/mobile-structured-agent-session-launch.ts +++ b/mobile/src/session/mobile-structured-agent-session-launch.ts @@ -2,6 +2,7 @@ import type { AgentSessionAttachResult, AgentSessionMutationResult } from '../../../src/shared/agent-session-wire' +import { isDefinitiveAgentSessionCreateRefusal } from '../../../src/shared/agent-session-definitive-refusal' import { structuredAgentSessionPayloadFingerprint } from '../../../src/shared/structured-agent-session-mutation' import type { RpcClient } from '../transport/rpc-client' import { structuredSessionOperationId } from './mobile-structured-agent-session-rpc' @@ -66,6 +67,13 @@ function unknownCreateResult(error: unknown): MobileStructuredCodexLaunchResult } } +function classifyCreateRefusal(code: string, message: string): MobileStructuredCodexLaunchResult { + if (!isDefinitiveAgentSessionCreateRefusal(code)) { + return unknownCreateResult(new Error(message)) + } + return { kind: 'failed', message: message || 'Could not open Codex chat.' } +} + export async function createMobileStructuredCodexSession( client: RpcClient, worktreeId: string @@ -125,10 +133,7 @@ export async function createMobileStructuredCodexSession( ) { return unknownCreateResult(new Error('The Codex chat result could not be confirmed.')) } - if (response.error.code === 'agent_session_operation_unknown') { - return unknownCreateResult(new Error(response.error.message)) - } - return { kind: 'failed', message: response.error.message || 'Could not open Codex chat.' } + return classifyCreateRefusal(response.error.code, response.error.message) } const result = response.result as AgentSessionMutationResult if (!result || typeof result !== 'object' || typeof result.ok !== 'boolean') { @@ -142,10 +147,7 @@ export async function createMobileStructuredCodexSession( ) { return unknownCreateResult(new Error('The Codex chat result could not be confirmed.')) } - if (result.refusal.code === 'agent_session_operation_unknown') { - return unknownCreateResult(new Error(result.refusal.message)) - } - return { kind: 'failed', message: result.refusal.message || 'Could not open Codex chat.' } + return classifyCreateRefusal(result.refusal.code, result.refusal.message) } if ( !result.value || diff --git a/mobile/src/session/use-mobile-native-chat-controller.ts b/mobile/src/session/use-mobile-native-chat-controller.ts index bf944398e87..a946956f8d6 100644 --- a/mobile/src/session/use-mobile-native-chat-controller.ts +++ b/mobile/src/session/use-mobile-native-chat-controller.ts @@ -10,9 +10,8 @@ import { useMobileNativeChatDrafts } from './use-mobile-native-chat-drafts' import { useMobileNativeChatFileSearch } from './use-mobile-native-chat-file-search' import { useMobileNativeChatMessageSend } from './use-mobile-native-chat-message-send' import { mobileNativeChatStreamPreview } from './mobile-native-chat-streaming-gate' -import { useMobileNativeChatSession } from './use-mobile-native-chat-session' import { useMobileNativeChatSessionOptionController } from './use-mobile-native-chat-session-option-controller' -import { useMobileStructuredAgentSession } from './use-mobile-structured-agent-session' +import { useMobileNativeChatSessionLane } from './use-mobile-native-chat-session-lane' import { useMobileStructuredNativeChatSendBridge } from './use-mobile-structured-native-chat-send-bridge' import { useMobileNativeChatPrompts } from './use-mobile-native-chat-prompts' import { useMobileNativeChatStop } from './use-mobile-native-chat-stop' @@ -82,27 +81,19 @@ export function useMobileNativeChatController(args: { nativeChatTranscriptIsLocalReadable }) - const legacyNativeChatSession = useMobileNativeChatSession({ - client, - sourceIdentity, - agent: activeChatStructured ? null : (activeChatResolution?.agent ?? null), - sessionId: activeChatStructured ? null : activeChatSessionId, - transcriptPath: activeChatStructured ? null : (activeChatResolution?.transcriptPath ?? null) - }) - const structuredNativeChat = useMobileStructuredAgentSession({ - client, - sessionId: activeChatStructured ? activeChatSessionId : null, - sourceIdentity, - enabled: showNativeChat, - // Holds are connection-scoped; dropping this on transport loss lets the hook - // reacquire the provider without clearing the cached transcript. - connected: connState === 'connected', - agent: activeChatStructured ? activeChatAgent : null, - onSendError - }) - const nativeChatSession = activeChatStructured - ? structuredNativeChat.session - : legacyNativeChatSession + const { structuredSession: structuredNativeChat, session: nativeChatSession } = + useMobileNativeChatSessionLane({ + client, + structured: activeChatStructured, + agent: activeChatAgent, + resolvedAgent: activeChatResolution?.agent ?? null, + transcriptPath: activeChatResolution?.transcriptPath ?? null, + sessionId: activeChatSessionId, + sourceIdentity, + enabled: showNativeChat, + connState, + onSendError + }) const { composerText: chatComposerText, setComposerText: setChatComposerText, @@ -303,6 +294,8 @@ export function useMobileNativeChatController(args: { chatPending, chatImagePreviewsByMessageId, nativeChatSession, + /** Structured lane: drives the per-turn status row and live tool progress. */ + nativeChatStructured: activeChatStructured, nativeChatAgentWorking, nativeChatStreamingText, nativeChatStreamLive, diff --git a/mobile/src/session/use-mobile-native-chat-session-lane.ts b/mobile/src/session/use-mobile-native-chat-session-lane.ts new file mode 100644 index 00000000000..fc465de902b --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-session-lane.ts @@ -0,0 +1,59 @@ +import type { RpcClient } from '../transport/rpc-client' +import type { ConnectionState } from '../transport/types' +import { useMobileNativeChatSession } from './use-mobile-native-chat-session' +import { useMobileStructuredAgentSession } from './use-mobile-structured-agent-session' + +/** Mounts both transcript sources and hands back the one this tab's lane owns. + * Both hooks always run (hook order is fixed); the inactive lane is starved of + * its identity inputs rather than unmounted, so a lane flip keeps its cache. */ +export function useMobileNativeChatSessionLane({ + client, + structured, + agent, + resolvedAgent, + transcriptPath, + sessionId, + sourceIdentity, + enabled, + connState, + onSendError +}: { + client: RpcClient | null + structured: boolean + /** Agent id for the structured provider session. */ + agent: string | null + /** Agent resolved from the terminal, for the bridge transcript reader. */ + resolvedAgent: string | null + transcriptPath: string | null + sessionId: string | null + sourceIdentity: Parameters[0]['sourceIdentity'] + enabled: boolean + connState: ConnectionState + onSendError: (message: string) => void +}): { + structuredSession: ReturnType + session: ReturnType +} { + const bridgeSession = useMobileNativeChatSession({ + client, + sourceIdentity, + agent: structured ? null : resolvedAgent, + sessionId: structured ? null : sessionId, + transcriptPath: structured ? null : transcriptPath + }) + const structuredSession = useMobileStructuredAgentSession({ + client, + sessionId: structured ? sessionId : null, + sourceIdentity, + enabled, + // Holds are connection-scoped; dropping this on transport loss lets the hook + // reacquire the provider without clearing the cached transcript. + connected: connState === 'connected', + agent: structured ? agent : null, + onSendError + }) + return { + structuredSession, + session: structured ? structuredSession.session : bridgeSession + } +} diff --git a/mobile/src/session/use-mobile-native-chat-session-options.ts b/mobile/src/session/use-mobile-native-chat-session-options.ts index 66a929aeb56..6acdf8b3750 100644 --- a/mobile/src/session/use-mobile-native-chat-session-options.ts +++ b/mobile/src/session/use-mobile-native-chat-session-options.ts @@ -168,7 +168,8 @@ export function useMobileNativeChatSessionOptions(args: { models: activeModels(catalog, record), record, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) }, [agent, catalog, scopeKey, version]) diff --git a/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx b/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx new file mode 100644 index 00000000000..8467684ce16 --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx @@ -0,0 +1,153 @@ +import { createElement } from 'react' +import { act, create, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { NativeChatMessage } from '../../../src/shared/native-chat-types' +import { useMobileNativeChatTurnDisclosure } from './use-mobile-native-chat-turn-disclosure' + +function userMessage(id: string): NativeChatMessage { + return { + id, + role: 'user', + blocks: [{ type: 'text', text: id }], + timestamp: null, + source: 'transcript' + } +} + +function Harness({ + messages, + enabled, + isWorking = true, + scopeKey = 'host\0worktree\0tab-a' +}: { + messages: readonly NativeChatMessage[] + enabled: boolean + isWorking?: boolean + scopeKey?: string +}): React.JSX.Element { + const disclosure = useMobileNativeChatTurnDisclosure({ + messages, + enabled, + isWorking, + scopeKey + }) + return createElement('result', { disclosure }) +} + +describe('useMobileNativeChatTurnDisclosure', () => { + let renderer: ReactTestRenderer | null = null + + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + }) + + it('does not scan bridge-lane transcripts', () => { + const messages: NativeChatMessage[] = [ + { + id: 'u1', + role: 'user', + blocks: [{ type: 'text', text: 'go' }], + timestamp: null, + source: 'transcript' + } + ] + const findLastIndex = vi.spyOn(messages, 'findLastIndex') + const slice = vi.spyOn(messages, 'slice') + const filter = vi.spyOn(messages, 'filter') + const map = vi.spyOn(messages, 'map') + + act(() => { + renderer = create(createElement(Harness, { messages, enabled: false })) + }) + + expect(findLastIndex).not.toHaveBeenCalled() + expect(slice).not.toHaveBeenCalled() + expect(filter).not.toHaveBeenCalled() + expect(map).not.toHaveBeenCalled() + }) + + it('keeps a settled turn handler stable for NUL-delimited scope keys', () => { + vi.useFakeTimers() + try { + vi.setSystemTime(1_000) + const messages: NativeChatMessage[] = [ + { + id: 'u1', + role: 'user', + blocks: [{ type: 'text', text: 'go' }], + timestamp: null, + source: 'transcript' + } + ] + act(() => { + renderer = create(createElement(Harness, { messages, enabled: true })) + }) + vi.setSystemTime(6_000) + act(() => { + renderer?.update(createElement(Harness, { messages, enabled: true, isWorking: false })) + }) + const first = renderer!.root.findByType('result').props.disclosure.resolveRow(0, messages[0]) + + const refreshed = [...messages] + act(() => { + renderer?.update( + createElement(Harness, { messages: refreshed, enabled: true, isWorking: false }) + ) + }) + const second = renderer!.root + .findByType('result') + .props.disclosure.resolveRow(0, refreshed[0]) + + // The row carries the key; the handler itself lives on the hook and stays + // stable for the scope, so a re-render never disturbs a row's memo. + expect(first.turnKey).toBe('u1') + expect(second.turnKey).toBe('u1') + const firstHandler = renderer!.root.findByType('result').props.disclosure.onToggleTurn + expect(firstHandler).toBeTypeOf('function') + act(() => { + renderer?.update( + createElement(Harness, { messages: [...refreshed], enabled: true, isWorking: false }) + ) + }) + expect(renderer!.root.findByType('result').props.disclosure.onToggleTurn).toBe(firstHandler) + } finally { + vi.useRealTimers() + } + }) + + it('keeps at most the latest 128 turns expanded', () => { + vi.useFakeTimers() + try { + let messages: NativeChatMessage[] = [] + for (let index = 0; index < 129; index++) { + messages = messages.concat(userMessage(`u${index}`)) + vi.setSystemTime(index * 2_000) + act(() => { + if (renderer) { + renderer.update(createElement(Harness, { messages, enabled: true })) + } else { + renderer = create(createElement(Harness, { messages, enabled: true })) + } + }) + vi.setSystemTime(index * 2_000 + 1_000) + act(() => { + renderer?.update(createElement(Harness, { messages, enabled: true, isWorking: false })) + }) + const disclosureNow = renderer!.root.findByType('result').props.disclosure + const row = disclosureNow.resolveRow(index, messages[index]) + act(() => disclosureNow.onToggleTurn(row.turnKey)) + } + + const disclosure = renderer!.root.findByType('result').props.disclosure + const expanded = messages.filter( + (message, index) => disclosure.resolveRow(index, message).turnExpanded + ) + expect(expanded).toHaveLength(128) + expect(disclosure.resolveRow(0, messages[0]).turnExpanded).toBe(false) + expect(disclosure.resolveRow(128, messages[128]).turnExpanded).toBe(true) + } finally { + vi.useRealTimers() + } + }) +}) diff --git a/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts b/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts new file mode 100644 index 00000000000..46b58f29cba --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts @@ -0,0 +1,126 @@ +import { useCallback, useMemo, useState } from 'react' +import type { NativeChatMessage } from '../../../src/shared/native-chat-types' +import { + MOBILE_UNANCHORED_TURN_KEY, + useMobileNativeChatTurnStatus, + type NativeChatTurnStatus +} from './use-mobile-native-chat-turn-status' + +const EMPTY_TURN_IDS: ReadonlySet = new Set() +const EMPTY_TURN_KEYS: readonly undefined[] = [] +const MAX_EXPANDED_TURNS = 128 + +export type MobileNativeChatTurnRow = { + turnStatus: NativeChatTurnStatus | null + turnExpanded: boolean + /** Set only on a settled turn — the one row that has activity to disclose. */ + turnKey?: string + activeTurnIsWorking: boolean +} + +/** Owns the transcript's per-turn status rows and their disclosure state, and + * resolves what one list row needs. Bridge-lane chats pass `enabled: false` and + * keep their single three-dot working indicator instead. */ +export function useMobileNativeChatTurnDisclosure({ + messages, + enabled, + isWorking, + scopeKey +}: { + messages: readonly NativeChatMessage[] + enabled: boolean + isWorking: boolean + /** Host/worktree/tab identity for timing and disclosure isolation. */ + scopeKey: string +}): { + active: NativeChatTurnStatus | null + /** True when the live turn has no user message to hang its status row under. */ + activeTurnIsUnanchored: boolean + onToggleTurn: (turnKey: string) => void + resolveRow: (index: number, message: NativeChatMessage) => MobileNativeChatTurnRow +} { + const turnStatuses = useMobileNativeChatTurnStatus({ + messages, + enabled, + isWorking, + scopeKey + }) + const [expandedTurns, setExpandedTurns] = useState<{ + scopeKey: string + turnIds: ReadonlySet + }>(() => ({ scopeKey, turnIds: new Set() })) + const expandedTurnIds = + expandedTurns.scopeKey === scopeKey ? expandedTurns.turnIds : EMPTY_TURN_IDS + const toggleExpandedTurn = useCallback( + (turnKey: string) => { + setExpandedTurns((current) => { + const next = new Set(current.scopeKey === scopeKey ? current.turnIds : []) + if (!next.delete(turnKey)) { + if (next.size >= MAX_EXPANDED_TURNS) { + const oldest = next.values().next().value + if (oldest) { + next.delete(oldest) + } + } + next.add(turnKey) + } + return { scopeKey, turnIds: next } + }) + }, + [scopeKey] + ) + // Resolve each row's turn boundary once — a findLast per row is quadratic on a + // long transcript. + const turnKeys = useMemo(() => { + if (!enabled) { + return EMPTY_TURN_KEYS + } + let turnKey: string | undefined + return messages.map((message) => { + if (message.role === 'user') { + turnKey = message.id + } + return turnKey + }) + }, [enabled, messages]) + + const { active, activeTurnKey, completedByTurn } = turnStatuses + const resolveRow = useCallback( + (index: number, message: NativeChatMessage): MobileNativeChatTurnRow => { + const turnKey = turnKeys[index] + const turnStatus = + !enabled || message.role !== 'user' + ? null + : turnKey === activeTurnKey + ? active + : turnKey + ? (completedByTurn[turnKey] ?? null) + : null + return { + turnStatus, + turnExpanded: turnKey ? expandedTurnIds.has(turnKey) : false, + // Why: the key travels and the row calls one stable handler with it. A + // closure per row would be a new identity every render of a streaming + // transcript, defeating the row's memo; caching one per turn would mean + // writing a ref during render, which react-freeze can discard. + turnKey: turnKey && turnStatus?.workedSeconds != null ? turnKey : undefined, + // With no user boundary at all, the session's working state stays authoritative. + activeTurnIsWorking: + enabled && + isWorking && + (turnKey === activeTurnKey || + (turnKey === undefined && activeTurnKey === MOBILE_UNANCHORED_TURN_KEY)) + } + }, + [turnKeys, enabled, activeTurnKey, active, completedByTurn, expandedTurnIds, isWorking] + ) + + return { + active, + /** Stable for a given chat scope, so it never disturbs a row's memo. */ + onToggleTurn: toggleExpandedTurn, + activeTurnIsUnanchored: + enabled && active != null && activeTurnKey === MOBILE_UNANCHORED_TURN_KEY, + resolveRow + } +} diff --git a/mobile/src/session/use-mobile-native-chat-turn-status.ts b/mobile/src/session/use-mobile-native-chat-turn-status.ts new file mode 100644 index 00000000000..13afbe70c09 --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-turn-status.ts @@ -0,0 +1,105 @@ +import { useEffect, useMemo, useRef, useState } from 'react' +import type { NativeChatMessage } from '../../../src/shared/native-chat-types' +import { + nativeChatTurnHasResponse, + reduceNativeChatTurnTiming, + selectNativeChatTurnStatuses, + type NativeChatTurnStatus, + type NativeChatTurnTimingByTurn +} from '../../../src/shared/native-chat-turn-status' + +export type { NativeChatTurnStatus } + +export const MOBILE_UNANCHORED_TURN_KEY = '__unanchored__' +const EMPTY_TURN_TIMING_BY_TURN: NativeChatTurnTimingByTurn = Object.freeze({}) + +type ScopedTurnTiming = { + scopeKey: string + timingByTurn: NativeChatTurnTimingByTurn +} + +/** Per-turn "Thinking / Working for N / Worked for N" timing, on the same shared + * state machine the desktop renderer uses so the two surfaces stamp turns alike. */ +export function useMobileNativeChatTurnStatus({ + messages, + enabled, + isWorking, + workingStartedAt, + scopeKey +}: { + messages: readonly NativeChatMessage[] + enabled: boolean + isWorking: boolean + workingStartedAt?: number | null + /** Host/worktree/tab identity. Timings never carry across chat surfaces. */ + scopeKey: string +}): { + active: NativeChatTurnStatus | null + completedByTurn: Readonly> + activeTurnKey: string +} { + const latestUserIndex = enabled + ? messages.findLastIndex((message) => message.role === 'user') + : -1 + const hasCurrentTurnResponse = enabled && nativeChatTurnHasResponse(messages, latestUserIndex) + const latestUserId = latestUserIndex !== -1 ? (messages[latestUserIndex]?.id ?? null) : null + const activeTurnKey = latestUserId ?? MOBILE_UNANCHORED_TURN_KEY + const [scopedTiming, setScopedTiming] = useState(() => ({ + scopeKey, + timingByTurn: {} + })) + // Do not expose the previous surface's state during the render before the + // timing effect adopts the new scope, or scan it while this UI is disabled. + const timingByTurn = + enabled && scopedTiming.scopeKey === scopeKey + ? scopedTiming.timingByTurn + : EMPTY_TURN_TIMING_BY_TURN + // An accepted send renders as `pending-N` until the transcript echo lands under + // its real id. That is one turn under two keys, so the clock must survive the swap. + const previousActiveTurn = useRef<{ scopeKey: string; turnKey: string } | null>(null) + + useEffect(() => { + if (!enabled) { + return + } + const validTurnKeys = new Set( + messages.filter((message) => message.role === 'user').map((message) => message.id) + ) + const previousActiveTurnKey = + previousActiveTurn.current?.scopeKey === scopeKey + ? previousActiveTurn.current.turnKey + : undefined + previousActiveTurn.current = { scopeKey, turnKey: activeTurnKey } + setScopedTiming((current) => { + const currentTiming = + current.scopeKey === scopeKey ? current.timingByTurn : EMPTY_TURN_TIMING_BY_TURN + const nextTiming = reduceNativeChatTurnTiming(currentTiming, { + activeTurnKey, + previousActiveTurnKey, + validTurnKeys, + isWorking, + workingStartedAt, + now: Date.now() + }) + return current.scopeKey === scopeKey && nextTiming === currentTiming + ? current + : { scopeKey, timingByTurn: nextTiming } + }) + }, [activeTurnKey, enabled, isWorking, messages, scopeKey, workingStartedAt]) + + // Why: the selection rebuilds its status objects on every call, and a streaming + // turn re-renders ~20x/s. Without this, every settled turn's row gets fresh + // props each tick and the memoized message rows all re-render. + const turnIsWorking = enabled && isWorking + const statuses = useMemo( + () => + selectNativeChatTurnStatuses(timingByTurn, { + activeTurnKey, + isWorking: turnIsWorking, + workingStartedAt, + hasCurrentTurnResponse + }), + [timingByTurn, activeTurnKey, turnIsWorking, workingStartedAt, hasCurrentTurnResponse] + ) + return { ...statuses, activeTurnKey } +} diff --git a/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts b/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts index c0a4e8368c5..e46bf087f38 100644 --- a/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts +++ b/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts @@ -137,14 +137,17 @@ describe('mobile + Codex tab creation routing', () => { expect(scope.setActiveSessionTabId).toHaveBeenCalledWith('terminal-tab-1') }) - it('falls back to a terminal when structured creation is refused', async () => { + it('falls back to a terminal when structured creation is definitively refused', async () => { const client = clientReturning( { ok: true, result: { supported: true } }, { ok: true, result: { ok: false, - refusal: { code: 'agent_session_ownership_unknown', message: 'provider unavailable' } + refusal: { + code: 'structured_agent_session_unsupported', + message: 'provider unavailable' + } } }, terminalCreateResponse() @@ -228,4 +231,34 @@ describe('mobile + Codex tab creation routing', () => { expect(scope.setCreateError).toHaveBeenCalledWith('still unknown') expect(scope.showToast).toHaveBeenCalledWith('still unknown', 1800) }) + + it.each(['agent_session_operation_unknown', 'runtime_error', 'future_unknown_code'])( + 'does not create a legacy sibling after a top-level %s response', + async (code) => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { ok: false, error: { code, message: 'create outcome ambiguous' } } + ) + const scope = createScope(client) + let actions: ReturnType | undefined + function Harness() { + actions = useMobileSessionTerminalCreateActions(scope as never) + return null + } + await act(async () => { + renderer = create(createElement(Harness)) + }) + await act(async () => { + await actions?.handleCreateTerminal('codex') + }) + + const sendRequest = client.sendRequest as unknown as ReturnType + expect(sendRequest.mock.calls.map(([method]) => method)).toEqual([ + 'agentSession.createSupport', + 'agentSession.create' + ]) + expect(scope.setCreateError).toHaveBeenCalledWith('create outcome ambiguous') + expect(scope.showToast).toHaveBeenCalledWith('create outcome ambiguous', 1800) + } + ) }) diff --git a/package.json b/package.json index 58f4805d937..d1d930be708 100644 --- a/package.json +++ b/package.json @@ -141,6 +141,10 @@ "bench:main-thread-jank": "pnpm run ensure:electron-runtime && node tests/tools/benchmarks/main-thread-jank-bench.mjs", "bench:worktree-deletion": "node tests/tools/benchmarks/worktree-deletion-dev-bench.mjs", "bench:zustand-selector-fanout": "node config/scripts/zustand-selector-fanout-benchmark.mjs", + "bench:agent-inspection-cadence": "node config/scripts/agent-inspection-cadence-batching-benchmark.mjs", + "bench:renderer-quadratic-scans": "node config/scripts/renderer-quadratic-scan-benchmark.mjs", + "bench:session-write-hot-path": "node config/scripts/session-write-hot-path-benchmark.mjs", + "bench:terminal-partial-escape-tail": "node config/scripts/terminal-partial-escape-tail-benchmark.mjs", "bench:worktree-refresh-churn": "node --disable-warning=MODULE_TYPELESS_PACKAGE_JSON config/scripts/worktree-refresh-churn-benchmark.mjs", "bench:multi-workspace-typing": "pnpm run ensure:electron-runtime && node config/scripts/run-multi-workspace-typing-bench.mjs", "bench:ai-vault-typing": "pnpm run ensure:electron-runtime && node config/scripts/run-ai-vault-typing-bench.mjs", @@ -153,6 +157,7 @@ "repro:live-remote-realistic-freeze": "node config/scripts/live-remote-realistic-freeze-repro.mjs" }, "dependencies": { + "@anthropic-ai/claude-agent-sdk": "0.3.251", "@electron-toolkit/preload": "^3.0.2", "@electron-toolkit/utils": "^4.0.0", "@floating-ui/dom": "1.7.6", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index d7481f2e556..6b59d23e026 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -122,6 +122,9 @@ importers: .: dependencies: + '@anthropic-ai/claude-agent-sdk': + specifier: 0.3.251 + version: 0.3.251(@anthropic-ai/sdk@0.122.0(zod@4.5.4))(@modelcontextprotocol/sdk@1.30.0(supports-color@7.2.0)(zod@4.5.4))(zod@4.5.4) '@electron-toolkit/preload': specifier: ^3.0.2 version: 3.0.2(electron@43.4.1(supports-color@7.2.0)) @@ -535,6 +538,23 @@ packages: '@antfu/install-pkg@1.1.0': resolution: {integrity: sha512-MGQsmw10ZyI+EJo45CdSER4zEb+p31LpDAFp2Z3gkSd1yqVZGi0Ebx++YTEMonJy4oChEMLsxZ64j8FH6sSqtQ==} + '@anthropic-ai/claude-agent-sdk@0.3.251': + resolution: {integrity: sha512-DqSi8mH2tQYRlVV0G+lJnQ/WbjJZ/a+8cJ3vPuYoqh8esIIvXHm1ZOXV1UPGsFYRnbBytEoiSGitguEXd+sQ+Q==} + engines: {node: '>=18.0.0'} + peerDependencies: + '@anthropic-ai/sdk': '>=0.93.0' + '@modelcontextprotocol/sdk': ^1.29.0 + zod: ^4.0.0 + + '@anthropic-ai/sdk@0.122.0': + resolution: {integrity: sha512-GGPNftt0caaz9MDlmNQGHX8855Ojaduyy5pm9Sm1h7HalCn0cWNb5/bweadJF+4yzbal+QL6ztBa09WAAOzLmQ==} + hasBin: true + peerDependencies: + zod: ^3.25.0 || ^4.0.0 + peerDependenciesMeta: + zod: + optional: true + '@babel/code-frame@7.29.7': resolution: {integrity: sha512-Aup7aUOfpbAUg2ROOJN6Iw5f9DMBlzu0mIkm/malLQFN/YQgO48wCj0Kxa3sEHJvPVFg7siR+qRInwXd2qhQKw==} engines: {node: '>=6.9.0'} @@ -2628,6 +2648,9 @@ packages: resolution: {integrity: sha512-tlqY9xq5ukxTUZBmoOp+m61cqwQD5pHJtFY3Mn8CA8ps6yghLH/Hw8UPdqg4OLmFW3IFlcXnQNmo/dh8HzXYIQ==} engines: {node: '>=18'} + '@stablelib/base64@1.0.1': + resolution: {integrity: sha512-1bnPQqSxSuc3Ii6MhBysoWCg58j97aUjuCSZrGSmDxNqtytIi0k8utUenAwTZN4V5mXXYGsVUI9zeBqy+jBOSQ==} + '@stablyai/playwright-base@2.1.14': resolution: {integrity: sha512-/iAgMW5tC0ETDo3mFyTzszRrD7rGFIT4fgDgtZxqa9vPhiTLix/1+GeOOBNY0uS+XRLFY0Uc/irsC3XProL47g==} engines: {node: '>=18'} @@ -4493,6 +4516,9 @@ packages: resolution: {integrity: sha512-7MptL8U0cqcFdzIzwOTHoilX9x5BrNqye7Z/LuC7kCMRio1EMSyqRK3BEAUD7sXRq4iT4AzTVuZdhgQ2TCvYLg==} engines: {node: '>=8.6.0'} + fast-sha256@1.3.0: + resolution: {integrity: sha512-n11RGP/lrWEFI/bWdygLxhI+pVeo1ZYIVwvvPkW7azl/rOy+F3HYRZ2K5zeE9mmkhQppyv9sQFx0JM9UabnpPQ==} + fast-string-truncated-width@3.0.3: resolution: {integrity: sha512-0jjjIEL6+0jag3l2XWWizO64/aZVtpiGE3t0Zgqxv0DPuxiMjvB3M24fCyhZUO4KomJQPj3LTSUnDP3GpdwC0g==} @@ -5039,6 +5065,10 @@ packages: json-parse-even-better-errors@2.3.1: resolution: {integrity: sha512-xyFwyhro/JEof6Ghe2iz2NcXoj2sloNsWr/XsERDK/oiPCfaNhl5ONfp+jQdAZRQQ0IJWNzH9zIZF7li91kh2w==} + json-schema-to-ts@3.1.1: + resolution: {integrity: sha512-+DWg8jCJG2TEnpy7kOm/7/AxaYoaRbjVB4LFZLySZlWn8exGs3A4OLJR966cVvU26N7X9TWxl+Jsw7dzAqKT6g==} + engines: {node: '>=16'} + json-schema-traverse@1.0.0: resolution: {integrity: sha512-NM8/P9n3XjXhIZn1lLhkFaACTOURQXjWhV4BA/RnOv8xvgqtqpAX9IO4mRQxSx1Rlo4tqzeqb0sOlruaOy3dug==} @@ -6425,6 +6455,9 @@ packages: stackback@0.0.2: resolution: {integrity: sha512-1XMJE5fQo1jGH6Y/7ebnwPOBEkIEnT4QF32d5R1+VXdXveM0IBMJt8zfaxX1P3QhVwrYe+576+jkANtSS2mBbw==} + standardwebhooks@1.1.1: + resolution: {integrity: sha512-bCbX9ZEyFkWPsRz7Bl3NuQUJohmwGSev/yhr7vhaGPlc4AfIrspIRa6cPTBuI1ItmrTDJ4d/S2hCsfe4+vQGnQ==} + stat-mode@1.0.0: resolution: {integrity: sha512-jH9EhtKIjuXZ2cWxmXS8ZP80XyC3iasQxMDV8jzhNJpfDb7VbQLVW4Wvsxz9QZvzV+G4YoSfBUVKDOyxLzi/sg==} engines: {node: '>= 6'} @@ -6608,6 +6641,9 @@ packages: truncate-utf8-bytes@1.0.2: resolution: {integrity: sha512-95Pu1QXQvruGEhv62XCMO3Mm90GscOCClvrIUwCM0PYOXK3kaF3l3sIHxx71ThJfcbM2O5Au6SO3AWCSEfW4mQ==} + ts-algebra@2.0.0: + resolution: {integrity: sha512-FPAhNPFMrkwz76P7cdjdmiShwMynZYN6SgOujD1urY4oNm80Ou9oMdmbR45LotcKOXoy7wSmHkRFE6Mxbrhefw==} + ts-dedent@2.2.0: resolution: {integrity: sha512-q5W7tVM71e2xjHZTlgfTDoPF/SmqKG5hddq9SzR49CH2hayqRKJtQ4mtRlSxKaJlR/+9rEM+mnBHf7I2/BQcpQ==} engines: {node: '>=6.10'} @@ -7007,6 +7043,16 @@ packages: zwitch@2.0.4: resolution: {integrity: sha512-bXE4cR/kVZhKZX/RjPEflHaKVhUVl85noU3v6b8apfQEc1x4A+zBxjZ4lN8LqGd6WZ3dl98pY4o717VFmoPp+A==} +ignoredOptionalDependencies: + - '@anthropic-ai/claude-agent-sdk-darwin-arm64' + - '@anthropic-ai/claude-agent-sdk-darwin-x64' + - '@anthropic-ai/claude-agent-sdk-linux-arm64' + - '@anthropic-ai/claude-agent-sdk-linux-arm64-musl' + - '@anthropic-ai/claude-agent-sdk-linux-x64' + - '@anthropic-ai/claude-agent-sdk-linux-x64-musl' + - '@anthropic-ai/claude-agent-sdk-win32-arm64' + - '@anthropic-ai/claude-agent-sdk-win32-x64' + snapshots: '@adobe/css-tools@4.5.0': {} @@ -7016,6 +7062,19 @@ snapshots: package-manager-detector: 1.6.0 tinyexec: 1.1.2 + '@anthropic-ai/claude-agent-sdk@0.3.251(@anthropic-ai/sdk@0.122.0(zod@4.5.4))(@modelcontextprotocol/sdk@1.30.0(supports-color@7.2.0)(zod@4.5.4))(zod@4.5.4)': + dependencies: + '@anthropic-ai/sdk': 0.122.0(zod@4.5.4) + '@modelcontextprotocol/sdk': 1.30.0(supports-color@7.2.0)(zod@4.5.4) + zod: 4.5.4 + + '@anthropic-ai/sdk@0.122.0(zod@4.5.4)': + dependencies: + json-schema-to-ts: 3.1.1 + standardwebhooks: 1.1.1 + optionalDependencies: + zod: 4.5.4 + '@babel/code-frame@7.29.7': dependencies: '@babel/helper-validator-identifier': 7.29.7 @@ -7669,6 +7728,28 @@ snapshots: dependencies: '@chevrotain/types': 11.1.2 + '@modelcontextprotocol/sdk@1.30.0(supports-color@7.2.0)(zod@4.5.4)': + dependencies: + '@hono/node-server': 2.1.0(hono@4.13.0) + ajv: 8.20.0 + ajv-formats: 3.0.1(ajv@8.20.0) + content-type: 1.0.5 + cors: 2.8.6 + cross-spawn: 7.0.6 + eventsource: 3.0.7 + eventsource-parser: 3.0.8 + express: 5.2.1(supports-color@7.2.0) + express-rate-limit: 8.5.2(express@5.2.1(supports-color@7.2.0)) + hono: 4.13.0 + jose: 6.2.3 + json-schema-typed: 8.0.2 + pkce-challenge: 5.0.1 + raw-body: 3.0.2 + zod: 4.5.4 + zod-to-json-schema: 3.25.2(zod@4.5.4) + transitivePeerDependencies: + - supports-color + '@modelcontextprotocol/sdk@1.30.0(zod@3.25.76)': dependencies: '@hono/node-server': 2.1.0(hono@4.13.0) @@ -7679,8 +7760,8 @@ snapshots: cross-spawn: 7.0.6 eventsource: 3.0.7 eventsource-parser: 3.0.8 - express: 5.2.1 - express-rate-limit: 8.5.2(express@5.2.1) + express: 5.2.1(supports-color@7.2.0) + express-rate-limit: 8.5.2(express@5.2.1(supports-color@7.2.0)) hono: 4.13.0 jose: 6.2.3 json-schema-typed: 8.0.2 @@ -8917,6 +8998,8 @@ snapshots: '@sindresorhus/merge-streams@4.0.0': {} + '@stablelib/base64@1.0.1': {} + '@stablyai/playwright-base@2.1.14(@playwright/test@1.59.1)(zod@4.5.4)': dependencies: '@playwright/test': 1.59.1 @@ -9923,7 +10006,7 @@ snapshots: bluebird@3.7.2: {} - body-parser@2.3.0: + body-parser@2.3.0(supports-color@7.2.0): dependencies: bytes: 3.1.2 content-type: 2.0.0 @@ -10779,15 +10862,15 @@ snapshots: exponential-backoff@3.1.3: {} - express-rate-limit@8.5.2(express@5.2.1): + express-rate-limit@8.5.2(express@5.2.1(supports-color@7.2.0)): dependencies: - express: 5.2.1 + express: 5.2.1(supports-color@7.2.0) ip-address: 10.4.0 - express@5.2.1: + express@5.2.1(supports-color@7.2.0): dependencies: accepts: 2.0.0 - body-parser: 2.3.0 + body-parser: 2.3.0(supports-color@7.2.0) content-disposition: 1.1.0 content-type: 1.0.5 cookie: 0.7.2 @@ -10797,7 +10880,7 @@ snapshots: encodeurl: 2.0.0 escape-html: 1.0.3 etag: 1.8.1 - finalhandler: 2.1.1 + finalhandler: 2.1.1(supports-color@7.2.0) fresh: 2.0.0 http-errors: 2.0.1 merge-descriptors: 2.0.0 @@ -10808,9 +10891,9 @@ snapshots: proxy-addr: 2.0.7 qs: 6.15.2 range-parser: 1.2.1 - router: 2.2.0 - send: 1.2.1 - serve-static: 2.2.1 + router: 2.2.0(supports-color@7.2.0) + send: 1.2.1(supports-color@7.2.0) + serve-static: 2.2.1(supports-color@7.2.0) statuses: 2.0.2 type-is: 2.1.0 vary: 1.1.2 @@ -10831,6 +10914,8 @@ snapshots: merge2: 1.4.1 micromatch: 4.0.8 + fast-sha256@1.3.0: {} + fast-string-truncated-width@3.0.3: {} fast-string-width@3.0.2: @@ -10871,7 +10956,7 @@ snapshots: dependencies: to-regex-range: 5.0.1 - finalhandler@2.1.1: + finalhandler@2.1.1(supports-color@7.2.0): dependencies: debug: 4.4.3(supports-color@7.2.0) encodeurl: 2.0.0 @@ -11445,6 +11530,11 @@ snapshots: json-parse-even-better-errors@2.3.1: {} + json-schema-to-ts@3.1.1: + dependencies: + '@babel/runtime': 7.29.7 + ts-algebra: 2.0.0 + json-schema-traverse@1.0.0: {} json-schema-typed@8.0.2: {} @@ -13037,7 +13127,7 @@ snapshots: points-on-curve: 0.2.0 points-on-path: 0.2.1 - router@2.2.0: + router@2.2.0(supports-color@7.2.0): dependencies: debug: 4.4.3(supports-color@7.2.0) depd: 2.0.0 @@ -13080,7 +13170,7 @@ snapshots: semver@7.8.1: {} - send@1.2.1: + send@1.2.1(supports-color@7.2.0): dependencies: debug: 4.4.3(supports-color@7.2.0) encodeurl: 2.0.0 @@ -13107,12 +13197,12 @@ snapshots: transitivePeerDependencies: - typescript - serve-static@2.2.1: + serve-static@2.2.1(supports-color@7.2.0): dependencies: encodeurl: 2.0.0 escape-html: 1.0.3 parseurl: 1.3.3 - send: 1.2.1 + send: 1.2.1(supports-color@7.2.0) transitivePeerDependencies: - supports-color @@ -13267,6 +13357,11 @@ snapshots: stackback@0.0.2: {} + standardwebhooks@1.1.1: + dependencies: + '@stablelib/base64': 1.0.1 + fast-sha256: 1.3.0 + stat-mode@1.0.0: {} state-local@1.0.7: {} @@ -13437,6 +13532,8 @@ snapshots: dependencies: utf8-byte-length: 1.0.5 + ts-algebra@2.0.0: {} + ts-dedent@2.2.0: {} ts-morph@26.0.0: @@ -13786,6 +13883,10 @@ snapshots: dependencies: zod: 3.25.76 + zod-to-json-schema@3.25.2(zod@4.5.4): + dependencies: + zod: 4.5.4 + zod@3.25.76: {} zod@4.5.4: {} diff --git a/pnpm-workspace.yaml b/pnpm-workspace.yaml index d241459f884..97920f087c5 100644 --- a/pnpm-workspace.yaml +++ b/pnpm-workspace.yaml @@ -12,6 +12,20 @@ minimumReleaseAgeExclude: - zod@4.5.4 shamefullyHoist: true +# Orca always launches the user's own resolved Claude CLI via +# pathToClaudeCodeExecutable, so the SDK's bundled ~95 MB-per-platform CLI +# binaries must never be installed. Excluding them is what makes the path +# override mandatory rather than merely preferred. +ignoredOptionalDependencies: + - '@anthropic-ai/claude-agent-sdk-darwin-arm64' + - '@anthropic-ai/claude-agent-sdk-darwin-x64' + - '@anthropic-ai/claude-agent-sdk-linux-arm64' + - '@anthropic-ai/claude-agent-sdk-linux-arm64-musl' + - '@anthropic-ai/claude-agent-sdk-linux-x64' + - '@anthropic-ai/claude-agent-sdk-linux-x64-musl' + - '@anthropic-ai/claude-agent-sdk-win32-arm64' + - '@anthropic-ai/claude-agent-sdk-win32-x64' + supportedArchitectures: os: - current diff --git a/src/main/__fixtures__/shell-wrapper-snapshots/daemon-bash-rcfile.txt b/src/main/__fixtures__/shell-wrapper-snapshots/daemon-bash-rcfile.txt index 54837bf4d65..b79ed543494 100644 --- a/src/main/__fixtures__/shell-wrapper-snapshots/daemon-bash-rcfile.txt +++ b/src/main/__fixtures__/shell-wrapper-snapshots/daemon-bash-rcfile.txt @@ -127,10 +127,6 @@ __orca_osc133_precmd() { unset __orca_in_command fi printf "\033]133;A\007" - # Why: emit the shell-ready marker here (not a trailing PROMPT_COMMAND entry) - # so a framework that must be last in PROMPT_COMMAND — bash-preexec — is not - # displaced by one of Orca's own hooks. - [[ -n "$__orca_ready_marker" ]] && printf "\033]777;orca-shell-ready\007" return "$exit_code" } __orca_osc133_preexec() { @@ -188,6 +184,11 @@ __orca_osc133_epilogue() { unset __orca_in_prompt_command __orca_adopt_outer_debug_trap trap '__orca_osc133_preexec' DEBUG + # Readline renders PS1 after entering raw mode; prompt hooks still run in cooked mode. + if [[ -n "$__orca_ready_marker" ]]; then + PS1="${PS1-}"'\[\e]777;orca-shell-ready\a\]' + __orca_ready_marker="" + fi } __orca_normalize_prompt_command_part() { local __orca_value="$1" __orca_output_name="$2" __orca_character __orca_chunk diff --git a/src/main/automations/precheck-runner.ts b/src/main/automations/precheck-runner.ts index 753bd784b06..ab38fd42355 100644 --- a/src/main/automations/precheck-runner.ts +++ b/src/main/automations/precheck-runner.ts @@ -4,6 +4,7 @@ import type { AutomationPrecheck, AutomationPrecheckResult } from '../../shared/ import { MAX_AUTOMATION_PRECHECK_OUTPUT_CHARS } from '../../shared/automation-precheck' import { getSshConnectionManager } from '../ipc/ssh' import { shellEscape } from '../ssh/ssh-connection-utils' +import { admitSelfInitiatedTreeKill } from '../own-chromium-tree-kill-guard' type AutomationPrecheckExecutionTarget = | { @@ -73,7 +74,10 @@ function failedPrecheckResult( }) } -function killLocalPrecheckProcessTree(child: ChildProcess): ReturnType | null { +/** Exported for the refusal-fallback test; the timeout path is otherwise unreachable. */ +export function killLocalPrecheckProcessTree( + child: ChildProcess +): ReturnType | null { const pid = child.pid if (!pid) { child.kill() @@ -81,6 +85,18 @@ function killLocalPrecheckProcessTree(child: ChildProcess): ReturnType void) => Promise - } - PermissionStatus: PermissionStatusConstructor - dispatchPermissionChange: (name: string) => void - navigator: { - permissions: { - query: (descriptor: { name: string }) => Promise - } - } -} - -function createContext(args: { - nativeNotificationPermission: string - requestedNotificationPermission: string - rejectedPermissions?: string[] -}): AntiDetectionContext & Record { - class PermissionStatus extends EventTarget { - #state = 'denied' - #onchange: EventListener | null = null - marker = 'real-status' - - get state(): string { - return this.#state - } - - get onchange(): EventListener | null { - return this.#onchange - } - - set onchange(listener: EventListener | null) { - if (this.#onchange) { - super.removeEventListener('change', this.#onchange) - } - this.#onchange = typeof listener === 'function' ? listener : null - if (this.#onchange) { - super.addEventListener('change', this.#onchange) - } - } - } - - const statuses = new Map() - const rejectedPermissions = new Set(args.rejectedPermissions) - - class Permissions { - query(descriptor: { name: string }): Promise { - if (rejectedPermissions.has(descriptor.name)) { - return Promise.reject(new Error('Unsupported permission')) - } - const status = new PermissionStatus() - const permissionStatuses = statuses.get(descriptor.name) ?? [] - permissionStatuses.push(status) - statuses.set(descriptor.name, permissionStatuses) - return Promise.resolve(status) - } - } - - const Notification = { - permission: args.nativeNotificationPermission, - requestPermission(callback?: (permission: string) => void): Promise { - callback?.(args.requestedNotificationPermission) - return Promise.resolve(args.requestedNotificationPermission) - } - } - Object.defineProperty(Notification, 'permission', { - configurable: true, - get: () => args.nativeNotificationPermission - }) - - return { - Date, - Event, - EventTarget, - Object, - Promise, - Set, - performance: { now: () => 0 }, - // Why: the script's Firefox gate reads navigator.userAgent, so a non-Firefox UA keeps these - // tests on the ordinary-page path where the PermissionStatus override applies. - window: { chrome: {} }, - navigator: { - userAgent: - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/150.0.0.0 Safari/537.36', - plugins: [], - languages: [], - permissions: new Permissions() - }, - Permissions, - PermissionStatus, - Notification, - dispatchPermissionChange(name: string): void { - for (const status of statuses.get(name) ?? []) { - status.dispatchEvent(new Event('change')) - } - } - } as AntiDetectionContext & Record -} - -describe('ANTI_DETECTION_SCRIPT — PermissionStatus', () => { - it('keeps an existing notification status current after permission changes', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - const status = await context.navigator.permissions.query({ name: 'notifications' }) - - expect(context.Notification.permission).toBe('default') - expect(status.state).toBe('prompt') - - await expect(context.Notification.requestPermission()).resolves.toBe('granted') - - expect(context.Notification.permission).toBe('granted') - expect(status.state).toBe('granted') - }) - - it('preserves native PermissionStatus identity and methods', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - const status = await context.navigator.permissions.query({ name: 'camera' }) - const expectedSource = Function.prototype.toString.call( - context.PermissionStatus.prototype.addEventListener - ) - - expect(status).toBeInstanceOf(context.PermissionStatus) - expect(status.state).toBe('prompt') - expect(status.constructor.name).toBe('PermissionStatus') - expect(status.marker).toBe('real-status') - expect(status.addEventListener.name).toBe('addEventListener') - expect(status.addEventListener).toBe(status.addEventListener) - expect(Function.prototype.toString.call(status.addEventListener)).toBe(expectedSource) - expect(expectedSource).toContain('addEventListener') - }) - - it('delivers change events through the returned status with the overridden state', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - const status = await context.navigator.permissions.query({ name: 'notifications' }) - const events: { receiver: EventTarget; target: EventTarget | null; state: string }[] = [] - const recordEvent = function (this: EventTarget, event: Event): void { - events.push({ - receiver: this, - target: event.target, - state: (event.target as PermissionQueryResult).state - }) - } - - status.addEventListener('change', recordEvent) - expect(() => { - status.onchange = function (this: EventTarget, event): void { - recordEvent.call(this, event) - } - }).not.toThrow() - - await context.Notification.requestPermission() - context.dispatchPermissionChange('notifications') - - expect(events).toHaveLength(2) - expect(events).toEqual([ - { receiver: status, target: status, state: 'granted' }, - { receiver: status, target: status, state: 'granted' } - ]) - }) - - // Why: 'camera' rather than 'storage-access' — #14685 narrowed the intercepted set to - // camera/microphone, so a name outside it falls through to the real query and never reaches - // the fallback at all. - it('uses a non-enumerable EventTarget fallback when the native query rejects', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted', - rejectedPermissions: ['camera'] - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - const status = await context.navigator.permissions.query({ name: 'camera' }) - - expect(status).toBeInstanceOf(EventTarget) - expect(status).not.toBeInstanceOf(context.PermissionStatus) - expect(status.state).toBe('prompt') - expect(Object.keys(status)).toEqual([]) - }) -}) diff --git a/src/main/browser/anti-detection.test.ts b/src/main/browser/anti-detection.test.ts deleted file mode 100644 index ddb904cb436..00000000000 --- a/src/main/browser/anti-detection.test.ts +++ /dev/null @@ -1,157 +0,0 @@ -import { runInNewContext } from 'node:vm' -import { describe, expect, it } from 'vitest' - -import { ANTI_DETECTION_SCRIPT } from './anti-detection' -import { googleAuthUserAgent } from './browser-google-auth-ua' - -type PermissionQueryResult = { - state: string - onchange: null -} - -type AntiDetectionContext = { - Notification: { - permission: string - requestPermission: (callback?: (permission: string) => void) => Promise - } - navigator: { - userAgent: string - permissions: { - query: (descriptor: { name: string }) => Promise - } - } - window: { - chrome?: { - runtime?: unknown - csi?: () => unknown - loadTimes?: () => unknown - } - } -} - -function createContext(args: { - nativeNotificationPermission: string - requestedNotificationPermission: string - userAgent?: string -}): AntiDetectionContext & Record { - class Permissions { - query(): Promise { - return Promise.resolve({ state: 'denied', onchange: null }) - } - } - - const Notification = { - permission: args.nativeNotificationPermission, - requestPermission(callback?: (permission: string) => void): Promise { - callback?.(args.requestedNotificationPermission) - return Promise.resolve(args.requestedNotificationPermission) - } - } - Object.defineProperty(Notification, 'permission', { - configurable: true, - get: () => args.nativeNotificationPermission - }) - - return { - Date, - Object, - Promise, - Set, - performance: { now: () => 0 }, - // Electron 43 exposes this native object before the anti-detection script runs. - window: { chrome: {} }, - navigator: { - userAgent: - args.userAgent ?? - 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko) Chrome/151.0.0.0 Safari/537.36', - plugins: [], - languages: [], - permissions: new Permissions() - }, - Permissions, - Notification - } as AntiDetectionContext & Record -} - -describe('ANTI_DETECTION_SCRIPT', () => { - it('does not expose Chrome globals under a Firefox identity', () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'denied', - userAgent: googleAuthUserAgent() - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - expect(context.window.chrome).toBeUndefined() - expect('chrome' in context.window).toBe(false) - }) - - it('keeps Chrome API stubs aligned with an ordinary Chrome page', () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'denied' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - expect(context.window.chrome?.runtime).toBeUndefined() - expect(context.window.chrome?.csi).toBeTypeOf('function') - expect(context.window.chrome?.loadTimes).toBeTypeOf('function') - }) - - it.each(['geolocation', 'idle-detection', 'midi', 'storage-access'])( - 'passes non-intercepted permission queries through to the native state for %s', - async (name) => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'denied' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - await expect(context.navigator.permissions.query({ name })).resolves.toEqual({ - state: 'denied', - onchange: null - }) - } - ) - - it('reports notification permission as granted after a site permission request succeeds', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - expect(context.Notification.permission).toBe('default') - await expect(context.navigator.permissions.query({ name: 'notifications' })).resolves.toEqual({ - state: 'prompt', - onchange: null - }) - - await expect(context.Notification.requestPermission()).resolves.toBe('granted') - - expect(context.Notification.permission).toBe('granted') - await expect(context.navigator.permissions.query({ name: 'notifications' })).resolves.toEqual({ - state: 'granted', - onchange: null - }) - }) - - it('preserves notification permission when Electron already reports a grant', async () => { - const context = createContext({ - nativeNotificationPermission: 'granted', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - expect(context.Notification.permission).toBe('granted') - await expect(context.navigator.permissions.query({ name: 'notifications' })).resolves.toEqual({ - state: 'granted', - onchange: null - }) - }) -}) diff --git a/src/main/browser/anti-detection.ts b/src/main/browser/anti-detection.ts deleted file mode 100644 index d725fbf965f..00000000000 --- a/src/main/browser/anti-detection.ts +++ /dev/null @@ -1,161 +0,0 @@ -// Why: Cloudflare Turnstile and similar bot detectors probe multiple browser -// APIs beyond navigator.webdriver. This script runs via -// Page.addScriptToEvaluateOnNewDocument before any page JS to mask automation -// signals that CDP debugger attachment and Electron's webview expose. -export const ANTI_DETECTION_SCRIPT = `(function() { - Object.defineProperty(navigator, 'webdriver', { get: () => false }); - // Why: Electron webviews expose an empty plugins array. Real Chrome always - // has at least a few default plugins (PDF Viewer, etc.). An empty array is - // a strong automation signal. - if (navigator.plugins.length === 0) { - Object.defineProperty(navigator, 'plugins', { - get: () => [ - { name: 'Chrome PDF Plugin', filename: 'internal-pdf-viewer' }, - { name: 'Chrome PDF Viewer', filename: 'mhjfbmdgcfjbbpaeojofohoefgiehjai' }, - { name: 'Native Client', filename: 'internal-nacl-plugin' } - ] - }); - } - // Why: auth hosts present Firefox, where Electron's native window.chrome is an identity mismatch. - if (navigator.userAgent.includes('Firefox/')) { - try { - delete window.chrome; - if ('chrome' in window) { - window.chrome = undefined; - } - } catch {} - } else { - // Why: Electron webviews may not have the window.chrome object that real - // Chrome exposes. Turnstile checks for its presence. The csi() and - // loadTimes() stubs satisfy deeper probes of Chrome-specific APIs. - if (!window.chrome) { - window.chrome = {}; - } - if (!window.chrome.csi) { - window.chrome.csi = function() { - return { - startE: Date.now(), - onloadT: Date.now(), - pageT: performance.now(), - tran: 15 - }; - }; - } - if (!window.chrome.loadTimes) { - window.chrome.loadTimes = function() { - return { - commitLoadTime: Date.now() / 1000, - connectionInfo: 'h2', - finishDocumentLoadTime: Date.now() / 1000, - finishLoadTime: Date.now() / 1000, - firstPaintAfterLoadTime: 0, - firstPaintTime: Date.now() / 1000, - navigationType: 'Other', - npnNegotiatedProtocol: 'h2', - requestTime: Date.now() / 1000 - 0.16, - startLoadTime: Date.now() / 1000 - 0.3, - wasAlternateProtocolAvailable: false, - wasFetchedViaSpdy: true, - wasNpnNegotiated: true - }; - }; - } - } - // Why: Electron's Permission API defaults to 'denied' for most permissions, - // but real Chrome returns 'prompt' for ungranted permissions. Returning - // 'denied' is a strong bot signal. Override the query result for common - // permissions that Turnstile and similar detectors probe. - var notificationPermission = 'default'; - var setNotificationPermission = function(permission) { - if (permission === 'granted' || permission === 'denied') { - notificationPermission = permission; - return permission; - } - notificationPermission = 'default'; - return 'default'; - }; - var notificationPermissionState = function() { - return notificationPermission === 'default' ? 'prompt' : notificationPermission; - }; - try { - if (Notification.permission === 'granted') { - notificationPermission = 'granted'; - } - } catch {} - const promptPerms = new Set([ - 'camera', 'microphone' - ]); - const origQuery = Permissions.prototype.query; - // Why: sites must receive the genuine PermissionStatus so native events, brand checks and method - // identity survive. Shadow only state, and resolve it lazily so existing statuses stay current. - function withOverriddenState(realStatus, stateProvider) { - Object.defineProperty(realStatus, 'state', { - configurable: true, - get: stateProvider - }); - return realStatus; - } - // Why: some names the real implementation rejects outright; fall back to an EventTarget so - // listener registration still works instead of throwing. - function fallbackStatus(stateProvider) { - const status = new EventTarget(); - Object.defineProperties(status, { - state: { configurable: true, get: stateProvider }, - onchange: { configurable: true, value: null, writable: true } - }); - return status; - } - function queryWithState(permissions, desc, stateProvider) { - let real; - try { - real = origQuery.call(permissions, desc); - } catch { - return Promise.resolve(fallbackStatus(stateProvider)); - } - return Promise.resolve(real).then( - (status) => withOverriddenState(status, stateProvider), - () => fallbackStatus(stateProvider) - ); - } - Permissions.prototype.query = function(desc) { - if (desc.name === 'notifications') { - return queryWithState(this, desc, notificationPermissionState); - } - if (promptPerms.has(desc.name)) { - return queryWithState(this, desc, () => 'prompt'); - } - return origQuery.call(this, desc); - }; - // Why: Electron may report Notification.permission as 'denied' by default - // whereas real Chrome reports 'default' for sites that haven't been granted - // or blocked. Turnstile cross-references this with the Permissions API. - try { - Object.defineProperty(Notification, 'permission', { - get: () => notificationPermission - }); - const origRequestPermission = Notification.requestPermission; - if (typeof origRequestPermission === 'function') { - Notification.requestPermission = function(callback) { - var wrappedCallback = typeof callback === 'function' - ? function(permission) { - callback(setNotificationPermission(permission)); - } - : undefined; - var result = origRequestPermission.call(Notification, wrappedCallback); - if (result && typeof result.then === 'function') { - return result.then(function(permission) { - return setNotificationPermission(permission); - }); - } - return result; - }; - } - } catch {} - // Why: Electron webviews may have an empty languages array. Real Chrome - // always has at least one entry. An empty array is an automation signal. - if (!navigator.languages || navigator.languages.length === 0) { - Object.defineProperty(navigator, 'languages', { - get: () => ['en-US', 'en'] - }); - } -})()` diff --git a/src/main/browser/browser-google-auth-ua.ts b/src/main/browser/browser-google-auth-ua.ts index 16ed14eb80f..e9b802f6d70 100644 --- a/src/main/browser/browser-google-auth-ua.ts +++ b/src/main/browser/browser-google-auth-ua.ts @@ -1,12 +1,12 @@ // Why: Google binds a signed-in session to the browser identity that created it. -// Cookies copied in from another browser (or sent under an Electron/Chrome-shaped -// UA that doesn't match a real first-party browser) get flagged by anti-fraud on -// accounts.google.com and expire within ~1h. Presenting a Firefox identity scoped +// Cookies copied in from another browser (or sent under a UA that doesn't match a +// real first-party browser) get flagged by anti-fraud on accounts.google.com and +// expire within ~1h. Presenting a Firefox identity scoped // to Google's auth hosts lets the user sign in *inside* the embedded browser, so // Google issues cookies bound to THIS browser that self-refresh — instead of us // transplanting cookies that go stale. Scope is deliberately the auth hosts only: // post-auth app surfaces (mail.google.com, myaccount.google.com, drive, etc.) keep -// the profile's real Chrome-shaped identity so nothing else about the session shifts. +// the profile's real identity so nothing else about the session shifts. // Why: exact hostname match — subdomains such as myaccount.google.com are post-auth // app surfaces, not the sign-in flow, and must retain the profile's real identity. diff --git a/src/main/browser/browser-manager-auth-user-agent.test.ts b/src/main/browser/browser-manager-auth-user-agent.test.ts index 405b689e850..eb0bd75b354 100644 --- a/src/main/browser/browser-manager-auth-user-agent.test.ts +++ b/src/main/browser/browser-manager-auth-user-agent.test.ts @@ -51,7 +51,7 @@ import { import { createViewportGuestFactory, flushViewportOps, - GUEST_CLEAN_UA + GUEST_ELECTRON_UA } from './browser-manager-viewport-test-fixtures' const { @@ -197,8 +197,9 @@ describe('browserManager', () => { // Why: popup child windows get attachGuestPolicies but are never entered into tabIdByWebContentsId, // so a direct lookup of the UA mode misses the native opt-out. That is worse than doing nothing — - // native sessions skip setupClientHintsOverride, so the popup would send the raw Electron UA on the - // wire while navigator.userAgent claimed Firefox. Google sign-in popups are a first-class surface. + // native sessions never install the header-level Firefox switch, so the popup would send the + // Electron UA on the wire while navigator.userAgent claimed Firefox. Google sign-in popups are a + // first-class surface. it('leaves the UA untouched on auth hosts for a popup owned by a native-UA profile', () => { const ownerGuest = { id: 415, @@ -543,7 +544,7 @@ describe('browserManager', () => { ) expect(uaWrites.length).toBeGreaterThan(0) for (const [, params] of uaWrites) { - expect((params as { userAgent: string }).userAgent).toBe(GUEST_CLEAN_UA) + expect((params as { userAgent: string }).userAgent).toBe(GUEST_ELECTRON_UA) } }) }) diff --git a/src/main/browser/browser-manager-guest-lifecycle.test.ts b/src/main/browser/browser-manager-guest-lifecycle.test.ts index c8550083dfc..e136f532474 100644 --- a/src/main/browser/browser-manager-guest-lifecycle.test.ts +++ b/src/main/browser/browser-manager-guest-lifecycle.test.ts @@ -637,9 +637,9 @@ describe('browserManager', () => { ).toHaveLength(2) }) - it('cancels pending anti-detection reattach timers when unregistering a guest', () => { - vi.useFakeTimers() - + // Why: a plain browsing tab must never attach a debugger (Cloudflare treats CDP as a bot signal); + // the only debugger wiring it keeps is the detach listener that invalidates the auth-host UA override. + it('never attaches a debugger to a browsing guest and drops its detach listener on unregister', () => { const debuggerHandlers = new Map void>() const debuggerAttachMock = vi.fn() const guest = { @@ -670,18 +670,17 @@ describe('browserManager', () => { browserManager.attachGuestPolicies(guest as never) browserManager.registerGuest({ - browserPageId: 'browser-reattach', + browserPageId: 'browser-no-debugger', webContentsId: 809, rendererWebContentsId }) - debuggerHandlers.get('detach')?.() - expect(vi.getTimerCount()).toBe(1) + expect(debuggerAttachMock).not.toHaveBeenCalled() + expect(guest.debugger.sendCommand).not.toHaveBeenCalled() + expect(debuggerHandlers.has('detach')).toBe(true) - browserManager.unregisterGuest('browser-reattach') - expect(vi.getTimerCount()).toBe(0) - - vi.advanceTimersByTime(500) - expect(debuggerAttachMock).toHaveBeenCalledTimes(1) + browserManager.unregisterGuest('browser-no-debugger') + expect(debuggerHandlers.has('detach')).toBe(false) + expect(debuggerAttachMock).not.toHaveBeenCalled() }) }) diff --git a/src/main/browser/browser-manager-guest-policy-profile.test.ts b/src/main/browser/browser-manager-guest-policy-profile.test.ts index 47871f64ea9..c5c4c2521fe 100644 --- a/src/main/browser/browser-manager-guest-policy-profile.test.ts +++ b/src/main/browser/browser-manager-guest-policy-profile.test.ts @@ -52,6 +52,8 @@ type GuestFake = { isAttached: () => boolean attach: ReturnType sendCommand: ReturnType + on: ReturnType + off: ReturnType } on: (event: string, listener: (...args: never[]) => void) => void once: (event: string, listener: (...args: never[]) => void) => void @@ -81,7 +83,9 @@ function createGuest(id: number, url: string): GuestFake { debugger: { isAttached: () => true, attach: vi.fn(), - sendCommand: vi.fn(async () => undefined) + sendCommand: vi.fn(async () => undefined), + on: vi.fn(), + off: vi.fn() }, on: (event, listener) => { listeners.set(event, [...(listeners.get(event) ?? []), listener]) @@ -143,7 +147,7 @@ describe('guest policy profiles', () => { // The presence half of every absence below: a browsing guest observably takes all of it through // the same method, so a profile that fenced nothing — or an attach path that stopped installing // anything at all — cannot pass these by being uniformly empty. - it('gives a browsing guest link routing, popups and anti-detection', () => { + it('gives a browsing guest link routing, popups and auth-identity detach tracking', () => { const guest = createGuest(300, 'https://example.com/') browserManager.attachGuestPolicies(guest as never) @@ -151,7 +155,10 @@ describe('guest policy profiles', () => { expect(listenerCount(guest, 'dom-ready')).toBe(1) expect(listenerCount(guest, 'frame-created')).toBe(1) expect(listenerCount(guest, 'did-create-window')).toBe(1) - expect(guest.debugger.sendCommand).toHaveBeenCalled() + expect(guest.debugger.on).toHaveBeenCalledWith('detach', expect.any(Function)) + // Why: a plain browsing tab must never attach a debugger; Cloudflare treats CDP as a bot signal. + expect(guest.debugger.attach).not.toHaveBeenCalled() + expect(guest.debugger.sendCommand).not.toHaveBeenCalled() expect(navigateTo(guest, 'https://elsewhere.example/')).toBe(false) }) @@ -161,6 +168,7 @@ describe('guest policy profiles', () => { expect(listenerCount(guest, 'dom-ready')).toBe(0) expect(listenerCount(guest, 'frame-created')).toBe(0) expect(listenerCount(guest, 'did-create-window')).toBe(0) + expect(guest.debugger.on).not.toHaveBeenCalled() expect(guest.debugger.sendCommand).not.toHaveBeenCalled() expect(guest.executeJavaScriptInIsolatedWorld).not.toHaveBeenCalled() }) diff --git a/src/main/browser/browser-manager-guest-policy.ts b/src/main/browser/browser-manager-guest-policy.ts index c0d522235c8..3952a31c08b 100644 --- a/src/main/browser/browser-manager-guest-policy.ts +++ b/src/main/browser/browser-manager-guest-policy.ts @@ -35,8 +35,7 @@ export abstract class BrowserManagerGuestPolicy extends BrowserManagerGuestClean this.clickedLinkFrameNameByGuestId.set(guest.id, clickedLinkFrameName) } - // Why: bot detectors probe APIs that differ in Electron webviews; inject overrides each load so manual browsing passes. - const disposeAntiDetection = this.injectAntiDetection(guest) + const disposeAuthDetachTracking = this.trackDebuggerDetachForAuthUserAgent(guest) // Why: disable throttling so background screenshots still get frames; else the compositor stalls and capture returns empty. guest.setBackgroundThrottling(false) const disposePopupPolicy = this.installGuestPopupPolicy(guest, clickedLinkFrameName) @@ -44,14 +43,14 @@ export abstract class BrowserManagerGuestPolicy extends BrowserManagerGuestClean // Why: store cleanup so unregisterGuest can drop these listeners on teardown and let the WebContents wrapper GC. this.policyCleanupByGuestId.set(guest.id, () => { - disposeAntiDetection() + disposeAuthDetachTracking() disposePopupPolicy() disposeNavigationPolicy() }) } /** - * A workspace document is not the web: no popups, no link routing, no anti-detection, and no + * A workspace document is not the web: no popups, no link routing, no auth-identity tracking, and no * navigation bookkeeping for chrome it does not have. What it does share with a browsing guest is * this method's teardown, so a retired preview drops its listeners on the same path. */ diff --git a/src/main/browser/browser-manager-navigation.ts b/src/main/browser/browser-manager-navigation.ts index 4e061d288aa..5cb46ccb682 100644 --- a/src/main/browser/browser-manager-navigation.ts +++ b/src/main/browser/browser-manager-navigation.ts @@ -1,5 +1,4 @@ import { openPopupWithOriginBar, type PopupChildWindowOptions } from './popup-origin-bar-window' -import { cleanElectronUserAgent } from './browser-session-ua' import { getBrowserSessionUserAgentMode } from './browser-session-user-agent-mode' import { googleAuthUserAgent, isGoogleAuthUrl } from './browser-google-auth-ua' import { buildViewportUserAgentOverride } from './browser-viewport-user-agent' @@ -12,10 +11,10 @@ import { BrowserManagerVisibility } from './browser-manager-visibility' export abstract class BrowserManagerNavigation extends BrowserManagerVisibility { // Why: navigator.userAgent (read by Google's auth JS) reflects the WebContents UA, - // not the request header, so the header-level Firefox switch in setupClientHintsOverride + // not the request header, so the header-level Firefox switch in setupGoogleAuthUserAgentOverride // must be matched here per navigation or the two layers disagree — itself a bot tell. // Restores the session's base identity off the auth hosts. Native-UA profiles opt out - // of the whole clean-UA path, so they keep their untouched identity everywhere. + // of the Firefox switch, so they keep their untouched identity everywhere. protected applyGoogleAuthUserAgent( guest: Electron.WebContents, url: string, @@ -24,8 +23,8 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility const browserPageId = this.tabIdByWebContentsId.get(guest.id) // Why: popup child windows get these policies but are never in tabIdByWebContentsId, so a direct // lookup misses the native-UA opt-out and would hand a native profile's popup the Firefox UA. - // That is worse than doing nothing: native sessions skip setupClientHintsOverride entirely, so - // the popup would send the raw Electron UA on the wire while navigator.userAgent claims Firefox. + // That is worse than doing nothing: native sessions never install the header-level Firefox + // switch, so the popup would send the Electron UA on the wire while navigator.userAgent claims Firefox. const ownerTabId = this.resolveBrowserTabIdForGuestWebContentsId(guest.id) // Session state is authoritative before renderer registration and after a native profile imports a source UA. const mode = @@ -56,15 +55,14 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility // navigation (ERR_ABORTED) and replay the original request, which a POST-started OAuth chain // cannot survive — the sign-in lands on a blank tab. CDP retargets navigator.userAgent without // touching the navigation, and it outranks the WebContents UA from then on, so a guest that - // switches to it stays on it. The wire UA never depended on this write: setupClientHintsOverride - // rewrites User-Agent per request for auth-host URLs on its own. + // switches to it stays on it. The wire UA never depended on this write: + // setupGoogleAuthUserAgentOverride rewrites User-Agent per request for auth-host URLs on its own. if (options.duringRedirect === true || overrideState !== undefined) { if (this.canOverrideUserAgentOverCdp(guest)) { authOverrideIssuedOverCdp = true // Why: go through the viewport builder rather than writing nextUa raw, so both CDP writers - // resolve one identity for this URL — Firefox on auth hosts, the profile's clean base off - // them, any mobile preset preserved. Writing the session UA directly would put the - // unlaundered Electron token back on the wire. + // resolve one identity for this URL — Firefox on auth hosts, the session's base identity + // off them, any mobile preset preserved. void this.applyAuthUserAgentOverrideOverCdp( guest, (browserPageId ? this.viewportUaOverrideMobileByTabId.get(browserPageId) : undefined) ?? @@ -188,7 +186,7 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility // Why: Emulation.setUserAgentOverride is set once and stands across every later navigation, // outranking setUserAgent for navigator.userAgent. A viewport preset applied before reaching an - // auth host would otherwise pin navigator.userAgent to the Chrome-shaped preset UA while the + // auth host would otherwise pin navigator.userAgent to the session's preset UA while the // request header says Firefox — the two-layer disagreement this scope exists to remove. protected reapplyViewportUserAgentOverride( guest: Electron.WebContents, @@ -222,7 +220,7 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility // Why: the session UA is the profile's stable base identity. guest.getUserAgent() is not: // applyGoogleAuthUserAgent leaves it pinned to the Firefox auth UA once a guest switches to // the CDP override, so reading it back here would republish that identity on ordinary hosts. - baseUserAgent: cleanElectronUserAgent(baseUserAgent ?? guest.session.getUserAgent()) + baseUserAgent: baseUserAgent ?? guest.session.getUserAgent() }) ) } diff --git a/src/main/browser/browser-manager-state.ts b/src/main/browser/browser-manager-state.ts index bc65cc3d2dc..54b0f99b3c1 100644 --- a/src/main/browser/browser-manager-state.ts +++ b/src/main/browser/browser-manager-state.ts @@ -1,4 +1,3 @@ -import { ANTI_DETECTION_SCRIPT } from './anti-detection' import { BrowserGrabSessionController } from './browser-grab-session-controller' import type { BrowserCertificateTrustController } from './browser-certificate-trust-controller' import { @@ -190,56 +189,19 @@ export abstract class BrowserManagerState extends BrowserManagerViewportScrollSt this.settingsResolver = resolver } - // Why: addScriptToEvaluateOnNewDocument (CDP) is the only reliable pre-page-script hook per nav; executeJavaScript ran on the old page context. - protected injectAntiDetection(guest: Electron.WebContents): () => void { - let disposed = false - let reattachTimer: ReturnType | null = null - - const attach = (): void => { - if (disposed || guest.isDestroyed()) { - return - } - try { - if (!guest.debugger.isAttached()) { - guest.debugger.attach('1.3') - } - void guest.debugger - .sendCommand('Page.enable', {}) - .then(() => - guest.debugger.sendCommand('Page.addScriptToEvaluateOnNewDocument', { - source: ANTI_DETECTION_SCRIPT - }) - ) - .catch(() => {}) - } catch { - /* best-effort — debugger may be unavailable */ - } - } - - // Why: proxy/bridge stop detaches the debugger and drops injections; re-attach (500ms delay to avoid racing a mid-restart) to keep overrides. + // Why: a debugger detach clears every CDP override Chromium holds, including the Google auth-host + // UA override, so the confirmed-override record must be dropped or the next auth navigation + // believes the identity is still installed and skips the write. + protected trackDebuggerDetachForAuthUserAgent(guest: Electron.WebContents): () => void { const onDetach = (): void => { this.authUserAgentOverrideStateByGuestId.delete(guest.id) - if (!disposed && !guest.isDestroyed() && reattachTimer === null) { - reattachTimer = setTimeout(() => { - reattachTimer = null - attach() - }, 500) - } } - try { - attach() guest.debugger.on('detach', onDetach) } catch { - /* best-effort */ + /* debugger may be unavailable */ } - return () => { - disposed = true - if (reattachTimer !== null) { - clearTimeout(reattachTimer) - reattachTimer = null - } try { guest.debugger.off('detach', onDetach) } catch { diff --git a/src/main/browser/browser-manager-types.ts b/src/main/browser/browser-manager-types.ts index a1b832a65bc..b91e2b741fe 100644 --- a/src/main/browser/browser-manager-types.ts +++ b/src/main/browser/browser-manager-types.ts @@ -117,7 +117,7 @@ export type PopupOwnerContext = { /** * What a guest is allowed to be. A browsing guest is the web — popups, clicked-link routing and - * anti-detection all apply. A workspace-document guest renders one granted document and gets none + * auth-identity tracking all apply. A workspace-document guest renders one granted document and gets none * of that; `host` is the renderer that minted its grant, and the only sink for what it reports. */ export type BrowserGuestPolicy = diff --git a/src/main/browser/browser-manager-viewport-override.test.ts b/src/main/browser/browser-manager-viewport-override.test.ts index b7d3bbabe0a..0ffc3c2a6e1 100644 --- a/src/main/browser/browser-manager-viewport-override.test.ts +++ b/src/main/browser/browser-manager-viewport-override.test.ts @@ -49,7 +49,6 @@ import { import { createViewportGuestFactory, flushViewportOps, - GUEST_CLEAN_UA, GUEST_ELECTRON_UA } from './browser-manager-viewport-test-fixtures' @@ -207,7 +206,7 @@ describe('browserManager', () => { mobile: false }) expect(debuggerSendCommand).toHaveBeenLastCalledWith('Emulation.setUserAgentOverride', { - userAgent: GUEST_CLEAN_UA + userAgent: GUEST_ELECTRON_UA }) // Navigating to the auth host must move the standing override to the Firefox identity. @@ -218,11 +217,11 @@ describe('browserManager', () => { userAgent: googleAuthUserAgent() }) - // Leaving the auth host restores the clean Chrome-shaped preset UA. + // Leaving the auth host restores the session's own preset UA. debuggerSendCommand.mockClear() willRedirect({ preventDefault: vi.fn() }, 'https://example.com/', false, true) await flushViewportOps() - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) }) // Why: not an ordering race — debugger.sendCommand dispatches in call order over one channel, so @@ -241,9 +240,9 @@ describe('browserManager', () => { } // Why mobile: on the desktop branch the break is masked by coincidence — applyGoogleAuthUserAgent - // has already switched the WebContents UA to Firefox, and cleanElectronUserAgent passes a Firefox - // UA through untouched, so the stale-URL desktop path happens to emit Firefox anyway. The mobile - // branch derives a Chrome-shaped iPhone UA from that same base and exposes the real defect. + // has already switched the WebContents UA to Firefox, so the stale-URL desktop path happens to + // emit Firefox anyway. The mobile branch derives a Chrome-shaped iPhone UA from the session base + // and exposes the real defect. it('does not leave the Chrome preset UA standing when a mobile preset lands mid-navigation onto an auth host', async () => { const { guest, debuggerSendCommand } = makeGuest(4251, 'https://example.com/') // Hold the preset's first CDP command open so the navigation lands inside its await window. @@ -332,7 +331,7 @@ describe('browserManager', () => { // Without the fix the resuming preset re-reads getURL() as the auth host and clobbers the // navigation's correct write, stranding the Firefox UA on a non-auth page. - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) }) it('falls back to the committed URL once a navigation commits or fails', async () => { @@ -378,7 +377,7 @@ describe('browserManager', () => { await flushViewportOps() expect(guest.setUserAgent).toHaveBeenLastCalledWith(GUEST_ELECTRON_UA) - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) // A later preset must also resolve the committed, non-auth URL. debuggerSendCommand.mockClear() @@ -457,7 +456,7 @@ describe('browserManager', () => { expect(guest.setUserAgent).not.toHaveBeenCalled() expect(debuggerSendCommand).not.toHaveBeenCalledWith( 'Emulation.setUserAgentOverride', - expect.objectContaining({ userAgent: GUEST_CLEAN_UA }) + expect.objectContaining({ userAgent: GUEST_ELECTRON_UA }) ) }) @@ -517,7 +516,7 @@ describe('browserManager', () => { didFailLoad(null, -3, 'Aborted', 'https://accounts.google.com/redirected', true) await flushViewportOps() expect(guest.setUserAgent).not.toHaveBeenCalled() - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) }) it('preserves the auth identity when a viewport preset is cleared after a redirect', async () => { @@ -592,7 +591,7 @@ describe('browserManager', () => { didStartNavigation(null, 'https://example.com/', false, true) await flushViewportOps() - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) }) it('reapplies a preset when navigation starts during its final UA write', async () => { @@ -849,8 +848,7 @@ describe('browserManager', () => { expect(debuggerAttach).toHaveBeenCalledWith('1.3') expect(debuggerSendCommand).toHaveBeenCalled() - // Why: detaching would clear Page.addScriptToEvaluateOnNewDocument - // (anti-detection). Guard regression. + // Why: detaching would clear every standing CDP override (viewport, auth UA). Guard regression. expect((guest.debugger as { detach?: unknown }).detach ?? undefined).toBeUndefined() }) diff --git a/src/main/browser/browser-manager-viewport-test-fixtures.ts b/src/main/browser/browser-manager-viewport-test-fixtures.ts index 8d68977f62e..076524ce5d0 100644 --- a/src/main/browser/browser-manager-viewport-test-fixtures.ts +++ b/src/main/browser/browser-manager-viewport-test-fixtures.ts @@ -3,8 +3,6 @@ import type { BrowserManagerMocks } from './browser-manager-test-harness' export const GUEST_ELECTRON_UA = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) orca/1.0.0 Chrome/134.0.0.0 Electron/30.0.0 Safari/537.36' -export const GUEST_CLEAN_UA = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/134.0.0.0 Safari/537.36' // Why: viewport UA writes are queued on the per-tab chain, so draining it takes more than one // microtask hop; loop until the chain is empty rather than guessing a tick count. @@ -53,7 +51,9 @@ export function createViewportGuestFactory( debugger: { isAttached: debuggerIsAttached, attach: debuggerAttach, - sendCommand: debuggerSendCommand + sendCommand: debuggerSendCommand, + on: vi.fn(), + off: vi.fn() } } return { diff --git a/src/main/browser/browser-manager-viewport.ts b/src/main/browser/browser-manager-viewport.ts index 1e79e760942..ce31dbe37e1 100644 --- a/src/main/browser/browser-manager-viewport.ts +++ b/src/main/browser/browser-manager-viewport.ts @@ -30,7 +30,7 @@ export abstract class BrowserManagerViewport extends BrowserManagerDownloadLifec return true } - // Why: emulate viewport via CDP; never detach the debugger here or per-guest overrides (addScriptToEvaluateOnNewDocument) are cleared. + // Why: emulate viewport via CDP; never detach the debugger here or the agent bridge's per-guest state is cleared. async setViewportOverride( browserTabId: string, override: BrowserViewportOverride | null diff --git a/src/main/browser/browser-session-partition-policies.test.ts b/src/main/browser/browser-session-partition-policies.test.ts index 78ce34d95fd..952c199536c 100644 --- a/src/main/browser/browser-session-partition-policies.test.ts +++ b/src/main/browser/browser-session-partition-policies.test.ts @@ -82,8 +82,7 @@ vi.mock('./browser-media-access', () => ({ requestSystemMediaAccess: async () => false })) vi.mock('./browser-session-ua', () => ({ - cleanElectronUserAgent: (userAgent: string) => userAgent, - setupClientHintsOverride: vi.fn() + setupGoogleAuthUserAgentOverride: vi.fn() })) vi.mock('./browser-session-user-agent-mode', () => ({ setBrowserSessionUserAgentMode: vi.fn() diff --git a/src/main/browser/browser-session-partition-policies.ts b/src/main/browser/browser-session-partition-policies.ts index 9f25d8840a2..ec3c68fb45e 100644 --- a/src/main/browser/browser-session-partition-policies.ts +++ b/src/main/browser/browser-session-partition-policies.ts @@ -9,7 +9,7 @@ import { } from './browser-session-proxy' import { hasSystemMediaAccess, requestSystemMediaAccess } from './browser-media-access' import { isAutoGrantedBrowserSessionPermission } from './browser-session-permission-policy' -import { cleanElectronUserAgent, setupClientHintsOverride } from './browser-session-ua' +import { setupGoogleAuthUserAgentOverride } from './browser-session-ua' import { setBrowserSessionUserAgentMode } from './browser-session-user-agent-mode' import { allowsBrowserWebAuthnPermission, @@ -92,10 +92,8 @@ export function installBrowserSessionPartitionPolicies( } browserManager.installCertificateRequestGuard(sess) - if (profile.userAgentMode !== 'native' && typeof sess.getUserAgent === 'function') { - const cleanUA = cleanElectronUserAgent(sess.getUserAgent()) - sess.setUserAgent(cleanUA) - setupClientHintsOverride(sess, cleanUA) + if (profile.userAgentMode !== 'native') { + setupGoogleAuthUserAgentOverride(sess) } if (options?.permissions === 'deny') { sess.setPermissionRequestHandler((_webContents, _permission, callback) => callback(false)) @@ -191,11 +189,7 @@ export function applyBrowserSessionUserAgentModes(profiles: BrowserSessionProfil if (profile.userAgentMode === 'native') { continue } - - // Why: the default Electron UA leaks "Electron/X.X.X" + app name, which trips Cloudflare Turnstile. - const cleanUA = cleanElectronUserAgent(sess.getUserAgent()) - sess.setUserAgent(cleanUA) - setupClientHintsOverride(sess, cleanUA) + setupGoogleAuthUserAgentOverride(sess) } catch { /* session not available yet (e.g. unit tests or pre-ready) */ } diff --git a/src/main/browser/browser-session-partition-proxy-install.test.ts b/src/main/browser/browser-session-partition-proxy-install.test.ts index 0841ed94785..4beeaab04c9 100644 --- a/src/main/browser/browser-session-partition-proxy-install.test.ts +++ b/src/main/browser/browser-session-partition-proxy-install.test.ts @@ -44,8 +44,7 @@ vi.mock('./browser-media-access', () => ({ requestSystemMediaAccess: vi.fn(async () => false) })) vi.mock('./browser-session-ua', () => ({ - cleanElectronUserAgent: vi.fn((ua: string) => ua), - setupClientHintsOverride: vi.fn() + setupGoogleAuthUserAgentOverride: vi.fn() })) vi.mock('./browser-session-user-agent-mode', () => ({ setBrowserSessionUserAgentMode: vi.fn(), diff --git a/src/main/browser/browser-session-registry.persistence.test.ts b/src/main/browser/browser-session-registry.persistence.test.ts index fdba71e16d6..67653c0c76c 100644 --- a/src/main/browser/browser-session-registry.persistence.test.ts +++ b/src/main/browser/browser-session-registry.persistence.test.ts @@ -27,7 +27,7 @@ function installModuleMocks( copyFailures = new Set() ): { sessionFromPartitionMock: ReturnType - setupClientHintsOverrideMock: ReturnType + setupGoogleAuthUserAgentOverrideMock: ReturnType browserManagerHandleGuestWillDownloadMock: ReturnType browserManagerNotifyPermissionDeniedMock: ReturnType requestSystemMediaAccessMock: ReturnType @@ -36,6 +36,7 @@ function installModuleMocks( partition, setUserAgent: vi.fn(), getUserAgent: vi.fn(() => 'Mozilla/5.0 Electron/31 Orca'), + webRequest: { onBeforeSendHeaders: vi.fn() }, setPermissionRequestHandler: vi.fn(), setPermissionCheckHandler: vi.fn(), setDevicePermissionHandler: vi.fn(), @@ -45,7 +46,7 @@ function installModuleMocks( clearStorageData: vi.fn().mockResolvedValue(undefined), clearCache: vi.fn().mockResolvedValue(undefined) })) - const setupClientHintsOverrideMock = vi.fn() + const setupGoogleAuthUserAgentOverrideMock = vi.fn() const browserManagerHandleGuestWillDownloadMock = vi.fn() const browserManagerNotifyPermissionDeniedMock = vi.fn() const requestSystemMediaAccessMock = vi.fn().mockResolvedValue(true) @@ -119,8 +120,7 @@ function installModuleMocks( requestSystemMediaAccess: requestSystemMediaAccessMock })) vi.doMock('./browser-session-ua', () => ({ - cleanElectronUserAgent: vi.fn((ua: string) => ua.replace(/\s*Electron\/\S+/, '')), - setupClientHintsOverride: setupClientHintsOverrideMock + setupGoogleAuthUserAgentOverride: setupGoogleAuthUserAgentOverrideMock })) // This suite models replay with an in-memory filesystem. The real file-backed SQLite merge has // dedicated coverage; these fixtures are legacy unmarked images and keep the copy path. @@ -149,7 +149,7 @@ function installModuleMocks( return { sessionFromPartitionMock, - setupClientHintsOverrideMock, + setupGoogleAuthUserAgentOverrideMock, browserManagerHandleGuestWillDownloadMock, browserManagerNotifyPermissionDeniedMock, requestSystemMediaAccessMock @@ -234,21 +234,24 @@ describe('BrowserSessionRegistry persistence', () => { }) }) - it('keeps UA cleaning as the fallback for profiles without an override', async () => { + // Why: the stock Electron UA is what clears Cloudflare; only the Google auth switch installs. + it('keeps the stock UA and installs the Google auth switch for profiles without an override', async () => { const fsState = createFsState() - const { sessionFromPartitionMock, setupClientHintsOverrideMock } = installModuleMocks(fsState) + const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = + installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') await browserSessionRegistry.createProfile('isolated', 'Default identity') const profileSession = sessionFromPartitionMock.mock.results.at(-1)?.value - expect(profileSession.setUserAgent).toHaveBeenCalledWith('Mozilla/5.0 Orca') - expect(setupClientHintsOverrideMock).toHaveBeenCalledWith(profileSession, 'Mozilla/5.0 Orca') + expect(profileSession.setUserAgent).not.toHaveBeenCalled() + expect(setupGoogleAuthUserAgentOverrideMock).toHaveBeenCalledWith(profileSession) }) it('leaves UA and client hints untouched for native-mode profiles', async () => { const fsState = createFsState() - const { sessionFromPartitionMock, setupClientHintsOverrideMock } = installModuleMocks(fsState) + const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = + installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') await browserSessionRegistry.createProfile('isolated', 'Google', { userAgentMode: 'native' }) @@ -256,7 +259,7 @@ describe('BrowserSessionRegistry persistence', () => { const profileSession = sessionFromPartitionMock.mock.results.at(-1)?.value const { getBrowserSessionUserAgentMode } = await import('./browser-session-user-agent-mode') expect(profileSession.setUserAgent).not.toHaveBeenCalled() - expect(setupClientHintsOverrideMock).not.toHaveBeenCalled() + expect(setupGoogleAuthUserAgentOverrideMock).not.toHaveBeenCalled() expect(getBrowserSessionUserAgentMode(profileSession as never)).toBe('native') }) @@ -379,7 +382,7 @@ describe('BrowserSessionRegistry persistence', () => { // Why: imports before Aug 2026 persisted a synthesized source-browser UA // (fork imports as a broken Chrome/1.x, Chrome imports as a valid version). // Neither may ever be applied again — the engine-derived UA is the only one. - it('ignores legacy persisted UAs, valid or broken, and applies the engine UA', async () => { + it('ignores legacy persisted UAs, valid or broken, and keeps the engine UA', async () => { const importedPartition = 'persist:orca-browser-session-11111111-1111-4111-8111-111111111111' const brokenUa = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/1.158.1 Safari/537.36' @@ -405,7 +408,8 @@ describe('BrowserSessionRegistry persistence', () => { ] }) - const { sessionFromPartitionMock, setupClientHintsOverrideMock } = installModuleMocks(fsState) + const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = + installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') browserSessionRegistry.initializeBrowserSessionsFromPersistedState() @@ -413,16 +417,9 @@ describe('BrowserSessionRegistry persistence', () => { const appliedUas = sessionFromPartitionMock.mock.results.flatMap((r) => r.value.setUserAgent.mock.calls.map((c: unknown[]) => c[0]) ) - expect(appliedUas).not.toContain(brokenUa) - expect(appliedUas).not.toContain(validUa) - // Why: every non-native profile falls to Orca's own cleaned engine UA. - expect(appliedUas.length).toBeGreaterThan(0) - expect(appliedUas.every((ua) => ua === 'Mozilla/5.0 Orca')).toBe(true) - expect( - setupClientHintsOverrideMock.mock.calls.every( - (c: unknown[]) => c[1] !== brokenUa && c[1] !== validUa - ) - ).toBe(true) + // Why: no persisted UA is ever written back; every profile keeps the engine's stock UA. + expect(appliedUas).toEqual([]) + expect(setupGoogleAuthUserAgentOverrideMock).toHaveBeenCalled() }) it('never applies a legacy persisted UA to a native-mode profile', async () => { @@ -487,7 +484,8 @@ describe('BrowserSessionRegistry persistence', () => { ] }) - const { sessionFromPartitionMock, setupClientHintsOverrideMock } = installModuleMocks(fsState) + const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = + installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') browserSessionRegistry.initializeBrowserSessionsFromPersistedState() @@ -498,7 +496,7 @@ describe('BrowserSessionRegistry persistence', () => { expect(importedSessions.length).toBeGreaterThan(0) expect(importedSessions.every((sess) => sess.setUserAgent.mock.calls.length === 0)).toBe(true) expect( - setupClientHintsOverrideMock.mock.calls.some( + setupGoogleAuthUserAgentOverrideMock.mock.calls.some( ([sess]) => (sess as { partition?: string }).partition === importedPartition ) ).toBe(false) diff --git a/src/main/browser/browser-session-registry.test.ts b/src/main/browser/browser-session-registry.test.ts index d5111483fcf..ae81ffda4e2 100644 --- a/src/main/browser/browser-session-registry.test.ts +++ b/src/main/browser/browser-session-registry.test.ts @@ -33,7 +33,7 @@ vi.mock('./browser-manager', () => ({ import { browserSessionRegistry } from './browser-session-registry' import { googleAuthUserAgent } from './browser-google-auth-ua' -import { setupClientHintsOverride } from './browser-session-ua' +import { setupGoogleAuthUserAgentOverride } from './browser-session-ua' import { setBrowserNetworkProxySettingsResolver } from './browser-session-proxy' import { handleElectronProxyLogin } from '../network/electron-proxy-credentials' import { applyProxySettingsToSession } from '../network/proxy-settings' @@ -54,6 +54,7 @@ describe('BrowserSessionRegistry', () => { askForMediaAccessMock.mockResolvedValue(true) getMediaAccessStatusMock.mockReturnValue('granted') sessionFromPartitionMock.mockReturnValue({ + webRequest: { onBeforeSendHeaders: vi.fn() }, setPermissionRequestHandler: vi.fn(), setPermissionCheckHandler: vi.fn(), setDevicePermissionHandler: vi.fn(), @@ -528,80 +529,46 @@ describe('BrowserSessionRegistry', () => { }) }) - describe('setupClientHintsOverride', () => { - it('overrides sec-ch-ua headers for Edge UA', () => { + describe('setupGoogleAuthUserAgentOverride', () => { + const STOCK_UA = + 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) orca/1.0.0 Chrome/147.0.6890.3 Electron/43.0.0 Safari/537.36' + + function install(): (details: unknown, callback: ReturnType) => void { const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const edgeUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36 Edg/147.0.3210.5' - - setupClientHintsOverride(mockSess, edgeUa) - + setupGoogleAuthUserAgentOverride({ webRequest: { onBeforeSendHeaders } } as never) expect(onBeforeSendHeaders).toHaveBeenCalledWith( { urls: ['https://*/*'] }, expect.any(Function) ) + return onBeforeSendHeaders.mock.calls[0][1] + } + // Why: the Electron token is what clears Cloudflare Turnstile; a Chrome-shaped UA with no + // client hints is what it rejects, so ordinary hosts must see the session's UA untouched. + it('leaves the stock Electron UA and its client hints alone off the auth hosts', () => { + const listener = install() const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] listener( - { requestHeaders: { 'sec-ch-ua': 'old', 'sec-ch-ua-full-version-list': 'old' } }, + { + url: 'https://example.com/api', + requestHeaders: { 'User-Agent': STOCK_UA, 'sec-ch-ua': 'old', Cookie: 'abc=123' } + }, callback ) const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['sec-ch-ua']).toContain('Microsoft Edge') - expect(modified['sec-ch-ua']).toContain('"147"') - expect(modified['sec-ch-ua-full-version-list']).toContain('147.0.3210.5') - }) - - it('overrides sec-ch-ua headers for Chrome UA', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const chromeUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - - setupClientHintsOverride(mockSess, chromeUa) - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener({ requestHeaders: { 'sec-ch-ua': 'old' } }, callback) - const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['sec-ch-ua']).toContain('Google Chrome') - expect(modified['sec-ch-ua']).not.toContain('Microsoft Edge') - }) - - it('registers handler even for non-Chrome UA but leaves sec-ch-ua untouched off auth hosts', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - - // Why: the Google-auth Firefox switch must install regardless of the base UA. - setupClientHintsOverride(mockSess, 'Mozilla/5.0 (compatible; MSIE 10.0)') - - expect(onBeforeSendHeaders).toHaveBeenCalledWith( - { urls: ['https://*/*'] }, - expect.any(Function) - ) - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener({ url: 'https://example.com/', requestHeaders: { 'sec-ch-ua': 'old' } }, callback) - expect(callback.mock.calls[0][0].requestHeaders['sec-ch-ua']).toBe('old') + expect(modified['User-Agent']).toBe(STOCK_UA) + expect(modified['sec-ch-ua']).toBe('old') + expect(modified.Cookie).toBe('abc=123') }) it('presents a Firefox UA and strips client hints on Google auth hosts', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - setupClientHintsOverride( - mockSess, - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - ) - + const listener = install() const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] listener( { url: 'https://accounts.google.com/v3/signin/identifier', requestHeaders: { - 'User-Agent': 'Chrome/147', + 'User-Agent': STOCK_UA, 'sec-ch-ua': 'old', 'sec-ch-ua-full-version-list': 'old', 'sec-ch-ua-platform': '"macOS"' @@ -610,7 +577,7 @@ describe('BrowserSessionRegistry', () => { callback ) const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['User-Agent']).toMatch(/Firefox\/\d/) + expect(modified['User-Agent']).toBe(googleAuthUserAgent()) expect(modified['User-Agent']).not.toContain('Chrome') expect(modified['sec-ch-ua']).toBeUndefined() expect(modified['sec-ch-ua-full-version-list']).toBeUndefined() @@ -618,15 +585,8 @@ describe('BrowserSessionRegistry', () => { }) it('strips client hints on a cross-host request that carries the Firefox auth UA', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - setupClientHintsOverride( - mockSess, - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - ) - + const listener = install() const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] // Subresource/XHR to a non-auth Google host while the auth document is on // screen: the WebContents Firefox UA leaks onto the request header. listener( @@ -651,99 +611,19 @@ describe('BrowserSessionRegistry', () => { expect(modified['sec-ch-ua-mobile']).toBeUndefined() }) - it('keeps the clean Chrome identity on cross-host requests that carry the Chrome UA', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const chromeUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - setupClientHintsOverride(mockSess, chromeUa) - + it('keeps the session identity on Google app subdomains (not auth hosts)', () => { + const listener = install() const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - // Regression guard: non-Google sites (Cloudflare) must keep Chrome hints. listener( { - url: 'https://example.com/api', - requestHeaders: { 'User-Agent': chromeUa, 'sec-ch-ua': 'old' } - }, - callback - ) - expect(callback.mock.calls[0][0].requestHeaders['sec-ch-ua']).toContain('Google Chrome') - }) - - it('does not strip hints for the Firefox UA when googleAuthOverride is disabled', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const chromeUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - setupClientHintsOverride(mockSess, chromeUa, { googleAuthOverride: false }) - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener( - { - url: 'https://play.google.com/log', - requestHeaders: { 'User-Agent': googleAuthUserAgent(), 'sec-ch-ua': 'old' } - }, - callback - ) - // Imported-native profiles never install the Firefox switch, so the strip - // branch stays inert and hints are aligned to Chrome instead. - expect(callback.mock.calls[0][0].requestHeaders['sec-ch-ua']).toContain('Google Chrome') - }) - - it('keeps Chrome client hints on Google app subdomains (not auth hosts)', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - setupClientHintsOverride( - mockSess, - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - ) - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener( - { url: 'https://myaccount.google.com/', requestHeaders: { 'sec-ch-ua': 'old' } }, - callback - ) - expect(callback.mock.calls[0][0].requestHeaders['sec-ch-ua']).toContain('Google Chrome') - }) - - it('keeps an imported native UA on auth hosts while aligning its Chrome hints', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const importedUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - setupClientHintsOverride(mockSess, importedUa, { googleAuthOverride: false }) - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener( - { - url: 'https://accounts.google.com/v3/signin/identifier', - requestHeaders: { 'User-Agent': importedUa, 'sec-ch-ua': 'old' } + url: 'https://myaccount.google.com/', + requestHeaders: { 'User-Agent': STOCK_UA, 'sec-ch-ua': 'old' } }, callback ) const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['User-Agent']).toBe(importedUa) - expect(modified['sec-ch-ua']).toContain('Google Chrome') - }) - - it('leaves non-Client-Hints headers unchanged', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - setupClientHintsOverride(mockSess, 'Mozilla/5.0 Chrome/147.0.0.0 Safari/537.36') - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener( - { requestHeaders: { Cookie: 'abc=123', 'sec-ch-ua': 'old', Accept: 'text/html' } }, - callback - ) - const modified = callback.mock.calls[0][0].requestHeaders - expect(modified.Cookie).toBe('abc=123') - expect(modified.Accept).toBe('text/html') + expect(modified['User-Agent']).toBe(STOCK_UA) + expect(modified['sec-ch-ua']).toBe('old') }) }) }) diff --git a/src/main/browser/browser-session-ua-wire-identity.electron.test.ts b/src/main/browser/browser-session-ua-wire-identity.electron.test.ts new file mode 100644 index 00000000000..4e0719b2745 --- /dev/null +++ b/src/main/browser/browser-session-ua-wire-identity.electron.test.ts @@ -0,0 +1,180 @@ +import { spawnSync } from 'node:child_process' +import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { createRequire } from 'node:module' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterAll, describe, expect, it } from 'vitest' +import { build as buildVite } from 'vite' + +// Why this runs a real Electron: Cloudflare Turnstile rejects a Chrome-shaped UA that ships no +// client hints (error 600010) and clears a declared Electron client. The header layer is the +// only place that identity can be proven, and the vm-based unit tests cannot see Chromium's +// header emission at all. Every partition must therefore keep the stock Electron UA on the wire +// for ordinary hosts and present the Firefox identity on Google's sign-in hosts only. + +const electronBinary = createRequire(import.meta.url)('electron') as string +const fixtureRoots: string[] = [] + +afterAll(() => { + for (const root of fixtureRoots) { + rmSync(root, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }) + } +}) + +// Retry once when Electron startup times out before `ready`; keep later failures fatal. +const FIXTURE_LAUNCH_ATTEMPTS = 2 + +type CapturedRequest = { + url: string + userAgent: string | null + clientHints: string[] +} + +type FixtureResult = { + sessionUserAgent: string + navigatorUserAgent: string + requests: CapturedRequest[] +} + +function neverReachedElectronReady(fixtureResult: string): boolean { + try { + return (JSON.parse(fixtureResult) as { step?: string }).step === 'timed out after starting' + } catch { + return false + } +} + +function buildFixtureMain(modulePath: string, resultPath: string): string { + return ` +const { app, BrowserWindow, session } = require('electron') +const { writeFileSync } = require('node:fs') +const { setupGoogleAuthUserAgentOverride } = require(${JSON.stringify(modulePath)}) +const resultPath = ${JSON.stringify(resultPath)} +let currentStep = 'starting' +const mark = (step) => { + currentStep = step + writeFileSync(resultPath, JSON.stringify({ step })) +} + +async function run() { + const timeout = setTimeout(() => { + writeFileSync(resultPath, JSON.stringify({ step: 'timed out after ' + currentStep })) + app.exit(1) + }, 15000) + await app.whenReady() + mark('ready') + const partition = 'persist:wire-identity-test' + const sess = session.fromPartition(partition) + setupGoogleAuthUserAgentOverride(sess) + mark('auth switch installed') + + // Why: onSendHeaders reports the headers exactly as they leave the network stack, after the + // product's onBeforeSendHeaders listener has rewritten them. The requests must actually be + // dispatched for it to fire, so the session is pointed at a proxy that refuses every + // connection: nothing reaches the real hosts and every load fails fast. + await sess.setProxy({ proxyRules: 'http://127.0.0.1:9', proxyBypassRules: '<-loopback>' }) + const requests = [] + sess.webRequest.onSendHeaders({ urls: ['https://*/*'] }, (details) => { + const headers = details.requestHeaders || {} + const uaKey = Object.keys(headers).find((key) => key.toLowerCase() === 'user-agent') + requests.push({ + url: details.url, + userAgent: uaKey ? headers[uaKey] : null, + clientHints: Object.keys(headers) + .filter((key) => key.toLowerCase().startsWith('sec-ch-ua')) + .sort() + }) + }) + + const window = new BrowserWindow({ show: false, webPreferences: { partition } }) + mark('window created') + for (const url of ['https://example.com/', 'https://accounts.google.com/v3/signin/identifier']) { + await window.loadURL(url).catch(() => {}) + } + mark('navigations attempted') + const navigatorUserAgent = await window.webContents.executeJavaScript('navigator.userAgent') + clearTimeout(timeout) + writeFileSync(resultPath, JSON.stringify({ + sessionUserAgent: sess.getUserAgent(), + navigatorUserAgent, + requests + })) + window.destroy() + app.exit(0) +} + +run().catch((error) => { + writeFileSync(resultPath, JSON.stringify({ step: currentStep, error: String(error?.stack || error) })) + app.exit(1) +}) +` +} + +async function runFixture(): Promise { + const root = mkdtempSync(join(tmpdir(), 'orca-wire-identity-')) + fixtureRoots.push(root) + const modulePath = join(root, 'browser-session-ua.cjs') + const resultPath = join(root, 'result.json') + const fixturePath = join(root, 'main.cjs') + await buildVite({ + configFile: false, + logLevel: 'silent', + build: { + emptyOutDir: false, + lib: { + entry: join(process.cwd(), 'src/main/browser/browser-session-ua.ts'), + formats: ['cjs'], + fileName: () => 'browser-session-ua.cjs' + }, + outDir: root, + target: 'node20', + rollupOptions: { external: ['electron', /^node:/] } + } + }) + writeFileSync(fixturePath, buildFixtureMain(modulePath, resultPath)) + const { ELECTRON_RUN_AS_NODE: _electronRunAsNode, ...env } = process.env + const executable = process.platform === 'linux' ? 'xvfb-run' : electronBinary + for (let attempt = 1; ; attempt += 1) { + rmSync(resultPath, { force: true }) + // Why a fresh profile per attempt: a launch that never reached `ready` may have left the + // Chromium profile mid-initialization, and reusing it would bias the retry. + const electronArgs = [fixturePath, `--user-data-dir=${join(root, `profile-${attempt}`)}`] + const run = spawnSync( + executable, + process.platform === 'linux' + ? ['--auto-servernum', electronBinary, ...electronArgs, '--no-sandbox'] + : electronArgs, + { encoding: 'utf8', env, timeout: 60_000 } + ) + const fixtureResult = existsSync(resultPath) ? readFileSync(resultPath, 'utf8') : 'no result' + if (attempt < FIXTURE_LAUNCH_ATTEMPTS && neverReachedElectronReady(fixtureResult)) { + continue + } + expect(run.error).toBeUndefined() + expect(run.status, `${fixtureResult}\n${run.stdout}\n${run.stderr}`).toBe(0) + return JSON.parse(fixtureResult) as FixtureResult + } +} + +describe('browser session wire identity under Electron', () => { + it('sends the stock Electron UA to ordinary hosts and Firefox to Google auth hosts', async () => { + const result = await runFixture() + + // Presence precondition: the stock identity still carries the Electron token that the old + // Chrome-shaped rewrite stripped, so an identity check below cannot pass on an empty UA. + expect(result.sessionUserAgent).toMatch(/ Electron\/\d/) + + const ordinary = result.requests.find((request) => request.url === 'https://example.com/') + expect(ordinary, JSON.stringify(result.requests)).toBeDefined() + expect(ordinary?.userAgent).toBe(result.sessionUserAgent) + expect(result.navigatorUserAgent).toBe(result.sessionUserAgent) + + const auth = result.requests.find((request) => + request.url.startsWith('https://accounts.google.com/') + ) + expect(auth, JSON.stringify(result.requests)).toBeDefined() + expect(auth?.userAgent).toMatch(/Firefox\/\d/) + expect(auth?.userAgent).not.toContain('Chrome') + expect(auth?.clientHints).toEqual([]) + }) +}) diff --git a/src/main/browser/browser-session-ua.ts b/src/main/browser/browser-session-ua.ts index 96c55cf5ac9..1375ebc66f5 100644 --- a/src/main/browser/browser-session-ua.ts +++ b/src/main/browser/browser-session-ua.ts @@ -8,93 +8,28 @@ import { stripClientHints } from './browser-google-auth-ua' -// Why: Electron's default UA includes "Electron/X.X.X" and the app name -// (e.g. "orca/1.2.3"), which Cloudflare Turnstile and other bot detectors -// flag as non-human traffic. Strip those tokens so the webview's UA and -// sec-ch-ua Client Hints look like standard Chrome. -export function cleanElectronUserAgent(ua: string): string { - return ( - ua - .replace(/\s+Electron\/\S+/, '') - // Why: \S+ matches any non-whitespace token (e.g. "orca/1.3.8-rc.0") - // including pre-release semver strings that [\d.]+ would miss. - .replace(/(\)\s+)\S+\s+(Chrome\/)/, '$1$2') - ) -} - -// Why: Electron emits sec-ch-ua brands like "Not A(Brand" without a -// "Google Chrome" entry, which disagrees with the Chrome-shaped UA the session -// presents. Rewrite the hint headers to the brand set Chrome ships for the same -// engine version so the two surfaces tell one story. Also owns the Google -// auth-host Firefox switch, which must install even for a non-Chrome-shaped UA. -export function setupClientHintsOverride( - sess: Session, - ua: string, - options: { googleAuthOverride?: boolean } = {} -): void { - // Why: only Chrome-shaped base UAs carry sec-ch-ua hints to rewrite, but the - // Google-auth Firefox switch below must install regardless, so keep the hints - // optional rather than bailing out of the whole handler. - const chromeHints = buildChromeClientHints(ua) +// Why: the session keeps Electron's stock UA. Stripping the Electron/app tokens to look like +// plain Chrome is what Cloudflare Turnstile rejects (error 600010): a Chrome UA that ships no +// client hints reads as a spoof, while a declared Electron client clears the same challenge. +// This handler only owns the Google auth-host Firefox switch, which is a proven, host-scoped +// exception that must stay consistent across the header and every cross-host subresource. +export function setupGoogleAuthUserAgentOverride(sess: Session): void { const firefoxUa = googleAuthUserAgent() sess.webRequest.onBeforeSendHeaders({ urls: ['https://*/*'] }, (details, callback) => { const headers = details.requestHeaders - if (options.googleAuthOverride !== false && isGoogleAuthUrl(details.url)) { + if (isGoogleAuthUrl(details.url)) { // Why: present a Firefox identity on Google's sign-in hosts so the user logs // in inside the app and Google issues self-refreshing bound cookies. Strip // sec-ch-ua* because real Firefox sends none. setUserAgentHeader(headers, firefoxUa) stripClientHints(headers) - callback({ requestHeaders: headers }) - return - } - if (options.googleAuthOverride !== false && currentUserAgent(headers) === firefoxUa) { - // Why: while the auth document is on screen the WebContents UA is Firefox, - // so its cross-host subresource/XHR requests (gstatic, play.google.com, the - // sign-in challenge endpoints) reach here carrying the Firefox UA yet still - // bearing Chromium client hints. Rewriting those to Chrome pairs a Firefox - // UA with Chrome hints — a sharper cross-host identity tell than either - // alone, which can stall Google's password-submit challenge. Real Firefox - // sends no client hints, so strip them to keep one identity for the flow. + } else if (currentUserAgent(headers) === firefoxUa) { + // Why: while the auth document is on screen the WebContents UA is Firefox, so its + // cross-host subresource/XHR requests carry the Firefox UA yet still bear Chromium + // client hints — a sharper cross-host identity tell than either alone. stripClientHints(headers) - callback({ requestHeaders: headers }) - return - } - if (chromeHints) { - for (const key of Object.keys(headers)) { - const lower = key.toLowerCase() - if (lower === 'sec-ch-ua') { - headers[key] = chromeHints.secChUa - } else if (lower === 'sec-ch-ua-full-version-list') { - headers[key] = chromeHints.secChUaFull - } - } } callback({ requestHeaders: headers }) }) } - -function buildChromeClientHints(ua: string): { secChUa: string; secChUaFull: string } | null { - const chromeMatch = ua.match(/Chrome\/([\d.]+)/) - if (!chromeMatch) { - return null - } - const fullChromeVersion = chromeMatch[1] - const majorVersion = fullChromeVersion.split('.')[0] - - let brand = 'Google Chrome' - let brandFullVersion = fullChromeVersion - - const edgeMatch = ua.match(/Edg\/([\d.]+)/) - if (edgeMatch) { - brand = 'Microsoft Edge' - brandFullVersion = edgeMatch[1] - } - const brandMajor = brandFullVersion.split('.')[0] - - return { - secChUa: `"${brand}";v="${brandMajor}", "Chromium";v="${majorVersion}", "Not/A)Brand";v="24"`, - secChUaFull: `"${brand}";v="${brandFullVersion}", "Chromium";v="${fullChromeVersion}", "Not/A)Brand";v="24.0.0.0"` - } -} diff --git a/src/main/browser/browser-viewport-user-agent.ts b/src/main/browser/browser-viewport-user-agent.ts index 7dedf8d8a6e..b8159a44c2a 100644 --- a/src/main/browser/browser-viewport-user-agent.ts +++ b/src/main/browser/browser-viewport-user-agent.ts @@ -23,7 +23,7 @@ export type ViewportUserAgentOverride = { } // Why: responsive sites UA-sniff; this is Chrome DevTools' default iPhone UA template with the real -// Chrome major spliced in to keep sec-ch-ua consistent (see setupClientHintsOverride). +// Chrome major spliced in so the userAgentMetadata brands below agree with it. function buildMobileUserAgent(chromeMajor: string): string { return `Mozilla/5.0 (iPhone; CPU iPhone OS 17_0 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) CriOS/${chromeMajor}.0.0.0 Mobile/15E148 Safari/604.1` } @@ -44,7 +44,7 @@ export function buildViewportUserAgentOverride(args: { return { userAgent: googleAuthUserAgent() } } if (!args.mobile) { - // Why: desktop presets still need the clean (non-Electron) UA so Cloudflare/Turnstile don't flag the session. + // Why: desktop presets republish the session's own identity unchanged. return { userAgent: args.baseUserAgent } } const chromeMajor = extractChromeMajor(args.baseUserAgent) diff --git a/src/main/browser/browser-webauthn-profile-delete.test.ts b/src/main/browser/browser-webauthn-profile-delete.test.ts index 9a5e129885f..3c471fe2dc2 100644 --- a/src/main/browser/browser-webauthn-profile-delete.test.ts +++ b/src/main/browser/browser-webauthn-profile-delete.test.ts @@ -48,7 +48,8 @@ function mockSession(): MockSession { setDevicePermissionHandler: vi.fn(), setDisplayMediaRequestHandler: vi.fn(), setPermissionCheckHandler: vi.fn(), - setPermissionRequestHandler: vi.fn() + setPermissionRequestHandler: vi.fn(), + webRequest: { onBeforeSendHeaders: vi.fn() } }) as unknown as MockSession } diff --git a/src/main/browser/cdp-debugger-channel.ts b/src/main/browser/cdp-debugger-channel.ts index 352bc741ef0..18b819a1ce9 100644 --- a/src/main/browser/cdp-debugger-channel.ts +++ b/src/main/browser/cdp-debugger-channel.ts @@ -1,6 +1,5 @@ import { WebSocket } from 'ws' import type { WebContents } from 'electron' -import { ANTI_DETECTION_SCRIPT } from './anti-detection' import { acquireElectronDebugger, type ElectronDebuggerLease } from './electron-debugger-lease' import type { CdpClientResponseWriter } from './cdp-client-response-writer' import type { CdpSyntheticSessionRegistry } from './cdp-synthetic-session-registry' @@ -34,14 +33,8 @@ export class CdpDebuggerChannel { } this.attached = true - // Why: attaching the CDP debugger sets navigator.webdriver = true and - // exposes other automation signals that Cloudflare Turnstile checks. - // Inject before any page loads so challenges succeed. try { await this.webContents.debugger.sendCommand('Page.enable', {}) - await this.webContents.debugger.sendCommand('Page.addScriptToEvaluateOnNewDocument', { - source: ANTI_DETECTION_SCRIPT - }) } catch { /* best-effort — page domain may not be ready yet */ } diff --git a/src/main/browser/cdp-debugger-events.ts b/src/main/browser/cdp-debugger-events.ts index 023d648a8ca..59d30105910 100644 --- a/src/main/browser/cdp-debugger-events.ts +++ b/src/main/browser/cdp-debugger-events.ts @@ -32,9 +32,11 @@ export function createCdpDebuggerMessageListener( | undefined if (p?.sessionId && p.targetInfo?.type === 'iframe' && p.targetInfo.targetId) { state.iframeSessions.set(p.targetInfo.targetId, p.sessionId) + // Why: no Runtime.enable here. Cross-origin iframes include challenge widgets + // (Cloudflare Turnstile), and the Runtime domain's console/Error.stack serialization + // is the CDP tell they detect; nothing reads iframe Runtime events anyway. guest.debugger.sendCommand('DOM.enable', {}, p.sessionId).catch(() => {}) guest.debugger.sendCommand('Accessibility.enable', {}, p.sessionId).catch(() => {}) - guest.debugger.sendCommand('Runtime.enable', {}, p.sessionId).catch(() => {}) } } if (method === 'Target.detachedFromTarget') { diff --git a/src/main/browser/cdp-debugger-lifecycle.ts b/src/main/browser/cdp-debugger-lifecycle.ts index f969f113d59..225eb689dba 100644 --- a/src/main/browser/cdp-debugger-lifecycle.ts +++ b/src/main/browser/cdp-debugger-lifecycle.ts @@ -1,5 +1,4 @@ import type { WebContents } from 'electron' -import { ANTI_DETECTION_SCRIPT } from './anti-detection' import { BrowserError } from './browser-error' import type { CdpTabState } from './cdp-auxiliary-commands' import type { CdpCommandSender } from './snapshot-engine' @@ -62,11 +61,6 @@ export class CdpDebuggerLifecycle { flatten: true }) - // Why: CDP attach exposes automation signals (navigator.webdriver) that Cloudflare checks; override per new document. - await sender('Page.addScriptToEvaluateOnNewDocument', { - source: ANTI_DETECTION_SCRIPT - }) - // Why: only remove this bridge's listeners; screencast/proxy sessions share the debugger and own their teardown. this.removeDebuggerListeners(guest, state) diff --git a/src/main/browser/cdp-ws-proxy-focus-replay.test.ts b/src/main/browser/cdp-ws-proxy-focus-replay.test.ts index 71b00cd9ff6..8404614ef8d 100644 --- a/src/main/browser/cdp-ws-proxy-focus-replay.test.ts +++ b/src/main/browser/cdp-ws-proxy-focus-replay.test.ts @@ -52,7 +52,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(mock.webContents.focus).toHaveBeenCalledTimes(1) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 99 }], ['DOM.focus', { backendNodeId: 99 }], ['Input.insertText', { text: 'hello' }] @@ -80,7 +79,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(insertResponse.result).toEqual({}) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 123 }, 'oopif-session-123'], ['DOM.focus', { backendNodeId: 123 }, 'oopif-session-123'], ['Input.insertText', { text: 'frame text' }, 'oopif-session-123'] @@ -112,7 +110,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(mock.webContents.focus).toHaveBeenCalledTimes(1) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 44 }], ['Runtime.callFunctionOn', { functionDeclaration: '() => document.activeElement?.id' }], ['Input.insertText', { text: 'after eval' }] @@ -155,7 +152,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(mock.webContents.focus).toHaveBeenCalledTimes(1) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 55 }], ['Input.insertText', { text: 'fallback' }] ]) @@ -197,7 +193,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(mock.webContents.focus).toHaveBeenCalledTimes(1) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 77 }], ['DOM.focus', { backendNodeId: 77 }] ]) @@ -242,7 +237,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(insertResponse?.result).toEqual({}) expect(getSendCommandMethods(mock)).toEqual([ 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', 'DOM.focus', 'DOM.focus', 'Input.insertText' @@ -270,12 +264,7 @@ describe('CdpWsProxy DOM.focus replay', () => { // Why: both Page.bringToFront and Input.insertText natively call focus(), // independent of the (now-cleared) DOM.focus replay. expect(mock.webContents.focus).toHaveBeenCalledTimes(2) - expect(getSendCommandMethods(mock)).toEqual([ - 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', - 'DOM.focus', - 'Input.insertText' - ]) + expect(getSendCommandMethods(mock)).toEqual(['Page.enable', 'DOM.focus', 'Input.insertText']) client.close() }) @@ -298,7 +287,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(insertResponse.result).toEqual({}) expect(getSendCommandMethods(mock)).toEqual([ 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', 'DOM.focus', 'Page.captureScreenshot', 'Input.insertText' @@ -322,12 +310,7 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(insertResponse.id).toBe(34) expect(insertResponse.result).toEqual({}) - expect(getSendCommandMethods(mock)).toEqual([ - 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', - 'DOM.focus', - 'Input.insertText' - ]) + expect(getSendCommandMethods(mock)).toEqual(['Page.enable', 'DOM.focus', 'Input.insertText']) second.close() }) diff --git a/src/main/browser/cdp-ws-proxy.test.ts b/src/main/browser/cdp-ws-proxy.test.ts index c5f098ffa91..d5a2d14a05b 100644 --- a/src/main/browser/cdp-ws-proxy.test.ts +++ b/src/main/browser/cdp-ws-proxy.test.ts @@ -400,11 +400,7 @@ describe('CdpWsProxy', () => { }) expect(mock.webContents.focus).toHaveBeenCalledTimes(1) - expect(getSendCommandMethods(mock)).toEqual([ - 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', - 'Input.insertText' - ]) + expect(getSendCommandMethods(mock)).toEqual(['Page.enable', 'Input.insertText']) client.close() }) @@ -421,7 +417,6 @@ describe('CdpWsProxy', () => { expect(response.result).toEqual({}) expect(getSendCommandMethods(mock)).toEqual([ 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', 'Network.enable', 'Page.enable', 'Page.setLifecycleEventsEnabled', @@ -442,7 +437,6 @@ describe('CdpWsProxy', () => { expect(response.result).toEqual({}) expect(getSendCommandMethods(mock)).toEqual([ 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', 'Network.enable', 'Page.enable', 'Page.setLifecycleEventsEnabled' @@ -462,7 +456,7 @@ describe('CdpWsProxy', () => { sessionId: 'iframe-session-123' }) - expect(getSendCommandCalls(mock).slice(2)).toEqual([ + expect(getSendCommandCalls(mock).slice(1)).toEqual([ ['Network.enable', {}, 'iframe-session-123'], ['Page.enable', {}, 'iframe-session-123'], ['Page.setLifecycleEventsEnabled', { enabled: true }, 'iframe-session-123'], @@ -481,7 +475,7 @@ describe('CdpWsProxy', () => { sessionId: 'iframe-session-123' }) - expect(getSendCommandCalls(mock).slice(2)).toEqual([ + expect(getSendCommandCalls(mock).slice(1)).toEqual([ ['Network.enable', {}, 'iframe-session-123'], ['Page.enable', {}, 'iframe-session-123'], ['Page.setLifecycleEventsEnabled', { enabled: true }, 'iframe-session-123'], @@ -562,11 +556,7 @@ describe('CdpWsProxy', () => { expect(response.id).toBe(13) expect(response.result).toEqual({}) - expect(getSendCommandMethods(mock)).toEqual([ - 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', - 'Runtime.evaluate' - ]) + expect(getSendCommandMethods(mock)).toEqual(['Page.enable', 'Runtime.evaluate']) client.close() }) diff --git a/src/main/claude-accounts/claude-structured-auth-policy.test.ts b/src/main/claude-accounts/claude-structured-auth-policy.test.ts new file mode 100644 index 00000000000..a2d7c7d5da8 --- /dev/null +++ b/src/main/claude-accounts/claude-structured-auth-policy.test.ts @@ -0,0 +1,168 @@ +import { describe, expect, it } from 'vitest' +import type { GlobalSettings } from '../../shared/global-settings-types' +import type { ClaudeManagedAccount } from '../../shared/managed-account-types' +import { + CLAUDE_AUTH_ENV_VARS, + hasClaudeAuthEnvConflict, + shouldStripClaudeAuthEnvForAccount +} from './environment' +import { + normalizeTuiAgentEnvRecord, + resolveTuiAgentLaunchEnv +} from '../../shared/tui-agent-launch-defaults' +import { claudeStructuredAuthPolicyForSettings } from './claude-structured-auth-policy' + +const HOST_ACCOUNT = { id: 'host-a', managedAuthRuntime: 'host' } as ClaudeManagedAccount +const WSL_ACCOUNT = { id: 'wsl-b', managedAuthRuntime: 'wsl' } as ClaudeManagedAccount +const LEGACY_ACCOUNT = { id: 'legacy-c' } as ClaudeManagedAccount + +function settings( + overrides: Partial< + Pick< + GlobalSettings, + | 'claudeManagedAccounts' + | 'activeClaudeManagedAccountId' + | 'activeClaudeManagedAccountIdsByRuntime' + > + > +): Parameters[0] { + return { + claudeManagedAccounts: [HOST_ACCOUNT, WSL_ACCOUNT, LEGACY_ACCOUNT], + activeClaudeManagedAccountId: null, + ...overrides + } as Parameters[0] +} + +// The predicate now backs BOTH transports (runtime-auth-preparation.ts and the +// structured wiring), so it needs a test of its own: forcing it to a constant used +// to leave ~1000 tests green. +describe('shouldStripClaudeAuthEnvForAccount', () => { + it('does not strip when no managed account is selected', () => { + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT], null)).toBe(false) + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT], undefined)).toBe(false) + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT], '')).toBe(false) + }) + + it('strips for a host-managed account', () => { + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT, WSL_ACCOUNT], 'host-a')).toBe(true) + }) + + it('strips for an account with no explicit runtime (the legacy host shape)', () => { + expect(shouldStripClaudeAuthEnvForAccount([LEGACY_ACCOUNT], 'legacy-c')).toBe(true) + }) + + it('does not strip for a WSL-managed account, matching runtime-auth-preparation', () => { + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT, WSL_ACCOUNT], 'wsl-b')).toBe(false) + }) + + it('strips for a selected id no account list explains', () => { + // Fail-safe: an id we cannot resolve is treated as a pinned account, never as + // "no account", so an unreadable settings blob cannot open the strip. + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT], 'deleted-d')).toBe(true) + expect(shouldStripClaudeAuthEnvForAccount(undefined, 'deleted-d')).toBe(true) + expect(shouldStripClaudeAuthEnvForAccount([], 'deleted-d')).toBe(true) + }) +}) + +describe('claudeStructuredAuthPolicyForSettings', () => { + it('reads the host runtime selection, not the legacy flat field alone', () => { + expect( + claudeStructuredAuthPolicyForSettings( + settings({ + activeClaudeManagedAccountId: 'host-a', + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: {} } + }) + ) + ).toEqual({ stripAuthEnv: true }) + }) + + it('strips when a host account is pinned by runtime selection', () => { + expect( + claudeStructuredAuthPolicyForSettings( + settings({ activeClaudeManagedAccountIdsByRuntime: { host: 'host-a', wsl: {} } }) + ) + ).toEqual({ stripAuthEnv: true }) + }) + + it('does not strip for system auth, so an API-key-only user keeps their sign-in', () => { + expect(claudeStructuredAuthPolicyForSettings(settings({}))).toEqual({ stripAuthEnv: false }) + }) + + it('ignores a WSL-only selection: the structured child is always a native host process', () => { + expect( + claudeStructuredAuthPolicyForSettings( + settings({ + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-b' } } + }) + ) + ).toEqual({ stripAuthEnv: false }) + }) +}) + +describe('the strip vocabulary the policy governs', () => { + it('covers every Anthropic auth variable the terminal path knows about', () => { + // A new auth var added to the list without a matching refusal/strip path is the + // shape of the leak this lane already shipped once. + expect([...CLAUDE_AUTH_ENV_VARS]).toEqual([ + 'ANTHROPIC_API_KEY', + 'ANTHROPIC_AUTH_TOKEN', + 'CLAUDE_CODE_OAUTH_TOKEN', + 'AWS_BEARER_TOKEN_BEDROCK' + ]) + }) +}) + +// The refusal has to cover exactly what the strip removes. Anything narrower lets an +// override reach the child that applyClaudeEnvPatch would have deleted. +describe('hasClaudeAuthEnvConflict matches the strip it guards', () => { + it('refuses each Anthropic auth variable', () => { + for (const key of CLAUDE_AUTH_ENV_VARS) { + expect(hasClaudeAuthEnvConflict({ [key]: 'v' }, 'linux')).toBe(true) + } + }) + + // `ANTHROPIC_API_KEY=` in the agent env box is how a user blanks a variable, and the + // settings pipeline preserves the empty value (agent-default-env-draft.ts assigns + // everything after the `=`; normalizeTuiAgentEnvRecord drops empty KEYS only). An + // empty value cannot beat the pinned account and the strip removes the name anyway, + // so refusing it would break a terminal launch that works today for no security gain. + it('admits an override whose value is empty, the documented way to blank a variable', () => { + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_API_KEY: '' }, 'linux')).toBe(false) + expect(hasClaudeAuthEnvConflict({ anthropic_api_key: '' }, 'win32')).toBe(false) + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_CUSTOM_HEADERS: '' }, 'linux')).toBe(false) + }) + + it('still refuses the same names once they carry a value', () => { + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_API_KEY: 'sk-ant' }, 'linux')).toBe(true) + }) + + // The end-to-end shape the regression actually took: settings text -> normalized + // record -> launch env -> the predicate the terminal preflight gates on. + it('admits a blanked variable all the way from the settings record', () => { + const configured = normalizeTuiAgentEnvRecord({ claude: { ANTHROPIC_API_KEY: '' } }) + const launchEnv = resolveTuiAgentLaunchEnv('claude', configured) + + expect(launchEnv).toEqual({ ANTHROPIC_API_KEY: '' }) + expect(hasClaudeAuthEnvConflict(launchEnv, 'linux')).toBe(false) + }) + + it('folds case on win32, where the OS does', () => { + expect(hasClaudeAuthEnvConflict({ anthropic_api_key: 'sk-lower' }, 'win32')).toBe(true) + expect(hasClaudeAuthEnvConflict({ Anthropic_Custom_Headers: 'x-api-key: v' }, 'win32')).toBe( + true + ) + }) + + it('keeps env names case-sensitive off win32', () => { + expect(hasClaudeAuthEnvConflict({ anthropic_api_key: 'sk-lower' }, 'linux')).toBe(false) + }) + + it('admits non-auth Anthropic settings on both platforms', () => { + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_BASE_URL: 'https://gw.test' }, 'linux')).toBe(false) + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_BASE_URL: 'https://gw.test' }, 'win32')).toBe(false) + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_CUSTOM_HEADERS: 'X-Trace: 1' }, 'linux')).toBe( + false + ) + expect(hasClaudeAuthEnvConflict(undefined, 'linux')).toBe(false) + }) +}) diff --git a/src/main/claude-accounts/claude-structured-auth-policy.ts b/src/main/claude-accounts/claude-structured-auth-policy.ts new file mode 100644 index 00000000000..c30cd69b827 --- /dev/null +++ b/src/main/claude-accounts/claude-structured-auth-policy.ts @@ -0,0 +1,37 @@ +import type { GlobalSettings } from '../../shared/global-settings-types' +import { shouldStripClaudeAuthEnvForAccount } from './environment' +import { getSelectedClaudeAccountIdForTarget } from './runtime-selection' + +/** The structured mirror of the terminal preflight's `prepareClaudeAuth` result: + * the one field a launch resolution needs from the managed-account state. */ +export type ClaudeStructuredAuthPolicy = { + stripAuthEnv: boolean +} + +/** + * The only supported way to build a structured launch's auth policy. + * + * It exists as a named function rather than an inline object at the wiring site so + * that the settings-to-policy mapping is testable on its own: the one production + * wiring lives in a `@ts-nocheck` file, where neither the compiler nor a type test + * can see a dropped field. + * + * Structured Claude always spawns a native local-host child — the launch resolver + * refuses any record with a remote execution host or a WSL distro — so the host + * selection, not the platform default target, owns its auth. + */ +export function claudeStructuredAuthPolicyForSettings( + settings: Pick< + GlobalSettings, + | 'claudeManagedAccounts' + | 'activeClaudeManagedAccountId' + | 'activeClaudeManagedAccountIdsByRuntime' + > +): ClaudeStructuredAuthPolicy { + return { + stripAuthEnv: shouldStripClaudeAuthEnvForAccount( + settings.claudeManagedAccounts, + getSelectedClaudeAccountIdForTarget(settings, { runtime: 'host' }) + ) + } +} diff --git a/src/main/claude-accounts/environment.ts b/src/main/claude-accounts/environment.ts index 83fe3b40209..b85dd60a854 100644 --- a/src/main/claude-accounts/environment.ts +++ b/src/main/claude-accounts/environment.ts @@ -1,3 +1,5 @@ +import type { ClaudeManagedAccount } from '../../shared/managed-account-types' + export const CLAUDE_AUTH_ENV_VARS = [ 'ANTHROPIC_API_KEY', 'ANTHROPIC_AUTH_TOKEN', @@ -13,14 +15,21 @@ export type ClaudeEnvPatch = { export function applyClaudeEnvPatch( baseEnv: Record, patch: ClaudeEnvPatch, - options?: { stripAuthEnv?: boolean } + options?: { stripAuthEnv?: boolean; platform?: NodeJS.Platform } ): Record { if (options?.stripAuthEnv) { for (const key of CLAUDE_AUTH_ENV_VARS) { delete baseEnv[key] } - if (isAuthLikeCustomHeaders(baseEnv.ANTHROPIC_CUSTOM_HEADERS)) { - delete baseEnv.ANTHROPIC_CUSTOM_HEADERS + const platform = options.platform ?? process.platform + for (const key of Object.keys(baseEnv)) { + const normalized = platform === 'win32' ? key.toUpperCase() : key + if ( + (platform === 'win32' && CLAUDE_AUTH_ENV_VARS.some((authKey) => authKey === normalized)) || + (normalized === 'ANTHROPIC_CUSTOM_HEADERS' && isAuthLikeCustomHeaders(baseEnv[key])) + ) { + delete baseEnv[key] + } } } @@ -34,16 +43,94 @@ export function applyClaudeEnvPatch( return baseEnv } -export function hasClaudeAuthEnvConflict(env: Record | undefined): boolean { - if (!env) { +/** One string for every transport, so a terminal launch and a structured launch + * cannot drift into telling the user two different things about one refusal. */ +export const CLAUDE_AUTH_ENV_CONFLICT_MESSAGE = + 'This Claude launch defines explicit Anthropic auth environment variables. Remove those overrides before using a managed Claude account.' + +export const CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE = + 'A Claude account switch is in progress. Try again after it finishes.' + +/** + * Whether a launch on the host runtime must drop inherited Anthropic auth. + * + * Only a pinned host-managed account owns the credential, so only it may strip: + * with no managed account the user's own `ANTHROPIC_*` is their sign-in, and + * removing it signs them out of a CLI that would otherwise have worked. + */ +export function shouldStripClaudeAuthEnvForAccount( + accounts: readonly ClaudeManagedAccount[] | undefined, + activeAccountId: string | null | undefined +): boolean { + if (!activeAccountId) { return false } return ( - CLAUDE_AUTH_ENV_VARS.some((key) => Boolean(env[key])) || - isAuthLikeCustomHeaders(env.ANTHROPIC_CUSTOM_HEADERS) + (accounts ?? []).find((account) => account.id === activeAccountId)?.managedAuthRuntime !== 'wsl' ) } +/** + * Whether a launch's explicit env carries Anthropic auth a managed account must own. + * + * The key comparison mirrors applyClaudeEnvPatch's strip exactly: case-insensitive on + * win32, where the OS folds env names so `anthropic_api_key` is an effective + * `ANTHROPIC_API_KEY`, and case-sensitive elsewhere. A refusal narrower than the strip + * lets an override through that the strip would have removed. + * + * A non-empty value is what makes it a conflict. `ANTHROPIC_API_KEY=` in the agent env + * box is how a user blanks a variable — the settings pipeline preserves that empty value + * (normalizeTuiAgentEnvRecord drops empty KEYS only) — and an empty override can neither + * authenticate nor beat the pinned account, while the strip removes the name regardless. + * Refusing it would break a terminal launch that works today for no security gain. + */ +/** + * The inherited Anthropic auth a non-stripping launch has to carry forward explicitly. + * + * applyClaudeEnvPatch always strips the inherited half of a child env, and the + * configured half is what overrides it — so a system-auth user's own key only survives + * if the caller puts it back deliberately. Returns the exact keys present, so a + * win32 `anthropic_api_key` is carried under the name the OS actually has. + */ +export function claudeAuthEnvCarriedForward( + inherited: NodeJS.ProcessEnv, + platform: NodeJS.Platform = process.platform +): Record { + const carried: Record = {} + for (const [key, value] of Object.entries(inherited)) { + if (value === undefined) { + continue + } + const normalized = platform === 'win32' ? key.toUpperCase() : key + if ( + CLAUDE_AUTH_ENV_VARS.some((authKey) => authKey === normalized) || + (normalized === 'ANTHROPIC_CUSTOM_HEADERS' && isAuthLikeCustomHeaders(value)) + ) { + carried[key] = value + } + } + return carried +} + +export function hasClaudeAuthEnvConflict( + env: Record | undefined, + platform: NodeJS.Platform = process.platform +): boolean { + if (!env) { + return false + } + for (const [key, value] of Object.entries(env)) { + const normalized = platform === 'win32' ? key.toUpperCase() : key + if (value && CLAUDE_AUTH_ENV_VARS.some((authKey) => authKey === normalized)) { + return true + } + if (normalized === 'ANTHROPIC_CUSTOM_HEADERS' && isAuthLikeCustomHeaders(value)) { + return true + } + } + return false +} + function isAuthLikeCustomHeaders(value: string | undefined): boolean { if (!value) { return false diff --git a/src/main/claude-accounts/live-pty-gate.ts b/src/main/claude-accounts/live-pty-gate.ts index 9e30b621924..caab66c3430 100644 --- a/src/main/claude-accounts/live-pty-gate.ts +++ b/src/main/claude-accounts/live-pty-gate.ts @@ -5,6 +5,13 @@ const liveClaudePtyIds = new Set() // survived the app restart inside the daemon. const seededUnconfirmedPtyIds = new Set() let switchInProgress = false +// Woken by endClaudeAuthSwitch so a caller past the point of no return can wait the +// swap out instead of refusing. See whenClaudeAuthSwitchSettles. +const switchSettledListeners = new Set<() => void>() + +/** A managed account swap is a credential-file rewrite, not a network round trip; + * anything past this is a wedged switch, and refusing beats waiting forever. */ +export const CLAUDE_AUTH_SWITCH_SETTLE_TIMEOUT_MS = 15_000 export type ClaudeLivePtyPersistence = { addClaudeLivePtySessionId(sessionId: string): void @@ -81,6 +88,35 @@ export function markClaudePtyExited(ptyId: string): void { notifyDrainedOnTransition(hadLivePtys) } +/** + * Register a structured Claude child with the same gate the terminal path uses. + * + * The gate is what makes the managed OAuth refresh defer instead of rotating a + * single-use refresh token out from under a running Claude (runtime-auth-sync.ts). + * A structured session's child is as much a live Claude as a PTY's is, so it has to + * hold the gate too — otherwise a refresh mid-turn breaks its next API call while an + * identical terminal session is protected. + * + * Deliberately not persisted, unlike markClaudePtySpawned: these children are direct + * children of this process and cannot survive a restart, so seeding them back on the + * next launch would hold the gate closed for a process that is provably gone. + */ +export function markClaudeStructuredChildSpawned(childKey: string): void { + liveClaudePtyIds.add(structuredChildGateId(childKey)) +} + +export function markClaudeStructuredChildExited(childKey: string): void { + const hadLivePtys = liveClaudePtyIds.size > 0 + liveClaudePtyIds.delete(structuredChildGateId(childKey)) + notifyDrainedOnTransition(hadLivePtys) +} + +// Namespaced so a structured child can never collide with a daemon PTY session id, +// which confirmSeededClaudeLivePtys reconciles against the daemon's own list. +function structuredChildGateId(childKey: string): string { + return `claude-structured:${childKey}` +} + export function hasLiveClaudePtys(): boolean { return liveClaudePtyIds.size > 0 } @@ -93,7 +129,44 @@ export function beginClaudeAuthSwitch(): void { } export function endClaudeAuthSwitch(): void { + const wasInProgress = switchInProgress switchInProgress = false + if (!wasInProgress) { + return + } + // Each listener removes itself as it settles; Set iteration is defined over that. + for (const listener of switchSettledListeners) { + listener() + } +} + +/** + * Resolves `true` once no account switch is running, `false` if one is still running + * at the deadline. + * + * Exists for callers that have already done irreversible work — a structured acquire + * has closed the old child by the time it resolves its launch, so turning a switch + * into a refusal there strands the user with a dead session and no replacement. + * Waiting for the swap and then launching against it is the recoverable answer; + * refusing is only correct when nothing has been torn down yet. + */ +export function whenClaudeAuthSwitchSettles( + timeoutMs = CLAUDE_AUTH_SWITCH_SETTLE_TIMEOUT_MS +): Promise { + if (!switchInProgress) { + return Promise.resolve(true) + } + return new Promise((resolve) => { + const settle = (settled: boolean): void => { + switchSettledListeners.delete(listener) + clearTimeout(timer) + resolve(settled) + } + const listener = (): void => settle(true) + switchSettledListeners.add(listener) + const timer = setTimeout(() => settle(false), timeoutMs) + timer.unref?.() + }) } export function isClaudeAuthSwitchInProgress(): boolean { diff --git a/src/main/claude-accounts/runtime-auth/runtime-auth-preparation.ts b/src/main/claude-accounts/runtime-auth/runtime-auth-preparation.ts index ae79c4c7bbb..dabcd9d472f 100644 --- a/src/main/claude-accounts/runtime-auth/runtime-auth-preparation.ts +++ b/src/main/claude-accounts/runtime-auth/runtime-auth-preparation.ts @@ -2,6 +2,7 @@ import { join } from 'node:path' import type { ClaudeManagedAccount } from '../../../shared/managed-account-types' import { resolveLocalAccountRuntimeTarget } from '../../../shared/local-account-runtime' import { parseWslUncPath } from '../../../shared/wsl-paths' +import { shouldStripClaudeAuthEnvForAccount } from '../environment' import { getDefaultWslDistro, getWslHome } from '../../wsl' import { getSelectedClaudeAccountIdForTarget, @@ -69,7 +70,10 @@ export class ClaudeRuntimeAuthPreparationService extends ClaudeRuntimeAuthSnapsh wslDistro: null, wslLinuxConfigDir: null, envPatch: paths.envPatch, - stripAuthEnv: Boolean(activeAccountId && activeAccount?.managedAuthRuntime !== 'wsl'), + stripAuthEnv: shouldStripClaudeAuthEnvForAccount( + settings.claudeManagedAccounts, + activeAccountId + ), managedRefreshDeferredByLivePty: Boolean( activeAccountId && activeAccount?.managedAuthRuntime !== 'wsl' && diff --git a/src/main/claude/__fixtures__/claude-agent-sdk-scripted-cli.mjs b/src/main/claude/__fixtures__/claude-agent-sdk-scripted-cli.mjs new file mode 100644 index 00000000000..4f2a09425fd --- /dev/null +++ b/src/main/claude/__fixtures__/claude-agent-sdk-scripted-cli.mjs @@ -0,0 +1,144 @@ +// Scripted stand-in for the Claude Code CLI, driven by the SDK contract-pin +// tests. It speaks just enough stream-json to satisfy the SDK: it answers every +// inbound control_request with a success control_response, records everything it +// observes to a report file, and plays back the steps listed in a scenario file. +// +// Env contract (set by the test): +// ORCA_SDK_CONTRACT_SCENARIO_PATH — JSON file +// { steps: Step[], controlResponses?: { [subtype]: } } where a Step is +// { emit: } +// ORCA_SDK_CONTRACT_REPORT_PATH — where argv/env observations are written +// ORCA_SDK_CONTRACT_IGNORE_SIGTERM — trap SIGTERM/SIGINT and outlive stdin close +// ORCA_SDK_CONTRACT_IGNORE_CONTROL_REQUESTS — record control requests but never answer +// ORCA_SDK_CONTRACT_DESCENDANT — fork an idle grandchild and report its pid +import { spawn } from 'node:child_process' +import { readFileSync, writeFileSync } from 'node:fs' +import { createInterface } from 'node:readline' + +const scenarioPath = process.env.ORCA_SDK_CONTRACT_SCENARIO_PATH +const reportPath = process.env.ORCA_SDK_CONTRACT_REPORT_PATH + +const report = { + argv: process.argv.slice(1), + execPath: process.execPath, + controlRequests: [], + controlResponses: [], + userMessages: [], + descendantPid: null +} +const writeReport = () => { + if (reportPath) { + writeFileSync(reportPath, JSON.stringify(report)) + } +} +// Written immediately so a test can prove which script the SDK executed even if +// the session dies before the scenario completes. +writeReport() + +const scenario = scenarioPath ? JSON.parse(readFileSync(scenarioPath, 'utf8')) : { steps: [] } + +if (process.env.ORCA_SDK_CONTRACT_IGNORE_SIGTERM) { + process.on('SIGTERM', () => {}) + process.on('SIGINT', () => {}) + setInterval(() => {}, 1_000_000) +} +if (process.env.ORCA_SDK_CONTRACT_DESCENDANT) { + const descendant = spawn(process.execPath, ['-e', 'setInterval(() => {}, 1000000)'], { + stdio: 'ignore' + }) + descendant.unref() + report.descendantPid = descendant.pid ?? null + writeReport() +} + +const emit = (frame) => process.stdout.write(`${JSON.stringify(frame)}\n`) + +const waiters = [] +const settle = (kind, requestId) => { + for (let i = waiters.length - 1; i >= 0; i--) { + const waiter = waiters[i] + if ( + waiter.kind === kind && + (waiter.requestId === undefined || waiter.requestId === requestId) + ) { + waiters.splice(i, 1) + waiter.resolve() + } + } +} +const waitFor = (kind, requestId) => { + if (kind === 'user' && report.userMessages.length > 0) { + return Promise.resolve() + } + if ( + kind === 'control_response' && + report.controlResponses.some((frame) => frame.response?.request_id === requestId) + ) { + return Promise.resolve() + } + return new Promise((resolve) => waiters.push({ kind, requestId, resolve })) +} + +createInterface({ input: process.stdin }).on('line', (line) => { + let frame + try { + frame = JSON.parse(line) + } catch { + return + } + if (frame.type === 'control_request') { + report.controlRequests.push(frame) + writeReport() + if (process.env.ORCA_SDK_CONTRACT_IGNORE_CONTROL_REQUESTS) { + return + } + emit({ + type: 'control_response', + response: { + subtype: 'success', + request_id: frame.request_id, + response: scenario.controlResponses?.[frame.request?.subtype] ?? { + commands: [], + models: [] + } + } + }) + return + } + if (frame.type === 'control_response') { + report.controlResponses.push(frame) + writeReport() + settle('control_response', frame.response?.request_id) + return + } + if (frame.type === 'user') { + report.userMessages.push(frame) + writeReport() + settle('user') + } +}) + +// Never outlive a wedged test: the readline subscription would otherwise hold +// this process open forever if the SDK side stops driving the scenario. +setTimeout(() => process.exit(3), 20_000).unref() + +for (const step of scenario.steps) { + if (step.emit) { + emit(step.emit) + } else if (step.stderr !== undefined) { + process.stderr.write(step.stderr) + } else if (step.awaitUserMessage) { + await waitFor('user') + } else if (step.awaitControlResponse !== undefined) { + await waitFor('control_response', step.awaitControlResponse) + } else if (step.delayMs) { + await new Promise((resolve) => setTimeout(resolve, step.delayMs)) + } else if (step.exit !== undefined) { + // A CLI that refuses to start: leave with its own status, stderr already written. + writeReport() + process.exit(step.exit) + } +} +writeReport() +process.exit(0) diff --git a/src/main/claude/claude-agent-sdk-contract-pins.test.ts b/src/main/claude/claude-agent-sdk-contract-pins.test.ts new file mode 100644 index 00000000000..46bcb7219d3 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-contract-pins.test.ts @@ -0,0 +1,519 @@ +import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { createRequire } from 'node:module' +import { tmpdir } from 'node:os' +import { dirname, join } from 'node:path' +import { + query, + type CanUseTool, + type Options, + type SDKUserMessage, + type SpawnedProcess as SdkSpawnedProcess, + type SpawnOptions as SdkSpawnOptions +} from '@anthropic-ai/claude-agent-sdk' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { spawnProcess } from '../../shared/child-process/run-process' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import type { AgentSessionRecordStore } from '../runtime/agent-session-record-store' +import { claudeQuerySettingsReader } from './claude-agent-sdk-control-requests' +import { createClaudeStructuredLaunchResolver } from './claude-structured-launch-resolution' + +// Contract pins for @anthropic-ai/claude-agent-sdk, run against the real SDK +// driving a scripted fake CLI (never the real Claude binary). These tests exist +// to catch a future SDK version drifting under Orca: unknown-frame pass-through, +// spawner env fidelity, argument parity with the pre-SDK argv, +// permission-callback semantics, and executable-path override. + +const FAKE_CLI = join(__dirname, '__fixtures__', 'claude-agent-sdk-scripted-cli.mjs') +const SESSION_ID = '5348c19f-6a54-4c2e-9c68-9c2b1a3d4e5f' +const LEAF_UUID = 'ad0f7c9e-1b2c-4d3e-8f90-abc123def456' +const PINNED_SDK_VERSION = '0.3.251' +const SDK_PLATFORM_PACKAGE_BASENAMES = [ + 'claude-agent-sdk-darwin-arm64', + 'claude-agent-sdk-darwin-x64', + 'claude-agent-sdk-linux-arm64', + 'claude-agent-sdk-linux-arm64-musl', + 'claude-agent-sdk-linux-x64', + 'claude-agent-sdk-linux-x64-musl', + 'claude-agent-sdk-win32-arm64', + 'claude-agent-sdk-win32-x64' +] + +/** + * The exact argv the hand-rolled transport built before the SDK swap. Frozen here + * as the parity oracle: CLAUDE_STRUCTURED_BASE_OPTIONS has to keep producing it. + */ +const PRE_SDK_ARGV = [ + '-p', + '--input-format', + 'stream-json', + '--output-format', + 'stream-json', + '--include-partial-messages', + '--verbose', + '--replay-user-messages', + '--permission-prompt-tool', + 'stdio', + '--setting-sources', + 'user,project,local' +] + +const RESULT_FRAME = { + type: 'result', + subtype: 'success', + is_error: false, + duration_ms: 1, + duration_api_ms: 1, + num_turns: 1, + result: 'ok', + session_id: SESSION_ID, + total_cost_usd: 0, + usage: { input_tokens: 1, output_tokens: 1 }, + uuid: 'uuid-result-1' +} + +type ScenarioStep = Record +type SpawnSeen = { + command: string + args: string[] + cwd: string | undefined + env: Record +} +type ScriptedCliReport = { + argv: string[] + execPath: string + controlRequests: { request_id: string; request: { subtype: string } }[] + controlResponses: { response: { request_id: string; response?: Record } }[] + userMessages: Record[] +} + +const scratchDirs: string[] = [] +afterEach(() => { + vi.unstubAllEnvs() + for (const dir of scratchDirs.splice(0)) { + rmSync(dir, { recursive: true, force: true }) + } +}) + +function scriptScenario( + steps: ScenarioStep[], + controlResponses: Record = {} +): { + scenarioPath: string + reportPath: string + cwd: string + readReport: () => ScriptedCliReport +} { + const dir = mkdtempSync(join(tmpdir(), 'claude-sdk-contract-')) + scratchDirs.push(dir) + const scenarioPath = join(dir, 'scenario.json') + const reportPath = join(dir, 'report.json') + writeFileSync(scenarioPath, JSON.stringify({ steps, controlResponses })) + return { + scenarioPath, + reportPath, + cwd: dir, + readReport: () => JSON.parse(readFileSync(reportPath, 'utf8')) as ScriptedCliReport + } +} + +function scenarioEnv(scenario: { scenarioPath: string; reportPath: string }) { + return { + PATH: process.env.PATH, + ORCA_SDK_CONTRACT_SCENARIO_PATH: scenario.scenarioPath, + ORCA_SDK_CONTRACT_REPORT_PATH: scenario.reportPath + } +} + +function recordingSpawner(spawns: SpawnSeen[]) { + return (opts: SdkSpawnOptions): SdkSpawnedProcess => { + spawns.push({ + command: opts.command, + args: [...opts.args], + cwd: opts.cwd, + env: { ...opts.env } + }) + return spawnProcess({ + program: opts.command, + args: opts.args, + cwd: opts.cwd, + env: opts.env as NodeJS.ProcessEnv, + signal: opts.signal + }) as unknown as SdkSpawnedProcess + } +} + +function resolvedLaunch(launchArgs: string[]) { + const record = { + sessionId: 'contract-pin-session', + provider: 'claude', + location: { + executionHostId: LOCAL_EXECUTION_HOST_ID, + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'folder' + }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/home/work/.claude' }, + providerHandleChain: [], + launchArgs + } as unknown as AgentSessionRecord + return createClaudeStructuredLaunchResolver({ + store: { getRecord: () => record } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async () => '/repos/workspace-1', + resolveCommand: () => FAKE_CLI, + resolveAuthPolicy: () => ({ stripAuthEnv: true }) + })({ identity: { sessionId: record.sessionId } as never }) +} + +function singleUserTurn(): AsyncIterable { + return (async function* () { + yield { + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'hello' }] }, + parent_tool_use_id: null, + session_id: SESSION_ID + } as SDKUserMessage + // Hold input open; the stream ends when the scripted CLI exits, and an + // unresolved bare promise does not keep the event loop alive. + await new Promise(() => {}) + })() +} + +async function drainQuery(options: Options): Promise[]> { + const messages: Record[] = [] + for await (const message of query({ prompt: singleUserTurn(), options })) { + messages.push(message as unknown as Record) + } + return messages +} + +/** Expand `--flag=value` argv entries so both SDK spellings compare equal. */ +function normalizeArgv(args: string[]): string[] { + return args.flatMap((arg) => { + if (!arg.startsWith('--')) { + return [arg] + } + const eq = arg.indexOf('=') + return eq === -1 ? [arg] : [arg.slice(0, eq), arg.slice(eq + 1)] + }) +} + +/** Group the pre-SDK argv into flag/value pairs. */ +function flagTable(args: readonly string[]): { flag: string; value: string | null }[] { + const table: { flag: string; value: string | null }[] = [] + for (let i = 0; i < args.length; i++) { + const flag = args[i]! + const next = args[i + 1] + if (next !== undefined && !next.startsWith('-')) { + table.push({ flag, value: next }) + i++ + } else { + table.push({ flag, value: null }) + } + } + return table +} + +describe('Claude Agent SDK contract pins', () => { + it('yields unknown types, unknown fields and unknown content blocks verbatim, and consumes keep_alive', async () => { + const unknownTopLevel = { + type: 'message_kind_from_the_future', + session_id: SESSION_ID, + uuid: 'uuid-unknown-1', + payload: { alpha: 1, nested: { flags: ['a', 'b'] } } + } + const assistantWithUnknowns = { + type: 'assistant', + message: { + id: 'msg-1', + type: 'message', + role: 'assistant', + model: 'claude-x', + content: [ + { type: 'text', text: 'hello back' }, + { type: 'content_block_from_the_future', payload: { depth: 3 } } + ], + stop_reason: null, + stop_sequence: null, + usage: { input_tokens: 1, output_tokens: 2 } + }, + parent_tool_use_id: null, + uuid: 'uuid-assistant-1', + session_id: SESSION_ID, + field_from_the_future: 'preserved' + } + const scenario = scriptScenario([ + { awaitUserMessage: true }, + { emit: { type: 'keep_alive' } }, + { emit: unknownTopLevel }, + { emit: assistantWithUnknowns }, + { emit: RESULT_FRAME } + ]) + const spawns: SpawnSeen[] = [] + const messages = await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + expect(messages.find((m) => m.uuid === 'uuid-unknown-1')).toEqual(unknownTopLevel) + expect(messages.find((m) => m.uuid === 'uuid-assistant-1')).toEqual(assistantWithUnknowns) + // The SDK intercepts keep_alive internally — a liveness signal must never + // be derived from it reaching the consumer, because it does not. + expect(messages.some((m) => m.type === 'keep_alive')).toBe(false) + expect(messages.some((m) => m.type === 'result')).toBe(true) + }) + + it('hands the custom spawner exactly the caller-supplied env, plus the two pinned SDK mutations', async () => { + vi.stubEnv('ANTHROPIC_API_KEY', 'ambient-key-must-not-leak') + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const spawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: { + ...scenarioEnv(scenario), + CLAUDE_CONFIG_DIR: '/pinned/claude-config', + ORCA_AGENT_SESSION_SPAWN_TOKEN: 'spawn-token-1', + NODE_OPTIONS: '--max-old-space-size=64' + }, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + const env = spawns[0]!.env + // Supplied values arrive verbatim: the config-dir pin and spawn token are + // observable at this boundary, so Orca's auth scrubbing stays assertable. + expect(env.CLAUDE_CONFIG_DIR).toBe('/pinned/claude-config') + expect(env.ORCA_AGENT_SESSION_SPAWN_TOKEN).toBe('spawn-token-1') + // Ambient process.env is NOT merged in when env is supplied. + expect(env.ANTHROPIC_API_KEY).toBeUndefined() + // The SDK's two documented mutations, pinned so a change is noticed. + expect(env.CLAUDE_CODE_ENTRYPOINT).toBe('sdk-ts') + expect('NODE_OPTIONS' in env).toBe(false) + }) + + it('inherits process.env into the child when env is omitted — the ambient-auth sharp edge', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + vi.stubEnv('ORCA_SDK_CONTRACT_SCENARIO_PATH', scenario.scenarioPath) + vi.stubEnv('ORCA_SDK_CONTRACT_REPORT_PATH', scenario.reportPath) + vi.stubEnv('ORCA_SDK_CONTRACT_AMBIENT_CANARY', 'inherited-from-process-env') + const spawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + // Omitting env reproduces the ambient-auth-leak failure mode: the child + // sees everything in process.env. Orca must therefore always pass an + // explicit, fully-constructed env. + expect(spawns[0]!.env.ORCA_SDK_CONTRACT_AMBIENT_CANARY).toBe('inherited-from-process-env') + }) + + it('emits --replay-user-messages only through extraArgs, never on its own', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const bareSpawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + spawnClaudeCodeProcess: recordingSpawner(bareSpawns) + }) + expect(bareSpawns[0]!.args).not.toContain('--replay-user-messages') + + const replayScenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const replaySpawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: replayScenario.cwd, + env: scenarioEnv(replayScenario), + extraArgs: { 'replay-user-messages': null }, + spawnClaudeCodeProcess: recordingSpawner(replaySpawns) + }) + const replayArgs = replaySpawns[0]!.args + expect(replayArgs.filter((arg) => arg === '--replay-user-messages')).toHaveLength(1) + }) + + it('produces a matching CLI flag for every pre-SDK argv entry', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const spawns: SpawnSeen[] = [] + // Driven by the real resolver, so the argv walk covers the durable-launchArgs + // translation and its merge order, not a hand-written options literal. + const launch = await resolvedLaunch(['--model', 'claude-sonnet-4-5', '--effort', 'high']) + await drainQuery({ + ...launch.options, + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + canUseTool: (async () => ({ behavior: 'deny', message: 'unused' })) as CanUseTool, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + expect(spawns).toHaveLength(1) + const argv = normalizeArgv(spawns[0]!.args) + // Typed-first translation must not also spell the flag through extraArgs. + for (const flag of ['--model', '--effort']) { + expect( + argv.filter((arg) => arg === flag), + `${flag} occurrences` + ).toHaveLength(1) + } + expect(argv[argv.indexOf('--model') + 1]).toBe('claude-sonnet-4-5') + expect(argv[argv.indexOf('--effort') + 1]).toBe('high') + // Headless print mode is the SDK's only mode; `query()` never passes `-p`, + // and if the SDK ever started passing it this pin would notice. + const impliedByHeadlessQuery = new Set(['-p']) + for (const entry of flagTable(PRE_SDK_ARGV)) { + if (impliedByHeadlessQuery.has(entry.flag)) { + expect(argv, `${entry.flag} is implied, never spelled`).not.toContain(entry.flag) + continue + } + const at = argv.indexOf(entry.flag) + expect(at, `SDK argv is missing ${entry.flag}`).toBeGreaterThanOrEqual(0) + if (entry.value !== null) { + expect(argv[at + 1], `value of ${entry.flag}`).toBe(entry.value) + } + } + // The launch resolver always carries one of --session-id / --resume. + const sessionAt = argv.indexOf('--session-id') + expect(sessionAt).toBeGreaterThanOrEqual(0) + expect(argv[sessionAt + 1]).toBe(launch.providerSessionId) + }) + + it('still exposes the runtime get_settings reader the auth diagnostic depends on', async () => { + // 0.3.251 ships getSettings() but redacts it from the Query declaration. This pin + // is the drift alarm: if a bump drops or reshapes it, the diagnostic degrades and + // this test says so instead of the degradation shipping silently. + const settings = { env: { ANTHROPIC_BASE_URL: 'https://settings.example.test' } } + const scenario = scriptScenario([{ delayMs: 3_000 }], { get_settings: settings }) + const session = query({ + prompt: singleUserTurn(), + options: { + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario) + } + }) + try { + const read = claudeQuerySettingsReader(session) + expect(read, 'the SDK no longer exposes get_settings at runtime').not.toBeNull() + await expect(read?.()).resolves.toEqual(settings) + } finally { + await session.return(undefined) + } + }) + + it('maps resume identity to --resume and --resume-session-at', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const spawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + resume: SESSION_ID, + resumeSessionAt: LEAF_UUID, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + const argv = normalizeArgv(spawns[0]!.args) + const resumeAt = argv.indexOf('--resume') + expect(resumeAt).toBeGreaterThanOrEqual(0) + expect(argv[resumeAt + 1]).toBe(SESSION_ID) + const leafAt = argv.indexOf('--resume-session-at') + expect(leafAt).toBeGreaterThanOrEqual(0) + expect(argv[leafAt + 1]).toBe(LEAF_UUID) + }) + + it('gives canUseTool the wire request_id and fires its abort signal on control_cancel_request', async () => { + const scenario = scriptScenario([ + { awaitUserMessage: true }, + { + emit: { + type: 'control_request', + request_id: 'perm-421', + request: { + subtype: 'can_use_tool', + tool_name: 'Bash', + input: { command: 'echo hi' }, + tool_use_id: 'tool-use-9' + } + } + }, + { delayMs: 120 }, + { emit: { type: 'control_cancel_request', request_id: 'perm-421' } }, + { awaitControlResponse: 'perm-421' }, + { emit: RESULT_FRAME } + ]) + const seen: { toolName: string; requestId: string; toolUseID: string }[] = [] + let abortFired = false + const canUseTool: CanUseTool = (toolName, _input, { signal, requestId, toolUseID }) => { + seen.push({ toolName, requestId, toolUseID }) + return new Promise((resolve) => { + signal.addEventListener('abort', () => { + abortFired = true + resolve({ behavior: 'deny', message: 'cancelled by test' }) + }) + }) + } + const spawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + canUseTool, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + expect(seen).toEqual([{ toolName: 'Bash', requestId: 'perm-421', toolUseID: 'tool-use-9' }]) + expect(abortFired).toBe(true) + // The callback's settlement is written back onto the wire against the same id. + const settled = scenario + .readReport() + .controlResponses.find((frame) => frame.response.request_id === 'perm-421') + expect(settled?.response.response?.behavior).toBe('deny') + // Exactly one process spawn per query, control traffic included. + expect(spawns).toHaveLength(1) + }) + + it('runs the executable given via pathToClaudeCodeExecutable under the default spawner', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const messages = await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario) + }) + + expect(messages.some((m) => m.type === 'result')).toBe(true) + const report = scenario.readReport() + // The SDK executed exactly the script we pointed it at — no bundled binary. + expect(report.argv[0]).toBe(FAKE_CLI) + expect(report.execPath).toContain('node') + // And the streaming handshake went to it: the SDK sent its initialize + // control request to our script. + expect(report.controlRequests.some((frame) => frame.request.subtype === 'initialize')).toBe( + true + ) + }) + + it('pins the SDK version the contract was verified against', () => { + const sdkEntry = createRequire(__filename).resolve('@anthropic-ai/claude-agent-sdk') + const manifest = JSON.parse(readFileSync(join(dirname(sdkEntry), 'package.json'), 'utf8')) as { + version: string + } + expect(manifest.version).toBe(PINNED_SDK_VERSION) + }) + + it('keeps the eight bundled CLI platform binaries out of the install', () => { + const sdkEntry = createRequire(__filename).resolve('@anthropic-ai/claude-agent-sdk') + // The SDK's own scoped directory is where pnpm would link its optional + // platform packages; ignoredOptionalDependencies must keep them all absent. + const scopeDir = dirname(dirname(sdkEntry)) + for (const basename of SDK_PLATFORM_PACKAGE_BASENAMES) { + expect( + existsSync(join(scopeDir, basename, 'package.json')), + `${basename} must not be installed` + ).toBe(false) + } + }) +}) diff --git a/src/main/claude/claude-agent-sdk-control-requests.test.ts b/src/main/claude/claude-agent-sdk-control-requests.test.ts new file mode 100644 index 00000000000..f76650d8151 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-control-requests.test.ts @@ -0,0 +1,26 @@ +import type { Query } from '@anthropic-ai/claude-agent-sdk' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { createClaudeControlSurface } from './claude-agent-sdk-control-requests' + +afterEach(() => { + vi.useRealTimers() +}) + +describe('createClaudeControlSurface stopTask', () => { + it('bounds a lost reply and permits a later stop request', async () => { + vi.useFakeTimers() + const stopTask = vi + .fn<() => Promise>() + .mockImplementationOnce(() => new Promise(() => {})) + .mockResolvedValueOnce() + const controls = createClaudeControlSurface({ stopTask } as unknown as Query) + const timedOut = expect(controls.stopTask('task-1', { timeoutMs: 25 })).rejects.toThrow( + 'claude stop_task request timed out' + ) + + await vi.advanceTimersByTimeAsync(25) + await timedOut + await expect(controls.stopTask('task-2', { timeoutMs: 25 })).resolves.toBeUndefined() + expect(stopTask).toHaveBeenCalledTimes(2) + }) +}) diff --git a/src/main/claude/claude-agent-sdk-control-requests.ts b/src/main/claude/claude-agent-sdk-control-requests.ts new file mode 100644 index 00000000000..71498f28421 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-control-requests.ts @@ -0,0 +1,159 @@ +import type { + PermissionMode, + Query, + SDKControlInterruptResponse +} from '@anthropic-ai/claude-agent-sdk' + +export class ClaudeControlRequestError extends Error { + constructor( + readonly subtype: string, + message: string + ) { + super(message) + this.name = 'ClaudeControlRequestError' + } +} + +export const CLAUDE_DEFAULT_REQUEST_TIMEOUT_MS = 30_000 + +/** The SDK closes a query out from under an in-flight control request with this exact message. */ +const QUERY_CLOSED_MESSAGE = 'Query closed before response received' + +/** 0.3.251 ships getSettings() but redacts it from the Query declaration; the typeof guard below is its degradation path. */ +type ClaudeQuerySettingsReader = { getSettings?: () => Promise } + +export function claudeQuerySettingsReader(query: Query): (() => Promise) | null { + const reader = (query as unknown as ClaudeQuerySettingsReader).getSettings + return typeof reader === 'function' ? reader.bind(query) : null +} + +/** + * cancel_async_message is a runtime Query method the shipped 0.3.251 declaration omits; + * it withdraws a single still-queued async user message by uuid so an interrupted turn + * cannot spawn a later unexpected turn. The typeof guard is its degradation path. + */ +type ClaudeQueryAsyncCanceller = { cancelAsyncMessage?: (uuid: string) => Promise } + +export function claudeQueryAsyncCanceller( + query: Query +): ((uuid: string) => Promise) | null { + const cancel = (query as unknown as ClaudeQueryAsyncCanceller).cancelAsyncMessage + return typeof cancel === 'function' ? cancel.bind(query) : null +} + +export type ClaudeControlOptions = { timeoutMs?: number } + +/** + * Run one native Query control method under Orca's deadline and error classification. + * + * The SDK owns correlation but applies no deadline, so the timeout stays here — and its + * message is load-bearing: the init proof matches on `claude initialize request timed out`. + * A closed query is a transport failure, not the CLI rejecting the request, so only the + * latter is re-thrown as a `ClaudeControlRequestError` a caller may surface as a rejection. + */ +export function runClaudeControl( + subtype: string, + run: () => Promise, + timeoutMs: number = CLAUDE_DEFAULT_REQUEST_TIMEOUT_MS +): Promise { + let timer: ReturnType | null = null + const deadline = new Promise((_resolve, reject) => { + timer = setTimeout(() => reject(new Error(`claude ${subtype} request timed out`)), timeoutMs) + timer.unref?.() + }) + return Promise.race([ + Promise.resolve() + .then(run) + .catch((error: unknown) => { + const message = error instanceof Error ? error.message : String(error) + if (error instanceof ClaudeControlRequestError || message === QUERY_CLOSED_MESSAGE) { + throw error + } + throw new ClaudeControlRequestError(subtype, message) + }), + deadline + ]).finally(() => { + if (timer) { + clearTimeout(timer) + } + }) +} + +/** The native control surface Orca drives, one method per Query control request. */ +export type ClaudeControlSurface = { + interrupt: ( + options?: ClaudeControlOptions & { cancelQueued?: boolean } + ) => Promise + cancelAsyncMessage: (uuid: string, options?: ClaudeControlOptions) => Promise + setModel: (model: string | undefined, options?: ClaudeControlOptions) => Promise + setPermissionMode: (mode: PermissionMode, options?: ClaudeControlOptions) => Promise + applyFlagSettings: ( + settings: Parameters[0], + options?: ClaudeControlOptions + ) => Promise + stopTask: (taskId: string, options?: ClaudeControlOptions) => Promise + supportedModels: (options?: ClaudeControlOptions) => Promise + initializationResult: (options?: ClaudeControlOptions) => Promise + getSettings: (options?: ClaudeControlOptions) => Promise +} + +type InterruptingQuery = { + interrupt: (options?: { + cancelQueued?: boolean + }) => Promise +} + +export function createClaudeControlSurface(query: Query): ClaudeControlSurface { + return { + interrupt: (options) => + runClaudeControl( + 'interrupt', + () => + (query as unknown as InterruptingQuery).interrupt( + options?.cancelQueued ? { cancelQueued: true } : undefined + ), + options?.timeoutMs + ), + cancelAsyncMessage: (uuid, options) => { + const cancel = claudeQueryAsyncCanceller(query) + return cancel + ? runClaudeControl('cancel_async_message', () => cancel(uuid), options?.timeoutMs).then( + () => {} + ) + : Promise.resolve() + }, + setModel: (model, options) => + runClaudeControl('set_model', () => query.setModel(model), options?.timeoutMs).then(() => {}), + setPermissionMode: (mode, options) => + runClaudeControl( + 'set_permission_mode', + () => query.setPermissionMode(mode), + options?.timeoutMs + ).then(() => {}), + applyFlagSettings: (settings, options) => + runClaudeControl( + 'apply_flag_settings', + () => query.applyFlagSettings(settings), + options?.timeoutMs + ).then(() => {}), + stopTask: (taskId, options) => + runClaudeControl('stop_task', () => query.stopTask(taskId), options?.timeoutMs).then( + () => {} + ), + supportedModels: (options) => + runClaudeControl('list_models', () => query.supportedModels(), options?.timeoutMs), + initializationResult: (options) => + runClaudeControl('initialize', () => query.initializationResult(), options?.timeoutMs), + getSettings: (options) => { + const read = claudeQuerySettingsReader(query) + return read + ? runClaudeControl('get_settings', read, options?.timeoutMs) + : Promise.reject( + new ClaudeControlRequestError( + 'get_settings', + 'this SDK exposes no get_settings request' + ) + ) + } + } +} diff --git a/src/main/claude/claude-agent-sdk-exit-proof-identity.test.ts b/src/main/claude/claude-agent-sdk-exit-proof-identity.test.ts new file mode 100644 index 00000000000..04d58067353 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-exit-proof-identity.test.ts @@ -0,0 +1,104 @@ +import { describe, expect, it, vi } from 'vitest' +import type { DescendantSnapshot } from '../pty-descendant-termination' +import type { WindowsDescendantSnapshot } from '../windows-descendant-exit-verification' +import { collectDescendantRows } from '../pty-descendant-termination' +import { createClaudeChildTreeReaper } from './claude-agent-sdk-exit-proof' +import { mergeClaudeCapturedTrees } from './claude-child-tree-snapshot' + +function posixSnapshot(capturedAtMs: number): DescendantSnapshot { + return { + root: { pid: 100, startedAt: 'Mon Jan 1 00:00:00 2026' }, + rootPgid: 100, + descendants: [{ pid: 200, ppid: 100, pgid: 100, startedAt: 'Mon Jan 1 00:00:01 2026' }], + capturedAtMs + } +} + +function windowsSnapshot(): WindowsDescendantSnapshot { + return { + root: { pid: 100, creationTimeMs: 5 }, + descendants: [{ pid: 200, creationTimeMs: 7 }], + unidentifiedCount: 0, + capturedAtMs: 1 + } +} + +describe('Claude child root identity', () => { + it('keeps a retained row boundary when a refresh observes no new descendants', () => { + const previous = posixSnapshot(1_700_000_000_900) + const next = posixSnapshot(1_700_000_002_100) + + expect( + mergeClaudeCapturedTrees( + { platform: 'posix', tree: previous }, + { platform: 'posix', tree: next } + ) + ).toEqual({ + platform: 'posix', + tree: { ...next, capturedAtMsByPid: { '200': previous.capturedAtMs } } + }) + }) + + it('keeps the descendant verdict when a POSIX root probe is unavailable', async () => { + const child = { pid: 100, kill: vi.fn(() => true) } + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => posixSnapshot(1)), + terminateDescendants, + verifyRootIdentity: vi.fn(async () => false) + }) + + // POSIX runs no bare-pid root operation, so a declined probe withholds + // nothing: the handle kill still lands and the verification still speaks. + await expect(tree.reap()).resolves.toBe('exited') + expect(terminateDescendants).toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('rejects mixed old and recycled root rows instead of making the tree killable', async () => { + const child = { pid: 100, kill: vi.fn(() => true) } + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => + collectDescendantRows( + 100, + [ + { pid: 100, ppid: 1, pgid: 100, startedAt: 'Mon Jan 1 00:00:00 2026' }, + { pid: 100, ppid: 1, pgid: 101, startedAt: 'Mon Jan 1 00:00:01 2026' }, + { pid: 200, ppid: 100, pgid: 200, startedAt: 'Mon Jan 1 00:00:00 2026' } + ], + 1 + ) + ), + terminateDescendants, + verifyRootIdentity: vi.fn(async () => true) + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + // No admissible snapshot means no row may be signalled from its number, but + // the root still leaves through the handle Node owns. + expect(terminateDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('fails closed when Windows root identity revalidation is unavailable', async () => { + const child = { pid: 100, kill: vi.fn(() => true) } + const terminateWindowsTree = vi.fn(async () => {}) + const terminateWindowsDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants: vi.fn(async () => windowsSnapshot()), + terminateWindowsTree, + terminateWindowsDescendants, + verifyRootIdentity: vi.fn(async () => false) + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + // taskkill /T /F addresses a bare pid and stays gated; the handle does not. + expect(terminateWindowsTree).not.toHaveBeenCalled() + expect(terminateWindowsDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) +}) diff --git a/src/main/claude/claude-agent-sdk-exit-proof.test.ts b/src/main/claude/claude-agent-sdk-exit-proof.test.ts new file mode 100644 index 00000000000..3f15f8e8fca --- /dev/null +++ b/src/main/claude/claude-agent-sdk-exit-proof.test.ts @@ -0,0 +1,934 @@ +import { execFileSync } from 'node:child_process' +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import { describe, expect, it, vi } from 'vitest' +import { spawnProcess, type SpawnedProcess } from '../../shared/child-process/run-process' +import type { DescendantTreeVerdict } from '../pty-descendant-exit-verification' +import type { DescendantSnapshot } from '../pty-descendant-termination' +import type { WindowsDescendantSnapshot } from '../windows-descendant-exit-verification' +import { + createClaudeChildTreeReaper as createClaudeChildTreeReaperImpl, + proveClaudeChildExit, + type ClaudeChildTreeReaper +} from './claude-agent-sdk-exit-proof' + +// The descendant models an MCP server: it either cooperates or, when it traps +// SIGTERM, only a forced, verified sweep can reach it. The root either traps +// SIGTERM too, or leaves promptly on stdin end the way a healthy CLI does — +// which is the path that used to skip descendant proof entirely. +function childWithDescendantScript(input: { + rootTrapsSigterm: boolean + descendantTrapsSigterm: boolean +}): string { + const descendantScript = `${input.descendantTrapsSigterm ? 'process.on("SIGTERM", () => {}); ' : ''}setInterval(() => {}, 1000000)` + const rootBehaviour = input.rootTrapsSigterm + ? `process.on('SIGTERM', () => {}) +process.on('SIGINT', () => {}) +setInterval(() => {}, 1000000)` + : `process.stdin.on('end', () => process.exit(0)) +process.stdin.resume()` + return ` +const descendant = require('node:child_process').spawn( + process.execPath, + ['-e', ${JSON.stringify(descendantScript)}], + { stdio: 'ignore' } +) +descendant.unref() +process.stdout.write(JSON.stringify({ descendantPid: descendant.pid }) + '\\n') +${rootBehaviour} +` +} + +const COOPERATIVE_CHILD = ` +process.stdin.on('end', () => process.exit(0)) +process.stdin.resume() +process.stdout.write('ready\\n') +` + +/** + * Sampled synchronously so it reads the exact moment the close boundary is + * crossed. A zombie has exited (its parent just has not reaped it yet), so a + * kill(pid, 0) probe would misreport it as running. + */ +function descendantState(pid: number): 'running' | 'exited' { + let state: string + try { + state = execFileSync('ps', ['-o', 'state=', '-p', String(pid)], { + encoding: 'utf8', + env: { ...process.env, LANG: 'C', LC_ALL: 'C' } + }).trim() + } catch (error) { + // ps exits 1 when no process matches; anything else is a failed probe, not an answer. + if ((error as { status?: number }).status !== 1) { + throw error + } + return 'exited' + } + return state.startsWith('Z') ? 'exited' : 'running' +} + +/** + * ps lstart is second-resolution, so the identity-safe sweep only SIGKILLs a row + * born strictly before the second the snapshot was captured in. The snapshot is + * armed the moment close begins, so a descendant born in that same second can + * only be asked, never forced — the same bound an MCP server spawned within a + * second of the user closing the chat would hit. + */ +function ageDescendantPastTheCaptureSecond(): Promise { + return new Promise((resolve) => setTimeout(resolve, 1_000 - (Date.now() % 1_000) + 20)) +} + +/** + * The close ladder as production drives it: `closeProcessRegistry` retries an + * unproven close, and each retry re-verifies the retained snapshot. A loaded + * host can spend one attempt's whole window inside `ps`, and reporting false + * there is the honest verdict — the requirement is that TRUE never outruns the + * observation, which the caller asserts at whichever boundary returns it. + */ +async function proveExitWithRetries( + input: Parameters[0], + attempts = 3 +): Promise { + for (let attempt = 1; attempt < attempts; attempt += 1) { + if (await proveClaudeChildExit(input)) { + return true + } + } + return proveClaudeChildExit(input) +} + +function spawnScript(script: string): ReturnType { + return spawnProcess({ + program: process.execPath, + args: ['-e', script], + stdio: ['pipe', 'pipe', 'pipe'] + }) +} + +function firstStdoutLine(child: ReturnType): Promise { + return new Promise((resolve) => { + child.stdout.setEncoding('utf8').once('data', (chunk: string) => resolve(chunk.trim())) + }) +} + +function observeExit(child: EventEmitter): { exitPromise: Promise; exited: () => boolean } { + let exited = false + const exitPromise = new Promise((resolve) => { + child.once('exit', () => { + exited = true + resolve() + }) + }) + return { exitPromise, exited: () => exited } +} + +/** `null` models a spawn that failed before a pid existed. */ +function mockChild( + pid: number | null = 424242 +): EventEmitter & + Pick & { kill: ReturnType } { + const child = new EventEmitter() + return Object.assign(child, { + pid: pid ?? undefined, + stdin: new PassThrough(), + kill: vi.fn(() => true) + }) as never +} + +/** A tree whose verdict is scripted per reap, recording when it was armed. */ +function mockTree(verdicts: DescendantTreeVerdict[]): ClaudeChildTreeReaper & { + capture: ReturnType + reap: ReturnType +} { + let treeVerdict: DescendantTreeVerdict = 'unverifiable' + return { + capture: vi.fn(async () => {}), + reap: vi.fn(async () => { + treeVerdict = verdicts.shift() ?? treeVerdict + return treeVerdict + }), + get treeVerdict() { + return treeVerdict + } + } +} + +function windowsSnapshotOf(descendantPid: number): WindowsDescendantSnapshot { + return { + root: { pid: 424242, creationTimeMs: 1_700_000_000_001 }, + descendants: [{ pid: descendantPid, creationTimeMs: 1_700_000_000_000 }], + unidentifiedCount: 0, + capturedAtMs: 1 + } +} + +function snapshotOf(descendantPid: number): DescendantSnapshot { + return { + root: { pid: 424242, startedAt: 'Mon Jan 1 00:00:00 2026' }, + rootPgid: 1, + descendants: [ + { pid: descendantPid, ppid: 424242, pgid: 1, startedAt: 'Mon Jan 1 00:00:00 2026' } + ], + capturedAtMs: 1 + } +} + +// Unit tests use synthetic process ids; production always supplies the fresh +// identity probe, so the harness explicitly models a matching probe. +function createClaudeChildTreeReaper( + child: Parameters[0], + deps: Parameters[1] = {} +): ReturnType { + return createClaudeChildTreeReaperImpl(child, { + verifyRootIdentity: async () => true, + ...deps + }) +} + +describe('claude child exit proof', () => { + it.runIf(process.platform !== 'win32')( + 'reports a proven exit only once a SIGTERM-resistant descendant is gone at the close boundary', + async () => { + const child = spawnScript( + childWithDescendantScript({ rootTrapsSigterm: true, descendantTrapsSigterm: true }) + ) + const { descendantPid } = JSON.parse(await firstStdoutLine(child)) as { + descendantPid: number + } + expect(descendantState(descendantPid)).toBe('running') + await ageDescendantPastTheCaptureSecond() + + try { + const proven = await proveExitWithRetries({ child, ...observeExit(child) }) + // Evaluated AT the boundary, not by polling until a deferred sweep timer + // wins: true releases the lease, so a descendant still running here is + // exactly the orphan the proof exists to prevent. False would be the + // honest verdict for a tree that outlived the bounded ladder. + expect({ proven, descendant: descendantState(descendantPid) }).toEqual({ + proven: true, + descendant: 'exited' + }) + } finally { + // Failure-safe only: the assertion above owns the requirement, this just + // stops a failing run from leaking a process. + try { + process.kill(descendantPid, 'SIGKILL') + } catch { + // Already gone. + } + } + }, + 20_000 + ) + + it.runIf(process.platform !== 'win32')( + 'proves a promptly exiting root only once its stubborn descendant is gone too', + async () => { + // The ordinary healthy close: the root leaves on stdin end within the graceful + // window. Its descendant must still be proven gone, not assumed gone with it. + const child = spawnScript( + childWithDescendantScript({ rootTrapsSigterm: false, descendantTrapsSigterm: true }) + ) + const { descendantPid } = JSON.parse(await firstStdoutLine(child)) as { + descendantPid: number + } + expect(descendantState(descendantPid)).toBe('running') + await ageDescendantPastTheCaptureSecond() + + try { + const proven = await proveExitWithRetries({ child, ...observeExit(child) }) + expect({ proven, descendant: descendantState(descendantPid) }).toEqual({ + proven: true, + descendant: 'exited' + }) + } finally { + try { + process.kill(descendantPid, 'SIGKILL') + } catch { + // Already gone. + } + } + }, + 20_000 + ) + + it.runIf(process.platform !== 'win32')( + 'still proves a stubborn child whose descendant honours SIGTERM', + async () => { + const child = spawnScript( + childWithDescendantScript({ rootTrapsSigterm: true, descendantTrapsSigterm: false }) + ) + const { descendantPid } = JSON.parse(await firstStdoutLine(child)) as { + descendantPid: number + } + try { + const proven = await proveExitWithRetries({ child, ...observeExit(child) }) + expect({ proven, descendant: descendantState(descendantPid) }).toEqual({ + proven: true, + descendant: 'exited' + }) + } finally { + try { + process.kill(descendantPid, 'SIGKILL') + } catch { + // Already gone. + } + } + }, + 20_000 + ) + + it('arms the snapshot before stdin closes and verifies it after a clean exit', async () => { + const child = spawnScript(COOPERATIVE_CHILD) + expect(await firstStdoutLine(child)).toBe('ready') + const exit = observeExit(child) + const tree = mockTree(['exited']) + let exitedWhenArmed: boolean | null = null + tree.capture.mockImplementation(async () => { + exitedWhenArmed = exit.exited() + }) + + await expect(proveClaudeChildExit({ child, ...exit, tree })).resolves.toBe(true) + // The snapshot is the only proof that survives the root: taken while it lived, + // verified once it left. A reap before the exit would have been the forced ladder. + expect(exitedWhenArmed).toBe(false) + expect(tree.reap).toHaveBeenCalledTimes(1) + expect(exit.exited()).toBe(true) + }, 20_000) + + it('proves a clean close of a childless root with one snapshot and no signal', async () => { + const child = spawnScript(COOPERATIVE_CHILD) + expect(await firstStdoutLine(child)).toBe('ready') + + await expect(proveClaudeChildExit({ child, ...observeExit(child) })).resolves.toBe(true) + }, 20_000) + + it('reports an unprovable exit as false rather than assuming the child died', async () => { + const child = mockChild() + const tree = mockTree(['exited']) + + await expect( + proveClaudeChildExit({ + child, + exitPromise: new Promise(() => {}), + exited: () => false, + tree + }) + ).resolves.toBe(false) + expect(tree.reap).toHaveBeenCalledTimes(1) + }, 20_000) + + it('reports false when the root exit was observed but a descendant was seen alive', async () => { + const child = mockChild() + const exit = observeExit(child) + const tree = mockTree(['live']) + tree.reap.mockImplementation(async () => { + child.emit('exit', null, 'SIGKILL') + return 'live' + }) + + await expect(proveClaudeChildExit({ child, ...exit, tree })).resolves.toBe(false) + expect(exit.exited()).toBe(true) + // One verification per attempt: the retried close re-verifies, this one does not. + expect(tree.reap).toHaveBeenCalledTimes(1) + }, 20_000) + + it('re-verifies an unproven tree on a retried close instead of trusting the dead root', async () => { + const child = mockChild() + const tree = mockTree(['exited']) + + await expect( + proveClaudeChildExit({ child, exitPromise: Promise.resolve(), exited: () => true, tree }) + ).resolves.toBe(true) + expect(tree.reap).toHaveBeenCalledTimes(1) + }) + + it('stays unproven for a root that left before any snapshot could be armed', async () => { + const child = mockChild() + const captureDescendants = vi.fn(async () => snapshotOf(4243)) + const terminateDescendants = vi.fn() + const tree = createClaudeChildTreeReaper(child, { + platform: 'darwin', + exited: () => true, + captureDescendants, + terminateDescendants + }) + + await expect( + proveClaudeChildExit({ child, exitPromise: Promise.resolve(), exited: () => true, tree }) + ).resolves.toBe(false) + // A dead root's descendants have reparented: walking its pid now could only + // sweep a stranger, so no walk is attempted and nothing is proven. + expect(captureDescendants).not.toHaveBeenCalled() + expect(terminateDescendants).not.toHaveBeenCalled() + expect(tree.treeVerdict).toBe('unverifiable') + }) +}) + +describe('claude child tree reaper', () => { + it('kills the root while verification runs and never stops it first', async () => { + const child = mockChild() + const release = Promise.withResolvers() + const terminateDescendants = vi.fn(() => release.promise) + const captureDescendants = vi.fn(async () => snapshotOf(4243)) + const tree = createClaudeChildTreeReaper(child, { + platform: 'darwin', + captureDescendants, + terminateDescendants + }) + + const first = tree.reap() + const second = tree.reap() + await vi.waitFor(() => expect(terminateDescendants).toHaveBeenCalledTimes(1)) + // A stopped root cannot verify: its killed children stay zombie rows in ps. + expect(child.kill.mock.calls).toEqual([['SIGKILL']]) + expect(tree.treeVerdict).toBe('unverifiable') + + release.resolve('exited') + await expect(Promise.all([first, second])).resolves.toEqual(['exited', 'exited']) + expect(captureDescendants).toHaveBeenCalledTimes(1) + expect(tree.treeVerdict).toBe('exited') + }) + + it('re-verifies the retained snapshot on a later reap rather than re-walking a dead root', async () => { + const child = mockChild() + const captureDescendants = vi.fn(async () => snapshotOf(4243)) + const terminateDescendants = vi + .fn() + .mockResolvedValueOnce('live') + .mockResolvedValueOnce('exited') + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('live') + expect(tree.treeVerdict).toBe('live') + await expect(tree.reap()).resolves.toBe('exited') + expect(captureDescendants).toHaveBeenCalledTimes(1) + expect(terminateDescendants).toHaveBeenNthCalledWith(2, snapshotOf(4243)) + expect(tree.treeVerdict).toBe('exited') + }) + + it('keeps an observed exit when a later re-read cannot see the table', async () => { + const child = mockChild() + const terminateDescendants = vi + .fn() + .mockResolvedValueOnce('exited') + .mockResolvedValueOnce('unverifiable') + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => snapshotOf(4243)), + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('exited') + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(tree.treeVerdict).toBe('exited') + }) + + it('keeps an observed live descendant when a later re-read cannot see the table', async () => { + const child = mockChild() + // Reap #1 completed and saw a descendant alive at its deadline; the root then + // left on its own and the re-verification on a loaded host could not read the + // table. "Could not look" must not erase "was seen alive": the lease release + // gate is exactly the pair this distinguishes. + const terminateDescendants = vi + .fn() + .mockResolvedValueOnce('live') + .mockResolvedValueOnce('unverifiable') + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => snapshotOf(4243)), + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('live') + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(tree.treeVerdict).toBe('live') + }) + + it('treats an unreadable process table as unproven and re-walks the live root', async () => { + const child = mockChild() + // A loaded host can miss the table's deadline; while the root still lives + // that is a retryable read, not evidence that it has no descendants. + const captureDescendants = vi + .fn() + .mockResolvedValueOnce(null) + .mockResolvedValueOnce(snapshotOf(4243)) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(terminateDescendants).not.toHaveBeenCalled() + await expect(tree.reap()).resolves.toBe('exited') + expect(captureDescendants).toHaveBeenCalledTimes(2) + }) + + it('does not latch a missing root while it is still live', async () => { + const child = mockChild() + const captureDescendants = vi + .fn() + .mockResolvedValueOnce({ rootPgid: null, descendants: [], capturedAtMs: 1 }) + .mockResolvedValueOnce(snapshotOf(4243)) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await expect(tree.reap()).resolves.toBe('exited') + expect(captureDescendants).toHaveBeenCalledTimes(2) + expect(terminateDescendants).toHaveBeenCalledWith(snapshotOf(4243)) + }) + + it('refreshes the live snapshot at close time so late descendants are included', async () => { + const child = mockChild() + const first = snapshotOf(4243) + const second = { + ...first, + descendants: [...first.descendants, { ...first.descendants[0], pid: 4244 }] + } + const captureDescendants = vi.fn().mockResolvedValueOnce(first).mockResolvedValueOnce(second) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + await tree.reap() + + expect(captureDescendants).toHaveBeenCalledTimes(2) + expect(terminateDescendants).toHaveBeenCalledWith(second) + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('keeps the original capture boundary for retained POSIX rows', async () => { + const child = mockChild() + const first = { + ...snapshotOf(4243), + capturedAtMs: 1_700_000_000_900 + } + const refreshed = { + ...first, + capturedAtMs: 1_700_000_002_100, + descendants: [ + ...first.descendants, + { + pid: 4244, + ppid: 424242, + pgid: 1, + startedAt: 'Tue Jan 2 00:00:00 2026' + } + ] + } + const captureDescendants = vi.fn().mockResolvedValueOnce(first).mockResolvedValueOnce(refreshed) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + await tree.reap() + + expect(terminateDescendants).toHaveBeenCalledWith({ + ...refreshed, + // The retained 4243 row was first observed in the earlier displayed + // second. Its per-row boundary must not advance with the refresh. + capturedAtMsByPid: { + '4243': first.capturedAtMs, + '4244': refreshed.capturedAtMs + } + }) + }) + + it('fails closed when a POSIX refresh reuses a PID with a new identity', async () => { + const child = mockChild() + const first = snapshotOf(4243) + const replacement = { + ...first, + descendants: [ + { + ...first.descendants[0], + pgid: 9, + startedAt: 'Tue Jan 2 00:00:00 2026' + } + ] + } + const captureDescendants = vi + .fn() + .mockResolvedValueOnce(first) + .mockResolvedValueOnce(replacement) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + + await expect(tree.reap()).resolves.toBe('unverifiable') + // The descendant evidence is discarded; the root's identity never was in doubt. + expect(terminateDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('fails closed when a Windows refresh reuses a PID with a new creation time', async () => { + const child = mockChild() + const first = windowsSnapshotOf(4243) + const replacement = { + ...first, + descendants: [{ pid: 4243, creationTimeMs: first.descendants[0].creationTimeMs + 1 }] + } + const captureWindowsDescendants = vi + .fn() + .mockResolvedValueOnce(first) + .mockResolvedValueOnce(replacement) + const terminateWindowsTree = vi.fn(async () => {}) + const terminateWindowsDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants, + terminateWindowsTree, + terminateWindowsDescendants + }) + + await tree.capture() + await tree.refresh?.() + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(terminateWindowsTree).not.toHaveBeenCalled() + expect(terminateWindowsDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('queues a fresh boundary behind an output-triggered capture already in flight', async () => { + const child = mockChild() + const firstDone = Promise.withResolvers() + const first = snapshotOf(4243) + const second = { + ...first, + descendants: [...first.descendants, { ...first.descendants[0], pid: 4244 }] + } + const captureDescendants = vi + .fn() + .mockImplementationOnce(async () => { + await firstDone.promise + return first + }) + .mockResolvedValueOnce(second) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + const outputCapture = tree.refresh!() + await vi.waitFor(() => expect(captureDescendants).toHaveBeenCalledTimes(1)) + const closeCapture = tree.refresh!() + await Promise.resolve() + expect(captureDescendants).toHaveBeenCalledTimes(1) + + firstDone.resolve() + await closeCapture + await tree.reap() + + expect(captureDescendants).toHaveBeenCalledTimes(2) + expect(terminateDescendants).toHaveBeenCalledWith(second) + await outputCapture + }) + + it('retains a replacement descendant when the prior identity exited', async () => { + const child = mockChild() + const first = snapshotOf(4243) + const replacement = snapshotOf(4244) + const captureDescendants = vi + .fn() + .mockResolvedValueOnce(first) + .mockResolvedValueOnce(replacement) + const terminateDescendants = vi.fn(async (snapshot: DescendantSnapshot) => + snapshot.descendants.some((row) => row.pid === 4244) ? ('live' as const) : ('exited' as const) + ) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + await expect(tree.reap()).resolves.toBe('live') + + expect(terminateDescendants).toHaveBeenCalledWith({ + ...replacement, + descendants: [...first.descendants, ...replacement.descendants] + }) + }) + + it('retains a Windows replacement descendant while preserving unidentified rows', async () => { + const child = mockChild() + const first = windowsSnapshotOf(4243) + const replacement = { + ...windowsSnapshotOf(4244), + unidentifiedCount: 0 + } + const captureWindowsDescendants = vi + .fn() + .mockResolvedValueOnce({ ...first, unidentifiedCount: 1 }) + .mockResolvedValueOnce(replacement) + const terminateWindowsTree = vi.fn(async () => {}) + const terminateWindowsDescendants = vi.fn(async (snapshot: WindowsDescendantSnapshot) => + snapshot.descendants.some((row) => row.pid === 4244) ? ('live' as const) : ('exited' as const) + ) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants, + terminateWindowsTree, + terminateWindowsDescendants + }) + + await tree.capture() + await tree.refresh?.() + await expect(tree.reap()).resolves.toBe('live') + + expect(terminateWindowsDescendants).toHaveBeenCalledWith({ + ...replacement, + descendants: [...first.descendants, ...replacement.descendants], + unidentifiedCount: 1 + }) + }) + + it('retains the prior identity-safe snapshot when a refresh is partial', async () => { + const child = mockChild() + const first = { + ...snapshotOf(4243), + descendants: [ + ...snapshotOf(4243).descendants, + { ...snapshotOf(4243).descendants[0], pid: 4244 } + ] + } + const captureDescendants = vi + .fn() + .mockResolvedValueOnce(first) + .mockResolvedValueOnce({ + ...first, + descendants: first.descendants.slice(0, 1) + }) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + await tree.reap() + + expect(terminateDescendants).toHaveBeenCalledWith(first) + }) + + it('stops re-walking once the root is gone, however the table behaved', async () => { + const child = mockChild() + let exited = false + const captureDescendants = vi.fn(async () => null) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => exited, + captureDescendants, + terminateDescendants: vi.fn() + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + // An unreadable table costs the snapshot, never the kill on the live root. + expect(child.kill).toHaveBeenCalledTimes(1) + exited = true + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(captureDescendants).toHaveBeenCalledTimes(1) + // The second attempt observes a dead root: Node has dropped the handle, so + // there is nothing left to signal and no recycled pid to reach. + expect(child.kill).toHaveBeenCalledTimes(1) + }) + + it('discards a walk that found no root instead of proving an empty tree', async () => { + const child = mockChild() + const captureDescendants = vi.fn(async () => ({ + rootPgid: null, + descendants: [], + capturedAtMs: 1 + })) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants: vi.fn() + }) + + await tree.capture() + await expect(tree.reap()).resolves.toBe('unverifiable') + // A vacuous walk remains retryable while the root is live; no empty-tree + // verdict is latched from a missing root row. + expect(captureDescendants).toHaveBeenCalledTimes(2) + }) + + it('discards a walk that raced the root exit instead of proving an empty tree', async () => { + const child = mockChild() + let exited = false + const captureDescendants = vi.fn(async () => { + exited = true + return { rootPgid: 1, descendants: [], capturedAtMs: 1 } + }) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => exited, + captureDescendants, + terminateDescendants: vi.fn() + }) + + await tree.capture() + await expect(tree.reap()).resolves.toBe('unverifiable') + }) + + it('proves a childless snapshot without signalling anything', async () => { + const child = mockChild() + const terminateDescendants = vi.fn() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => ({ + root: { pid: 424242, startedAt: 'Mon Jan 1 00:00:00 2026' }, + rootPgid: 1, + descendants: [], + capturedAtMs: 1 + })), + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('exited') + expect(terminateDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('waits for the Windows tree kill before releasing the root', async () => { + const child = mockChild() + const release = Promise.withResolvers() + const terminateWindowsTree = vi.fn(() => release.promise) + const captureDescendants = vi.fn() + const terminateWindowsDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureDescendants, + captureWindowsDescendants: vi.fn(async () => windowsSnapshotOf(4243)), + terminateWindowsTree, + terminateWindowsDescendants + }) + + const reap = tree.reap() + await vi.waitFor(() => + expect(terminateWindowsTree).toHaveBeenCalledWith({ + pid: 424242, + creationTimeMs: 1_700_000_000_001 + }) + ) + expect(child.kill).not.toHaveBeenCalled() + expect(terminateWindowsDescendants).not.toHaveBeenCalled() + release.resolve() + await expect(reap).resolves.toBe('exited') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + expect(terminateWindowsDescendants).toHaveBeenCalledWith(windowsSnapshotOf(4243)) + expect(captureDescendants).not.toHaveBeenCalled() + }) + + it('stays unproven on Windows when taskkill fails and a descendant is still observed', async () => { + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants: vi.fn(async () => windowsSnapshotOf(4243)), + terminateWindowsTree: vi.fn(async () => { + throw new Error('taskkill: access denied') + }), + terminateWindowsDescendants: vi.fn(async () => 'live' as const) + }) + + // taskkill's own outcome is not the proof; the table read after it is. + await expect(tree.reap()).resolves.toBe('live') + expect(tree.treeVerdict).toBe('live') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('stays unproven on Windows when taskkill resolves but a descendant survives it', async () => { + const child = mockChild() + const terminateWindowsTree = vi.fn(async () => {}) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants: vi.fn(async () => windowsSnapshotOf(4243)), + terminateWindowsTree, + terminateWindowsDescendants: vi.fn(async () => 'live' as const) + }) + + await expect(tree.reap()).resolves.toBe('live') + expect(terminateWindowsTree).toHaveBeenCalledTimes(1) + expect(tree.treeVerdict).toBe('live') + }) + + it('never taskkills a Windows root that already exited, but still verifies its snapshot', async () => { + const child = mockChild() + let exited = false + const terminateWindowsTree = vi.fn(async () => {}) + const terminateWindowsDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + exited: () => exited, + captureWindowsDescendants: vi.fn(async () => windowsSnapshotOf(4243)), + terminateWindowsTree, + terminateWindowsDescendants + }) + + await tree.capture() + exited = true + await expect(tree.reap()).resolves.toBe('exited') + // A dead root's pid may already belong to a stranger: taskkill /T /F on it + // would take down an unrelated tree. + expect(terminateWindowsTree).not.toHaveBeenCalled() + expect(terminateWindowsDescendants).toHaveBeenCalledWith(windowsSnapshotOf(4243)) + }) + + it('treats an unreadable Windows table as unproven', async () => { + const child = mockChild() + const terminateWindowsDescendants = vi.fn() + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants: vi.fn(async () => null), + terminateWindowsTree: vi.fn(async () => {}), + terminateWindowsDescendants + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(terminateWindowsDescendants).not.toHaveBeenCalled() + // A host that cannot supply creation times blocks taskkill, not the root kill. + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('has nothing to reap for a child that never spawned', async () => { + const child = mockChild(null) + const captureDescendants = vi.fn() + const tree = createClaudeChildTreeReaper(child, { platform: 'linux', captureDescendants }) + + await expect(tree.reap()).resolves.toBe('exited') + expect(captureDescendants).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/claude/claude-agent-sdk-exit-proof.ts b/src/main/claude/claude-agent-sdk-exit-proof.ts new file mode 100644 index 00000000000..17533f87a70 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-exit-proof.ts @@ -0,0 +1,366 @@ +import type { SpawnedProcess } from '../../shared/child-process/run-process' +import { + terminateDescendantSnapshotWithVerdict, + type DescendantTreeVerdict +} from '../pty-descendant-exit-verification' +import { + captureDescendantSnapshot, + type DescendantSnapshot, + type PosixProcessIdentity +} from '../pty-descendant-termination' +import { + captureWindowsDescendantSnapshot, + terminateIdentifiedWindowsProcessTree, + verifyWindowsDescendantSnapshotExit, + verifyWindowsProcessIdentity, + type WindowsDescendantSnapshot, + type WindowsProcessIdentity +} from '../windows-descendant-exit-verification' +import { mergeClaudeCapturedTrees, type ClaudeCapturedTree } from './claude-child-tree-snapshot' +import { terminateClaudeRoot, terminateClaudeWindowsRoot } from './claude-child-root-termination' +import { + proveClaudeChildExitWithReaper, + type ClaudeChildExitProofInput +} from './claude-child-exit-proof-ladder' + +/** + * A later reap may only raise the latched verdict. An observed exit is final, and + * a descendant seen alive at a deadline is never forgotten by a later look that + * could not read the table: the lease gate discriminates on exactly that pair. + */ +const TREE_VERDICT_TRUST: Record = { + unverifiable: 0, + live: 1, + exited: 2 +} + +type ReapableChild = Pick + +/** + * A walk is only admissible while the root it walked was alive. A POSIX walk + * that found no root says so with a null pgid; either platform's walk can also + * have raced the root's death. Both can only have missed descendants that + * already reparented away, so neither is evidence about the tree. + */ +function admissibleTree( + captured: DescendantSnapshot | WindowsDescendantSnapshot | null, + platform: NodeJS.Platform, + exited: boolean +): ClaudeCapturedTree | null { + if (!captured || exited) { + return null + } + if (platform === 'win32') { + return { platform: 'win32', tree: captured as WindowsDescendantSnapshot } + } + const tree = captured as DescendantSnapshot + return tree.rootPgid === null ? null : { platform: 'posix', tree } +} + +export type ClaudeChildTreeReaperDeps = { + platform?: NodeJS.Platform + /** Whether the root's exit has been observed; only a live root can be walked. */ + exited?: () => boolean + captureDescendants?: (rootPid: number) => Promise + terminateDescendants?: (snapshot: DescendantSnapshot) => Promise + terminateWindowsTree?: (root: WindowsProcessIdentity) => Promise + captureWindowsDescendants?: (rootPid: number) => Promise + terminateWindowsDescendants?: ( + snapshot: WindowsDescendantSnapshot + ) => Promise + /** Identity probe for the bare-pid tree kill; only Windows has one to gate. */ + verifyRootIdentity?: (root: PosixProcessIdentity | WindowsProcessIdentity) => Promise +} + +export type ClaudeChildTreeReaper = { + /** + * Snapshot the root's live descendants. The moment the root dies they reparent + * and no table walk can find them again, so this has to run before anything + * gives the root a reason to leave. Held once; later calls are no-ops. + */ + capture(): Promise + /** Refresh a live root's snapshot at the close boundary; a failed refresh keeps the prior proof. */ + refresh?: () => Promise + /** + * Kill the child's whole tree and report what the bounded verification + * observed. Concurrent calls share one reap, and a later call re-verifies the + * same snapshot rather than trusting a root that has since died on its own. + */ + reap(): Promise + /** + * `unverifiable` until a reap observes otherwise. `exited` is the only verdict + * that lets a close release the lease; `live` names a descendant that was seen + * still running, which no later caller may collapse into "unknown". + */ + readonly treeVerdict: DescendantTreeVerdict +} + +/** + * The same shared primitives the Codex structured provider composes: a raw + * pipe child owns no PTY job, so there is nothing for the PTY job sweep to + * terminate on Windows and no unref'd timer is allowed to outlive the proof. + * + * The proof is unproven by default. `treeVerdict` is assigned in exactly one + * place, from the verdict of `judgeTree`, so a code path that never reaches a + * verification cannot report the tree gone by omission. + */ +export function createClaudeChildTreeReaper( + child: ReapableChild, + deps: ClaudeChildTreeReaperDeps = {} +): ClaudeChildTreeReaper { + const platform = deps.platform ?? process.platform + const exited = deps.exited ?? (() => false) + // Undefined until captured; null when no admissible snapshot exists — the root + // was already gone, or the table could not be read while it was alive — which + // no later read can make up for. + let snapshot: ClaudeCapturedTree | null | undefined + let capturing: Promise | null = null + let refreshing: Promise | null = null + let queuedRefresh: Promise | null = null + let inFlight: Promise | null = null + let treeVerdict: DescendantTreeVerdict = 'unverifiable' + + // Consulted only on win32: POSIX signals descendants by revalidated identity + // and reaches the root solely through Node's handle, so neither needs a probe. + const verifyRoot = + deps.verifyRootIdentity ?? + ((root: PosixProcessIdentity | WindowsProcessIdentity) => + verifyWindowsProcessIdentity(root as WindowsProcessIdentity)) + + function captureOnce(): Promise { + if (refreshing) { + const pending = refreshing + return pending.then(() => queuedRefresh ?? undefined) + } + if (snapshot !== undefined) { + return Promise.resolve() + } + if (capturing) { + const pending = capturing + return pending.then(() => queuedRefresh ?? undefined) + } + const rootPid = child.pid + if (!rootPid || exited()) { + // Only the root's death makes a missing snapshot final: its descendants + // have reparented, and no later walk can reach them. + snapshot = exited() ? null : snapshot + return Promise.resolve() + } + const capture = + platform === 'win32' + ? (deps.captureWindowsDescendants ?? captureWindowsDescendantSnapshot) + : (deps.captureDescendants ?? captureDescendantSnapshot) + capturing = capture(rootPid) + .catch(() => null) + .then((captured) => { + // A walk that found no root, or that raced the root's death, can only + // have missed descendants that already reparented away. A table that + // could not be read in time is not an answer at all: while the root + // still lives the walk is simply retried, rather than latching a failed + // read as proof that there was nothing to find. + const rootExited = exited() + const tree = admissibleTree(captured, platform, rootExited) + if (tree) { + snapshot = tree + } else if (rootExited) { + // Once the root has exited its descendants may have reparented; no + // later table read can make an absent snapshot safe to signal. + snapshot = null + } else { + // A failed read or a walk that did not observe the live root is + // retryable while the root remains alive. Never latch a vacuous null. + snapshot = undefined + } + }) + .finally(() => { + capturing = null + }) + return capturing + } + + function startRefresh(): Promise { + if (exited()) { + return Promise.resolve() + } + const rootPid = child.pid + if (!rootPid) { + return Promise.resolve() + } + const capture = + platform === 'win32' + ? (deps.captureWindowsDescendants ?? captureWindowsDescendantSnapshot) + : (deps.captureDescendants ?? captureDescendantSnapshot) + const operation = (async () => { + const captured = await capture(rootPid).catch(() => null) + if (exited()) { + return + } + const tree = admissibleTree(captured, platform, false) + if (!tree) { + return + } + if (snapshot === undefined) { + snapshot = tree + return + } + if (snapshot !== null) { + // A merge that returns null saw a same-PID identity change: a + // recycle/replace decision, not an absent descendant, so no row here may + // be signalled from its number. Only the descendant evidence is lost — + // the root still leaves through the handle no recycled pid can reach. + snapshot = mergeClaudeCapturedTrees(snapshot, tree) + } + // Keep an earlier admissible snapshot when this close-boundary read fails; + // it remains the only identity-safe evidence after root exit. + })() + refreshing = operation + const clearRefreshing = (): void => { + if (refreshing === operation) { + refreshing = null + } + } + void operation.then(clearRefreshing, clearRefreshing) + return operation + } + + function queueRefreshAfter(pending: Promise): Promise { + if (queuedRefresh) { + return queuedRefresh + } + const operation = pending.then(() => { + if (exited()) { + return + } + return startRefresh() + }) + queuedRefresh = operation + const clearQueuedRefresh = (): void => { + if (queuedRefresh === operation) { + queuedRefresh = null + } + } + void operation.then(clearQueuedRefresh, clearQueuedRefresh) + return operation + } + + async function refresh(): Promise { + const pending = capturing ?? refreshing + if (pending) { + await queueRefreshAfter(pending) + return + } + if (queuedRefresh) { + await queuedRefresh + return + } + try { + await startRefresh() + } catch { + // A refresh is advisory; capture failures leave the prior proof intact. + } + } + + /** The only source of a tree verdict: every `exited` here is an observation. */ + async function judgeTree(): Promise { + const killRoot = (): boolean => terminateClaudeRoot({ child, exited }) + const rootPid = child.pid + if (!rootPid) { + // Never spawned, so the OS never created a tree to orphan. + return 'exited' + } + await captureOnce() + if (platform === 'win32') { + // Why taskkill's own outcome is never the verdict: it resolves identically + // on a timeout, an access denial, a recycled root and a real kill. + const { rootVerified } = await terminateClaudeWindowsRoot({ + snapshot: snapshot?.platform === 'win32' ? snapshot.tree : null, + exited, + verifyRoot: (root) => verifyRoot(root), + terminateTree: (root) => + deps.terminateWindowsTree + ? deps.terminateWindowsTree(root) + : terminateIdentifiedWindowsProcessTree(root, { + ownsRoot: () => !exited() + }).then(() => undefined), + killRoot + }) + if (!rootVerified && !exited()) { + return 'unverifiable' + } + return snapshot?.platform === 'win32' + ? await (deps.terminateWindowsDescendants ?? verifyWindowsDescendantSnapshotExit)( + snapshot.tree + ) + : 'unverifiable' + } + if (snapshot?.platform !== 'posix') { + killRoot() + return 'unverifiable' + } + if (snapshot.tree.descendants.length === 0) { + // Read while the root was alive and childless: a later table read has no + // row it could match, so it would add nothing to this observation. + killRoot() + return 'exited' + } + // Why the root is killed while verification is already running, and never + // SIGSTOPped first the way the Codex non-group path does: measured on macOS, a + // killed child of a stopped parent stays a zombie row in ps with its lstart + // and pgid intact, so verification cannot pass until the root is dead. The + // descendants are signalled by the verifier as soon as it revalidates their + // identities; the root's death then reparents any zombies to init, which + // reaps them. After a root exit the kill is a no-op: Node drops the handle + // on exit and never signals a possibly recycled pid. + const verdictPromise = deps.terminateDescendants + ? deps.terminateDescendants(snapshot.tree) + : terminateDescendantSnapshotWithVerdict(snapshot.tree, { + requireIdentityBeforeSignal: true + }) + killRoot() + // What the verification observed is the verdict: a kill that reports no + // signal means the handle was already gone, never that the tree survived. + return verdictPromise + } + + return { + capture: captureOnce, + refresh, + reap() { + if (inFlight) { + return inFlight + } + const attempt = judgeTree() + .catch((): DescendantTreeVerdict => 'unverifiable') + .then((verdict) => { + treeVerdict = + TREE_VERDICT_TRUST[verdict] > TREE_VERDICT_TRUST[treeVerdict] ? verdict : treeVerdict + return verdict + }) + inFlight = attempt + void attempt.finally(() => { + if (inFlight === attempt) { + inFlight = null + } + }) + return attempt + }, + get treeVerdict() { + return treeVerdict + } + } +} + +/** + * Orca's own shutdown ladder on the child it spawned, kept because the SDK's + * close path returns no proof and Orca never releases a lease on an assumed exit. + * + * Resolves true only after the child actually emitted exit and its snapshotted + * descendants were observed gone; false is unproven. A root that left on its + * own before a snapshot could be armed stays unproven: its descendants had + * already reparented out of reach when the ladder first looked. + */ +export function proveClaudeChildExit(input: ClaudeChildExitProofInput): Promise { + return proveClaudeChildExitWithReaper(input, () => + createClaudeChildTreeReaper(input.child, { exited: input.exited }) + ) +} diff --git a/src/main/claude/claude-agent-sdk-import-boundary.test.ts b/src/main/claude/claude-agent-sdk-import-boundary.test.ts new file mode 100644 index 00000000000..f1a38466d41 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-import-boundary.test.ts @@ -0,0 +1,154 @@ +import { existsSync, readFileSync, statSync } from 'node:fs' +import { dirname, join, relative, resolve } from 'node:path' +import { describe, expect, it } from 'vitest' +import { spawnProcess } from '../../shared/child-process/run-process' + +/** + * Keep the agent SDK on the structured-Claude side of the toggle. + * + * A user who never leaves the terminal/TUI Claude path must not pay for the SDK: + * importing it evaluates a package that rewrites + * `process.env.NoDefaultCurrentDirectoryInExePath`, changing how Windows resolves + * executables for every later subprocess, and a missing or incompatible install + * would take normal runtime startup down with it. The ordinary + * `OrcaRuntimeService` graph reaches the Claude transport module, so only a + * deferred import keeps that boundary — and only a walk of the real import graph + * keeps the next static import from quietly restoring it. + */ +const SDK_PACKAGE = '@anthropic-ai/claude-agent-sdk' +const REPO_ROOT = resolve(__dirname, '..', '..', '..') + +/** The Electron main entry: everything the app loads before any session exists. */ +const ROOT = 'src/main/index.ts' +/** Proof the walk goes all the way into the Claude transport rather than stopping short. */ +const TRANSPORT_MODULE = 'src/main/claude/claude-stream-json-connection.ts' + +/** + * Static, value-carrying specifiers only, read statement by statement so a + * multi-line `import { ... } from '...'` counts. `import type` is erased before + * the module ever loads and a bare `import(...)` is the deferral this guards, so + * neither is an edge the runtime traverses at load time. + */ +const STATEMENT_START = /^\s*(?:import|export)\b/ +const TYPE_ONLY = /^\s*(?:import|export)\s+type\b/ +const FROM_SPECIFIER = /(?:^|\s)from\s*['"]([^'"]+)['"]/ +const SIDE_EFFECT_IMPORT = /^\s*import\s*['"]([^'"]+)['"]/ +/** An import statement never spans more lines than its longest specifier list. */ +const MAX_STATEMENT_LINES = 60 + +function readSpecifiers(source: string): string[] { + const lines = source.split('\n') + const found: string[] = [] + for (let index = 0; index < lines.length; index += 1) { + const first = lines[index] as string + if (!STATEMENT_START.test(first) || TYPE_ONLY.test(first)) { + continue + } + const sideEffect = SIDE_EFFECT_IMPORT.exec(first) + if (sideEffect) { + found.push(sideEffect[1] as string) + continue + } + for (let scan = index; scan < Math.min(lines.length, index + MAX_STATEMENT_LINES); scan += 1) { + if (scan > index && STATEMENT_START.test(lines[scan] as string)) { + break + } + const specifier = FROM_SPECIFIER.exec(lines[scan] as string) + if (specifier) { + found.push(specifier[1] as string) + break + } + } + } + return found +} + +/** Resolve a relative specifier the way the bundler does; unresolvable means not a module. */ +function resolveRelative(fromFile: string, specifier: string): string | null { + const base = join(dirname(fromFile), specifier) + for (const candidate of [base, `${base}.ts`, `${base}.tsx`, join(base, 'index.ts')]) { + if (existsSync(candidate) && statSync(candidate).isFile()) { + return candidate + } + } + return null +} + +function walkStaticImports(rootFile: string): { visited: Set; sdkImporters: string[] } { + const visited = new Set() + const sdkImporters: string[] = [] + const queue = [resolve(REPO_ROOT, rootFile)] + while (queue.length > 0) { + const file = queue.pop() as string + const key = relative(REPO_ROOT, file).split('\\').join('/') + if (visited.has(key)) { + continue + } + visited.add(key) + for (const specifier of readSpecifiers(readFileSync(file, 'utf8'))) { + if (specifier === SDK_PACKAGE || specifier.startsWith(`${SDK_PACKAGE}/`)) { + sdkImporters.push(key) + continue + } + if (!specifier.startsWith('.')) { + continue + } + const target = resolveRelative(file, specifier) + if (target) { + queue.push(target) + } + } + } + return { visited, sdkImporters } +} + +describe('claude agent SDK import boundary', () => { + const walk = walkStaticImports(ROOT) + + it('walks a graph deep enough to reach the Claude transport', () => { + // Without this the guard passes for the wrong reason the moment the walk breaks. + expect(walk.visited.size).toBeGreaterThan(500) + expect([...walk.visited]).toContain(TRANSPORT_MODULE) + }) + + it('never reaches the SDK through a static import from the main entry', () => { + expect( + walk.sdkImporters, + `${SDK_PACKAGE} must stay behind the structured-Claude boundary. Load it with a deferred import inside the session path instead.` + ).toEqual([]) + }) + + it('leaves the Windows executable-search environment alone when the runtime loads', async () => { + // A vitest file runs in its own fork, so this is a clean process; the ambient + // value is cleared first because the developer's own shell may carry one. + delete process.env.NoDefaultCurrentDirectoryInExePath + await import('../runtime/structured-agent-session-runtime') + + expect(process.env.NoDefaultCurrentDirectoryInExePath).toBeUndefined() + }) + + it('still lets the SDK set it, so the guard above is not measuring nothing', async () => { + // A separate process, not this fork: the assertion has to be about a first + // evaluation of the package, which a cached module registry cannot give. + const { NoDefaultCurrentDirectoryInExePath: _cleared, ...env } = process.env + const probe = spawnProcess({ + program: process.execPath, + args: [ + '-e', + `import(${JSON.stringify(SDK_PACKAGE)}).then(() => console.log(String(process.env.NoDefaultCurrentDirectoryInExePath)))` + ], + cwd: REPO_ROOT, + env: env as Record, + stdio: ['ignore', 'pipe', 'ignore'] + }) + const observed = await new Promise((settle) => { + let output = '' + probe.stdout?.setEncoding('utf8').on('data', (chunk: string) => { + output += chunk + }) + probe.once('close', () => settle(output.trim())) + }) + + expect(observed).toBe('1') + }) +}) diff --git a/src/main/claude/claude-agent-sdk-process-spawn.test.ts b/src/main/claude/claude-agent-sdk-process-spawn.test.ts new file mode 100644 index 00000000000..cd3520cf6d5 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-process-spawn.test.ts @@ -0,0 +1,107 @@ +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import { describe, expect, it, vi } from 'vitest' +import type { SpawnOptions as SdkSpawnOptions } from '@anthropic-ai/claude-agent-sdk' +import { resolveSpawn, type spawnProcess } from '../../shared/child-process/run-process' +import type { ProcessSpec } from '../../shared/child-process/process-spec' +import { createClaudeCodeProcessSpawn } from './claude-agent-sdk-process-spawn' + +type FakeChild = EventEmitter & { + pid: number + stdin: PassThrough + stdout: PassThrough + stderr: PassThrough + kill: ReturnType +} + +function fakeSpawn() { + const child = new EventEmitter() as FakeChild + child.pid = 4321 + child.stdin = new PassThrough() + child.stdout = new PassThrough() + child.stderr = new PassThrough() + child.kill = vi.fn(() => true) + const specs: ProcessSpec[] = [] + const spawnImpl = ((spec: ProcessSpec) => { + specs.push(spec) + return child + }) as unknown as typeof spawnProcess + return { child, spawnImpl, specs } +} + +function sdkOptions(overrides: Partial = {}): SdkSpawnOptions { + return { + command: '/usr/local/bin/claude', + args: ['--output-format', 'stream-json'], + cwd: '/work/repo', + env: { PATH: '/usr/bin', CLAUDE_CONFIG_DIR: '/accounts/one', UNSET: undefined }, + signal: new AbortController().signal, + ...overrides + } +} + +describe('claude agent SDK process spawn', () => { + it('routes the SDK spawn through Orca and retains the pid the lease adjudicates on', () => { + const process = fakeSpawn() + const spawn = createClaudeCodeProcessSpawn(process.spawnImpl) + + expect(spawn.pid).toBeUndefined() + expect(spawn.child).toBeNull() + const child = spawn.spawn(sdkOptions()) + + expect(child).toBe(process.child) + expect(spawn.child).toBe(process.child) + expect(spawn.pid).toBe(4321) + expect(process.specs[0]).toEqual({ + program: '/usr/local/bin/claude', + args: ['--output-format', 'stream-json'], + cwd: '/work/repo', + env: { PATH: '/usr/bin', CLAUDE_CONFIG_DIR: '/accounts/one' }, + stdio: ['pipe', 'pipe', 'pipe'] + }) + }) + + it('keeps the child out of the SDK abort path so exit proof stays Orca-owned', () => { + const process = fakeSpawn() + const controller = new AbortController() + createClaudeCodeProcessSpawn(process.spawnImpl).spawn(sdkOptions({ signal: controller.signal })) + + // Node's spawn({signal}) kills the child on abort; Orca's ladder must be the + // only thing that can end this process, or close() would report an assumed exit. + expect(process.specs[0]).not.toHaveProperty('signal') + }) + + it('drains stderr into a bounded tail so an exit error still carries it', async () => { + const process = fakeSpawn() + const spawn = createClaudeCodeProcessSpawn(process.spawnImpl) + spawn.spawn(sdkOptions()) + + process.child.stderr.write('x'.repeat(9000)) + process.child.stderr.write('claude: not signed in') + await new Promise((resolve) => setImmediate(resolve)) + + expect(spawn.stderrTail).toMatch(/claude: not signed in$/) + expect(spawn.stderrTail.length).toBe(8192) + }) + + it('hands a Windows .cmd shim to Orca\u2019s argument encoder', () => { + const process = fakeSpawn() + createClaudeCodeProcessSpawn(process.spawnImpl).spawn( + sdkOptions({ + command: 'C:\\Users\\dev\\AppData\\npm\\claude.cmd', + args: ['--setting-sources=user,project,local', '--session-id', 'a b&c'] + }) + ) + + // The spec the spawner builds is what Orca's Windows branch encodes; the SDK's + // own spawn would hand `.cmd` straight to Node and mangle the argument. + const resolved = resolveSpawn(process.specs[0] as ProcessSpec, 'win32') + expect(resolved.file.toLowerCase()).toContain('cmd.exe') + expect(resolved.options.windowsVerbatimArguments).toBe(true) + expect(resolved.args).toHaveLength(1) + // `/v:off` plus the quoted argument is what keeps `&` from splitting the line. + expect(resolved.args[0]).toContain('/v:off') + expect(resolved.args[0]).toContain('"a b&c"') + expect(resolved.args[0]).toContain('"--setting-sources=user,project,local"') + }) +}) diff --git a/src/main/claude/claude-agent-sdk-process-spawn.ts b/src/main/claude/claude-agent-sdk-process-spawn.ts new file mode 100644 index 00000000000..a2b1ad7f158 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-process-spawn.ts @@ -0,0 +1,69 @@ +import type { SpawnOptions as ClaudeAgentSdkSpawnOptions } from '@anthropic-ai/claude-agent-sdk' +import { spawnProcess } from '../../shared/child-process/run-process' + +/** Derived rather than imported: only src/shared/child-process may name node:child_process. */ +type ClaudeCodeChild = ReturnType + +const STDERR_TAIL_MAX_BYTES = 8192 + +export type ClaudeCodeProcessSpawn = { + /** Pass as the SDK's `spawnClaudeCodeProcess`; the SDK never learns the pid because it never owns it. */ + spawn: (options: ClaudeAgentSdkSpawnOptions) => ClaudeCodeChild + /** The retained child, so Orca keeps its own tree-kill and exit-proof ladder. Null until the SDK spawns. */ + readonly child: ClaudeCodeChild | null + /** Ownership proof: the durable lease adjudicates on this pid plus start time plus the spawn token. */ + readonly pid: number | undefined + readonly stderrTail: string +} + +function definedEnv(env: Record): Record { + const next: Record = {} + for (const [key, value] of Object.entries(env)) { + if (value !== undefined) { + next[key] = value + } + } + return next +} + +/** + * Orca supplies the Claude Code child rather than letting the SDK spawn it. + * + * Two independent reasons: the SDK's `SpawnedProcess` has no pid, and Orca's + * spawner is the only path that encodes `.cmd` arguments safely on Windows. + */ +export function createClaudeCodeProcessSpawn( + spawnImpl: typeof spawnProcess = spawnProcess +): ClaudeCodeProcessSpawn { + let child: ClaudeCodeChild | null = null + let stderrTail = '' + return { + spawn: (options) => { + // Why `options.signal` is dropped: it would let the SDK kill the child outside + // Orca's ladder, and close() may never report an exit it did not observe. + const spawned = spawnImpl({ + program: options.command, + args: [...options.args], + ...(options.cwd === undefined ? {} : { cwd: options.cwd }), + env: definedEnv(options.env), + stdio: ['pipe', 'pipe', 'pipe'] + }) + child = spawned + // The SDK drains stderr only for its own local spawn, so a custom spawner must: + // otherwise the child blocks on a full pipe and exit errors lose their tail. + spawned.stderr.setEncoding('utf8').on('data', (chunk: string) => { + stderrTail = (stderrTail + chunk).slice(-STDERR_TAIL_MAX_BYTES) + }) + return spawned + }, + get child() { + return child + }, + get pid() { + return child?.pid + }, + get stderrTail() { + return stderrTail + } + } +} diff --git a/src/main/claude/claude-agent-sdk-root-kill-fallback.test.ts b/src/main/claude/claude-agent-sdk-root-kill-fallback.test.ts new file mode 100644 index 00000000000..9f843527e30 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-root-kill-fallback.test.ts @@ -0,0 +1,190 @@ +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import { describe, expect, it, vi } from 'vitest' +import type { SpawnedProcess } from '../../shared/child-process/run-process' +import type { DescendantSnapshot } from '../pty-descendant-termination' +import type { WindowsDescendantSnapshot } from '../windows-descendant-exit-verification' +import { createClaudeChildTreeReaper } from './claude-agent-sdk-exit-proof' +import { mergeClaudeCapturedTrees } from './claude-child-tree-snapshot' + +const ROOT_PID = 424242 +const ROOT_STARTED_AT = 'Mon Jan 1 00:00:00 2026' +const ROOT_FORK_MS = Date.parse(ROOT_STARTED_AT) + +function mockChild(): EventEmitter & + Pick & { kill: ReturnType } { + return Object.assign(new EventEmitter(), { + pid: ROOT_PID, + stdin: new PassThrough(), + kill: vi.fn(() => true) + }) as never +} + +function posixSnapshot(input: { + capturedAtMs: number + descendants?: DescendantSnapshot['descendants'] +}): DescendantSnapshot { + return { + root: { pid: ROOT_PID, startedAt: ROOT_STARTED_AT }, + rootPgid: ROOT_PID, + descendants: input.descendants ?? [], + capturedAtMs: input.capturedAtMs + } +} + +function windowsSnapshot(capturedAtMs = 1): WindowsDescendantSnapshot { + return { + root: { pid: ROOT_PID, creationTimeMs: 1_700_000_000_001 }, + descendants: [{ pid: 4243, creationTimeMs: 1_700_000_000_000 }], + unidentifiedCount: 0, + capturedAtMs + } +} + +describe('Claude root kill fallback', () => { + it('kills the root when the first capture landed in the fork second', async () => { + // The production POSIX verifier declines a root born in its capture second, + // and that verdict must not cost the tree the kill on Node's own handle. + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => false, + captureDescendants: vi.fn(async () => posixSnapshot({ capturedAtMs: ROOT_FORK_MS + 300 })) + }) + + await expect(tree.reap()).resolves.toBe('exited') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('kills the root after a recycled descendant pid voided the snapshot', async () => { + const child = mockChild() + const captureDescendants = vi + .fn() + .mockResolvedValueOnce( + posixSnapshot({ + capturedAtMs: ROOT_FORK_MS + 5_000, + descendants: [{ pid: 100, ppid: ROOT_PID, pgid: ROOT_PID, startedAt: ROOT_STARTED_AT }] + }) + ) + .mockResolvedValueOnce( + posixSnapshot({ + capturedAtMs: ROOT_FORK_MS + 6_000, + descendants: [ + { pid: 100, ppid: ROOT_PID, pgid: ROOT_PID, startedAt: 'Mon Jan 1 00:00:30 2026' } + ] + }) + ) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => false, + captureDescendants, + terminateDescendants: vi.fn(async () => 'exited' as const), + verifyRootIdentity: vi.fn(async () => true) + }) + + await tree.capture() + await tree.refresh?.() + // The descendant evidence is rightly discarded; the root's never was in doubt. + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('keeps an observed live descendant when the root identity probe declined', async () => { + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => false, + captureDescendants: vi.fn(async () => + posixSnapshot({ + capturedAtMs: ROOT_FORK_MS + 5_000, + descendants: [{ pid: 100, ppid: ROOT_PID, pgid: ROOT_PID, startedAt: ROOT_STARTED_AT }] + }) + ), + terminateDescendants: vi.fn(async () => 'live' as const), + verifyRootIdentity: vi.fn(async () => false) + }) + + await expect(tree.reap()).resolves.toBe('live') + expect(tree.treeVerdict).toBe('live') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('reports a Windows taskkill that worked as exited, not unverifiable', async () => { + const child = mockChild() + // Probe 1 gates taskkill; a later probe correctly finds the root already dead. + const verifyRootIdentity = vi.fn().mockResolvedValueOnce(true).mockResolvedValue(false) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + exited: () => false, + captureWindowsDescendants: vi.fn(async () => windowsSnapshot()), + terminateWindowsTree: vi.fn(async () => {}), + terminateWindowsDescendants: vi.fn(async () => 'exited' as const), + verifyRootIdentity + }) + + await expect(tree.reap()).resolves.toBe('exited') + }) + + it('kills the root when no POSIX snapshot could be read', async () => { + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => false, + captureDescendants: vi.fn(async () => null), + terminateDescendants: vi.fn() + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('kills the root when the Windows process table is unreadable', async () => { + const child = mockChild() + const terminateWindowsTree = vi.fn(async () => {}) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + exited: () => false, + captureWindowsDescendants: vi.fn(async () => null), + terminateWindowsTree, + terminateWindowsDescendants: vi.fn() + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + // No identity means no bare-pid tree kill, but the owned handle is still ours. + expect(terminateWindowsTree).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('never signals a root the reaper already saw exit', async () => { + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => true, + captureDescendants: vi.fn(async () => null) + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(child.kill).not.toHaveBeenCalled() + }) + + it('chains per-pid Windows boundaries across a second merge', async () => { + const first = windowsSnapshot(1_000) + const second: WindowsDescendantSnapshot = { + ...windowsSnapshot(2_000), + descendants: [ + { pid: 4243, creationTimeMs: 1_700_000_000_000 }, + { pid: 4244, creationTimeMs: 1_700_000_000_002 } + ] + } + const third: WindowsDescendantSnapshot = { ...second, capturedAtMs: 3_000 } + + const merged = mergeClaudeCapturedTrees( + { platform: 'win32', tree: first }, + { platform: 'win32', tree: second } + ) + expect(merged?.tree.capturedAtMsByPid).toEqual({ '4243': 1_000, '4244': 2_000 }) + const rechained = mergeClaudeCapturedTrees(merged!, { platform: 'win32', tree: third }) + + expect(rechained?.tree.capturedAtMsByPid).toEqual({ '4243': 1_000, '4244': 2_000 }) + }) +}) diff --git a/src/main/claude/claude-agent-sdk-user-message-queue.test.ts b/src/main/claude/claude-agent-sdk-user-message-queue.test.ts new file mode 100644 index 00000000000..62fbd8cf203 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-user-message-queue.test.ts @@ -0,0 +1,65 @@ +import type { SDKUserMessage } from '@anthropic-ai/claude-agent-sdk' +import { describe, expect, it } from 'vitest' +import { createClaudeUserMessageQueue } from './claude-agent-sdk-user-message-queue' + +/** + * The SDK's input pump is `for await (const frame of prompt) { await transport.write(frame) }`. + * A rejected write — or an abort — ends that loop abruptly, which calls the + * generator's `return()`. Everything below drives that exact shape, because the + * frame the pump already pulled is the one nothing else can reach. + */ +const frame = (text: string): SDKUserMessage => + ({ + type: 'user', + message: { role: 'user', content: [{ type: 'text', text }] } + }) as unknown as SDKUserMessage + +const settled = (promise: Promise): Promise<'settled' | 'pending'> => + Promise.race([ + promise.then( + () => 'settled' as const, + () => 'settled' as const + ), + new Promise<'pending'>((resolve) => setTimeout(() => resolve('pending'), 100)) + ]) + +describe('claude user message queue', () => { + it('rejects the frame the SDK pulled but abandoned without writing', async () => { + const queue = createClaudeUserMessageQueue() + const pump = queue.messages[Symbol.asyncIterator]() + const sent = queue.push(frame('hello')) + + await pump.next() + await pump.return?.(undefined) + + await expect(settled(sent)).resolves.toBe('settled') + await expect(sent).rejects.toThrow( + 'claude stream-json input ended before the frame was written' + ) + }) + + it('rejects an in-flight frame from fail() when the SDK never resumes the pump', async () => { + const queue = createClaudeUserMessageQueue() + const pump = queue.messages[Symbol.asyncIterator]() + const sent = queue.push(frame('hello')) + + await pump.next() + queue.fail(new Error('claude stream-json exited: child died')) + + await expect(settled(sent)).resolves.toBe('settled') + await expect(sent).rejects.toThrow('claude stream-json exited: child died') + }) + + it('still settles a written frame only once the pump asks for the next one', async () => { + const queue = createClaudeUserMessageQueue() + const pump = queue.messages[Symbol.asyncIterator]() + const sent = queue.push(frame('hello')) + + const pulled = await pump.next() + expect(pulled.value).toMatchObject({ type: 'user' }) + // The write proof is the pump coming back for more, exactly as before. + await expect(settled(sent)).resolves.toBe('pending') + void pump.next() + await expect(sent).resolves.toBeUndefined() + }) +}) diff --git a/src/main/claude/claude-agent-sdk-user-message-queue.ts b/src/main/claude/claude-agent-sdk-user-message-queue.ts new file mode 100644 index 00000000000..87fa6660159 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-user-message-queue.ts @@ -0,0 +1,100 @@ +import type { SDKUserMessage } from '@anthropic-ai/claude-agent-sdk' + +type QueuedMessage = { + message: SDKUserMessage + resolve: () => void + reject: (error: Error) => void +} + +export type ClaudeUserMessageQueue = { + /** The SDK's streaming-input prompt; it stays open until `end`. */ + messages: AsyncIterable + /** Resolves once the SDK has finished writing the frame to the child. */ + push: (message: SDKUserMessage) => Promise + /** Reject every unwritten frame, in-flight included; a caller waiting on a send must not hang past the exit. */ + fail: (error: Error) => void + end: () => void +} + +/** The rejection an abandoned frame carries when nothing else has named a cause yet. */ +const UNWRITTEN_FRAME_MESSAGE = 'claude stream-json input ended before the frame was written' + +export function createClaudeUserMessageQueue(): ClaudeUserMessageQueue { + const queued: QueuedMessage[] = [] + // The frame the SDK has taken but not yet acknowledged. It is out of `queued`, + // so it is unreachable from anywhere else and would otherwise never settle. + let inFlight: QueuedMessage | null = null + let wake: (() => void) | null = null + let ended = false + let failure: Error | null = null + const notify = (): void => { + wake?.() + wake = null + } + const rejectInFlight = (error: Error): void => { + const abandoned = inFlight + inFlight = null + abandoned?.reject(error) + } + + async function* drain(): AsyncGenerator { + for (;;) { + const next = queued.shift() + if (next) { + inFlight = next + let written = false + try { + yield next.message + written = true + } finally { + // The SDK's input pump abandons this iterator when its + // `await transport.write(...)` rejects or the query aborts, and the code + // after a `yield` never runs on that path. Settling here is the only + // place a frame it already took can be reached. + if (written) { + inFlight = null + // Resumed only after the SDK's `await transport.write(...)` settled, so this + // is the same "the frame reached the child" proof the hand-rolled write gave. + next.resolve() + } else { + rejectInFlight(failure ?? new Error(UNWRITTEN_FRAME_MESSAGE)) + } + } + continue + } + if (ended || failure) { + return + } + await new Promise((resolve) => { + wake = resolve + }) + } + } + + return { + messages: drain(), + push: (message) => + new Promise((resolve, reject) => { + if (failure) { + reject(failure) + return + } + queued.push({ message, resolve, reject }) + notify() + }), + fail: (error) => { + failure ??= error + for (const entry of queued.splice(0)) { + entry.reject(error) + } + // A pump that never resumes cannot run the generator's cleanup, so the + // exit path has to reach the in-flight frame itself. + rejectInFlight(error) + notify() + }, + end: () => { + ended = true + notify() + } + } +} diff --git a/src/main/claude/claude-background-task-tracker.test.ts b/src/main/claude/claude-background-task-tracker.test.ts new file mode 100644 index 00000000000..d8f316d7dcd --- /dev/null +++ b/src/main/claude/claude-background-task-tracker.test.ts @@ -0,0 +1,340 @@ +import { describe, expect, it } from 'vitest' +import { + ClaudeBackgroundTaskTracker, + classifyClaudeBackgroundTaskKind +} from './claude-background-task-tracker' + +function system(subtype: string, fields: Record): Record { + return { type: 'system', subtype, session_id: 'provider-1', uuid: crypto.randomUUID(), ...fields } +} + +function result(): Record { + return { type: 'result', subtype: 'success', session_id: 'provider-1', uuid: crypto.randomUUID() } +} + +function aggregate(tasks: unknown[]): Record { + return system('background_tasks_changed', { tasks }) +} + +describe('ClaudeBackgroundTaskTracker', () => { + it('classifies SDK task types without inferring them from descriptions', () => { + expect(classifyClaudeBackgroundTaskKind('local_agent')).toBe('agent') + expect(classifyClaudeBackgroundTaskKind('local_workflow')).toBe('workflow') + expect(classifyClaudeBackgroundTaskKind('local_bash')).toBe('command') + expect(classifyClaudeBackgroundTaskKind('monitor')).toBe('monitor') + expect(classifyClaudeBackgroundTaskKind('future_task')).toBe('unknown') + }) + + it('waits for the foreground turn to settle before monitoring a background task', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe({ type: 'user' }, true) + tracker.observe( + system('task_started', { + task_id: 'task-1', + task_type: 'local_agent', + is_backgrounded: true + }) + ) + expect(tracker.state).toBeNull() + + expect(tracker.observe(result())).toBe(true) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-1', kind: 'agent' }] + }) + }) + + it('uses an explicit background update for a foreground task and ignores progress alone', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe({ type: 'user' }, true) + tracker.observe( + system('task_started', { + task_id: 'task-1', + task_type: 'local_bash', + is_backgrounded: false + }) + ) + expect( + tracker.observe(system('task_progress', { task_id: 'task-1', description: 'still working' })) + ).toBe(false) + tracker.observe(result()) + expect(tracker.state).toBeNull() + + tracker.observe(system('task_updated', { task_id: 'task-1', patch: { is_backgrounded: true } })) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-1', kind: 'command' }] + }) + }) + + it('publishes bounded display details when a running task description changes', () => { + const tracker = new ClaudeBackgroundTaskTracker() + expect( + tracker.observe( + system('task_started', { + task_id: 'task-1', + task_type: 'local_bash', + is_backgrounded: true, + description: ' run\n the build ' + }) + ) + ).toBe(true) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-1', kind: 'command', description: 'run the build' }] + }) + + expect( + tracker.observe( + system('task_updated', { + task_id: 'task-1', + patch: { description: 'x'.repeat(600) } + }) + ) + ).toBe(true) + expect(tracker.state?.tasks?.[0]?.description).toHaveLength(512) + expect( + tracker.observe( + system('task_updated', { + task_id: 'task-1', + patch: { description: 'x'.repeat(600) } + }) + ) + ).toBe(false) + }) + + it('replaces its roster from aggregate lifecycle frames and preserves stoppable provider ids', () => { + const tracker = new ClaudeBackgroundTaskTracker() + expect( + tracker.observe( + aggregate([ + { task_id: 'task-agent', task_type: 'local_agent', description: 'agent' }, + { task_id: 'task-bash', task_type: 'local_bash', description: 'bash' } + ]) + ) + ).toBe(true) + expect(tracker.stoppableTaskIds).toEqual(['task-agent', 'task-bash']) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [ + { id: 'task-agent', kind: 'agent', description: 'agent' }, + { id: 'task-bash', kind: 'command', description: 'bash' } + ] + }) + + expect( + tracker.observe( + aggregate([{ task_id: 'task-next', task_type: 'local_workflow', description: 'workflow' }]) + ) + ).toBe(true) + expect(tracker.stoppableTaskIds).toEqual(['task-next']) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-next', kind: 'workflow', description: 'workflow' }] + }) + + expect(tracker.observe(aggregate([]))).toBe(true) + expect(tracker.stoppableTaskIds).toEqual([]) + expect(tracker.state).toBeNull() + }) + + it('excludes ambient aggregate tasks', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe( + aggregate([ + { task_id: 'ambient', task_type: 'monitor', description: 'watcher', ambient: true }, + { task_id: 'visible', task_type: 'local_bash', description: 'command' } + ]) + ) + + expect(tracker.stoppableTaskIds).toEqual(['visible']) + }) + + it('does not let late edge frames revive tasks cleared by an aggregate roster', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe( + aggregate([{ task_id: 'task-late', task_type: 'local_agent', description: 'agent' }]) + ) + tracker.observe(aggregate([])) + + tracker.observe( + system('task_started', { + task_id: 'task-late', + task_type: 'local_agent', + is_backgrounded: true + }) + ) + tracker.observe( + system('task_updated', { task_id: 'task-late', patch: { is_backgrounded: true } }) + ) + + expect(tracker.stoppableTaskIds).toEqual([]) + expect(tracker.state).toBeNull() + }) + + it('lets an authoritative aggregate roster replace earlier terminal-edge evidence', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe(system('task_notification', { task_id: 'task-live', status: 'completed' })) + + tracker.observe( + aggregate([{ task_id: 'task-live', task_type: 'local_agent', description: 'agent' }]) + ) + + expect(tracker.stoppableTaskIds).toEqual(['task-live']) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-live', kind: 'agent', description: 'agent' }] + }) + }) + + it('keeps terminal edges authoritative on either side of aggregate replacement', () => { + const terminalFirst = new ClaudeBackgroundTaskTracker() + terminalFirst.observe( + system('task_notification', { task_id: 'task-first', status: 'completed' }) + ) + terminalFirst.observe(aggregate([])) + terminalFirst.observe( + system('task_started', { + task_id: 'task-first', + task_type: 'local_agent', + is_backgrounded: true + }) + ) + expect(terminalFirst.state).toBeNull() + + const terminalLast = new ClaudeBackgroundTaskTracker() + terminalLast.observe( + aggregate([{ task_id: 'task-last', task_type: 'local_agent', description: 'agent' }]) + ) + terminalLast.observe(system('task_notification', { task_id: 'task-last', status: 'completed' })) + terminalLast.observe( + system('task_started', { + task_id: 'task-last', + task_type: 'local_agent', + is_backgrounded: true + }) + ) + expect(terminalLast.state).toBeNull() + }) + + it('keeps terminal evidence authoritative across duplicates and out-of-order starts', () => { + const tracker = new ClaudeBackgroundTaskTracker() + const terminal = system('task_notification', { task_id: 'task-late', status: 'completed' }) + tracker.observe(terminal) + tracker.observe(terminal) + tracker.observe( + system('task_started', { + task_id: 'task-late', + task_type: 'local_workflow', + is_backgrounded: true + }) + ) + expect(tracker.state).toBeNull() + + tracker.observe( + system('task_started', { + task_id: 'task-live', + task_type: 'monitor' + }) + ) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-live', kind: 'monitor' }] + }) + expect( + tracker.observe(system('task_updated', { task_id: 'task-live', patch: { status: 'killed' } })) + ).toBe(true) + expect(tracker.state).toBeNull() + }) + + it('recognizes task types that are registered only as background work', () => { + for (const taskType of ['local_workflow', 'monitor']) { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe(system('task_started', { task_id: taskType, task_type: taskType })) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: taskType, kind: taskType === 'local_workflow' ? 'workflow' : 'monitor' }] + }) + } + }) + + it('admits unknown background updates conservatively and bounds edge-only fallback ids', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe( + system('task_updated', { task_id: 'unknown', patch: { is_backgrounded: true } }) + ) + expect(tracker.stoppableTaskIds).toEqual(['unknown']) + + for (let index = 0; index < 400; index += 1) { + tracker.observe( + system('task_started', { + task_id: `task-${index}`, + task_type: 'local_agent', + is_backgrounded: true + }) + ) + } + expect(tracker.stoppableTaskIds.length).toBeLessThanOrEqual(256) + }) + + it('bounds aggregate rosters and resets to the edge-only fallback on clear', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe( + aggregate( + Array.from({ length: 400 }, (_, index) => ({ + task_id: `aggregate-${index}`, + task_type: 'local_bash', + description: 'command' + })) + ) + ) + expect(tracker.stoppableTaskIds).toHaveLength(256) + + tracker.clear() + tracker.observe( + system('task_started', { + task_id: 'edge-after-reset', + task_type: 'local_agent', + is_backgrounded: true + }) + ) + expect(tracker.stoppableTaskIds).toEqual(['edge-after-reset']) + }) + + it('gates aggregate monitoring behind foreground turn completion', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe({ type: 'user' }, true) + tracker.observe( + aggregate([{ task_id: 'task-live', task_type: 'local_bash', description: 'command' }]) + ) + expect(tracker.state).toBeNull() + + expect(tracker.observe(result())).toBe(true) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-live', kind: 'command', description: 'command' }] + }) + }) + + it('ignores ambient SDK tasks and clears all liveness when the session ends', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe( + system('task_started', { + task_id: 'ambient', + task_type: 'monitor', + is_backgrounded: true, + ambient: true + }) + ) + expect(tracker.state).toBeNull() + tracker.observe( + system('task_started', { + task_id: 'task-live', + task_type: 'local_agent', + is_backgrounded: true + }) + ) + expect(tracker.clear()).toBe(true) + expect(tracker.state).toBeNull() + }) +}) diff --git a/src/main/claude/claude-background-task-tracker.ts b/src/main/claude/claude-background-task-tracker.ts new file mode 100644 index 00000000000..a1504a65dec --- /dev/null +++ b/src/main/claude/claude-background-task-tracker.ts @@ -0,0 +1,251 @@ +import type { + AgentSessionBackgroundTask, + AgentSessionBackgroundTaskState +} from '../../shared/agent-session-wire' + +const MAX_TRACKED_TASKS = 256 +const MAX_TASK_ID_LENGTH = 512 +const MAX_TASK_DESCRIPTION_LENGTH = 512 +const TERMINAL_TASK_STATES = new Set(['completed', 'failed', 'killed', 'stopped']) + +export type ClaudeBackgroundTaskKind = AgentSessionBackgroundTask['kind'] + +type TrackedTask = { + backgrounded: boolean + kind: ClaudeBackgroundTaskKind + description?: string +} + +function record(value: unknown): Record | null { + return typeof value === 'object' && value !== null ? (value as Record) : null +} + +function taskId(message: Record): string | null { + const value = message.task_id + return typeof value === 'string' && value.length > 0 && value.length <= MAX_TASK_ID_LENGTH + ? value + : null +} + +function taskDescription(value: unknown): string | undefined { + if (typeof value !== 'string') { + return undefined + } + const trimmed = value.trim().replace(/\s+/g, ' ') + return trimmed.length > 0 ? trimmed.slice(0, MAX_TASK_DESCRIPTION_LENGTH) : undefined +} + +export function classifyClaudeBackgroundTaskKind(taskType: unknown): ClaudeBackgroundTaskKind { + switch (taskType) { + case 'local_agent': + return 'agent' + case 'local_workflow': + return 'workflow' + case 'local_bash': + return 'command' + case 'monitor': + return 'monitor' + default: + return 'unknown' + } +} + +export class ClaudeBackgroundTaskTracker { + private readonly tasks = new Map() + private readonly terminalTaskIds = new Set() + private aggregateRosterObserved = false + private foregroundTurnActive = false + private monitoring = false + private publishedTasksFingerprint = '' + + get state(): AgentSessionBackgroundTaskState | null { + if (!this.monitoring) { + return null + } + return { + state: 'monitoring', + tasks: this.backgroundTaskDetails() + } + } + + get stoppableTaskIds(): string[] { + const ids: string[] = [] + for (const [id, task] of this.tasks) { + if (task.backgrounded) { + ids.push(id) + } + } + return ids + } + + observe(message: Record, startsTurn = false): boolean { + if (startsTurn) { + this.foregroundTurnActive = true + } + if (message.type === 'result') { + this.foregroundTurnActive = false + } else if (message.type === 'system') { + if (!this.observeSystemFrame(message) && !startsTurn) { + return false + } + } else if (!startsTurn) { + return false + } + return this.refreshMonitoring() + } + + clear(): boolean { + this.tasks.clear() + this.terminalTaskIds.clear() + this.aggregateRosterObserved = false + this.foregroundTurnActive = false + return this.refreshMonitoring() + } + + private observeSystemFrame(message: Record): boolean { + if (message.subtype === 'background_tasks_changed') { + this.replaceAggregateRoster(message.tasks) + return true + } + const id = taskId(message) + if (!id) { + return false + } + if (message.subtype === 'task_notification') { + this.finish(id) + return true + } + if (message.subtype === 'task_updated') { + const patch = record(message.patch) + if (!patch) { + return false + } + if (TERMINAL_TASK_STATES.has(String(patch.status))) { + this.finish(id) + return true + } + const existing = this.tasks.get(id) + if ( + (patch.is_backgrounded === true || taskDescription(patch.description)) && + (!this.aggregateRosterObserved || existing) + ) { + this.upsert(id, { + backgrounded: patch.is_backgrounded === true || existing?.backgrounded === true, + kind: existing?.kind ?? 'unknown', + description: taskDescription(patch.description) ?? existing?.description + }) + return true + } + return false + } + if (message.subtype !== 'task_started' || this.terminalTaskIds.has(id)) { + return false + } + if (message.ambient === true || message.skip_transcript === true) { + this.finish(id) + return true + } + if (this.aggregateRosterObserved && !this.tasks.has(id)) { + return false + } + const kind = classifyClaudeBackgroundTaskKind(message.task_type) + this.upsert(id, { + backgrounded: message.is_backgrounded === true || kind === 'workflow' || kind === 'monitor', + kind, + description: taskDescription(message.description) + }) + return true + } + + private replaceAggregateRoster(value: unknown): void { + if (!Array.isArray(value)) { + return + } + this.aggregateRosterObserved = true + this.tasks.clear() + this.terminalTaskIds.clear() + for (const valueTask of value) { + if (this.tasks.size >= MAX_TRACKED_TASKS) { + break + } + const task = record(valueTask) + if (!task || task.ambient === true) { + continue + } + const id = taskId(task) + if (!id) { + continue + } + this.tasks.set(id, { + backgrounded: true, + kind: classifyClaudeBackgroundTaskKind(task.task_type), + description: taskDescription(task.description) + }) + } + } + + private upsert(id: string, task: TrackedTask): void { + const existing = this.tasks.get(id) + if (existing) { + this.tasks.set(id, { + backgrounded: existing.backgrounded || task.backgrounded, + kind: existing.kind === 'unknown' ? task.kind : existing.kind, + description: task.description ?? existing.description + }) + return + } + if (this.tasks.size >= MAX_TRACKED_TASKS) { + let foregroundId: string | undefined + for (const [candidateId, candidate] of this.tasks) { + if (!candidate.backgrounded) { + foregroundId = candidateId + break + } + } + if (!foregroundId) { + return + } + this.tasks.delete(foregroundId) + } + this.tasks.set(id, task) + } + + private finish(id: string): void { + this.tasks.delete(id) + this.terminalTaskIds.delete(id) + this.terminalTaskIds.add(id) + if (this.terminalTaskIds.size > MAX_TRACKED_TASKS) { + const oldest = this.terminalTaskIds.values().next() + if (!oldest.done) { + this.terminalTaskIds.delete(oldest.value) + } + } + } + + private refreshMonitoring(): boolean { + const details = this.foregroundTurnActive ? [] : this.backgroundTaskDetails() + const next = details.length > 0 + const fingerprint = next ? JSON.stringify(details) : '' + if (next === this.monitoring && fingerprint === this.publishedTasksFingerprint) { + return false + } + this.monitoring = next + this.publishedTasksFingerprint = fingerprint + return true + } + + private backgroundTaskDetails(): AgentSessionBackgroundTask[] { + const details: AgentSessionBackgroundTask[] = [] + for (const [id, task] of this.tasks) { + if (!task.backgrounded) { + continue + } + details.push({ + id, + kind: task.kind, + ...(task.description ? { description: task.description } : {}) + }) + } + return details + } +} diff --git a/src/main/claude/claude-child-exit-proof-ladder.ts b/src/main/claude/claude-child-exit-proof-ladder.ts new file mode 100644 index 00000000000..85ed629f1b9 --- /dev/null +++ b/src/main/claude/claude-child-exit-proof-ladder.ts @@ -0,0 +1,41 @@ +import type { SpawnedProcess } from '../../shared/child-process/run-process' +import { waitForProcessExitUntil } from '../codex/codex-process-exit-deadline' +import type { ClaudeChildTreeReaper } from './claude-agent-sdk-exit-proof' + +const GRACEFUL_EXIT_MS = 1_500 +const FORCED_EXIT_MS = 1_000 + +export type ClaudeChildExitProofInput = { + child: Pick + exitPromise: Promise + exited: () => boolean + tree?: ClaudeChildTreeReaper +} + +export async function proveClaudeChildExitWithReaper( + input: ClaudeChildExitProofInput, + createTree: () => ClaudeChildTreeReaper +): Promise { + const tree = input.tree ?? createTree() + // Arm before stdin closes: only a live root can identify its descendants. + await tree.capture() + try { + input.child.stdin?.end() + } catch { + // The reap below still owns the process. + } + let reaped = false + if (!input.exited()) { + await waitForProcessExitUntil(input.exitPromise, GRACEFUL_EXIT_MS) + if (!input.exited()) { + reaped = true + await tree.refresh?.() + await tree.reap() + await waitForProcessExitUntil(input.exitPromise, FORCED_EXIT_MS) + } + } + if (!reaped && input.exited() && tree.treeVerdict !== 'exited') { + await tree.reap() + } + return input.exited() && tree.treeVerdict === 'exited' +} diff --git a/src/main/claude/claude-child-process-environment.test.ts b/src/main/claude/claude-child-process-environment.test.ts new file mode 100644 index 00000000000..8299fdcdd61 --- /dev/null +++ b/src/main/claude/claude-child-process-environment.test.ts @@ -0,0 +1,63 @@ +import { describe, expect, it } from 'vitest' +import { applyClaudeEnvPatch } from '../claude-accounts/environment' +import { buildClaudeChildProcessEnv } from './claude-child-process-environment' + +describe('Claude child process environment', () => { + it('strips case-insensitive auth headers through the shared env patch on Windows', () => { + expect( + applyClaudeEnvPatch( + { + anthropic_api_key: 'inherited-key', + Anthropic_Custom_Headers: 'Authorization: inherited', + SAFE_VALUE: 'preserved' + }, + {}, + { stripAuthEnv: true, platform: 'win32' } + ) + ).toEqual({ SAFE_VALUE: 'preserved' }) + }) + + it('strips case-insensitive inherited auth and session stamps on Windows', () => { + const env = buildClaudeChildProcessEnv( + { + ANTHROPIC_AUTH_TOKEN: 'configured-token', + Claude_Code_Session_Id: 'configured-session' + }, + { + platform: 'win32', + inheritedEnv: { + anthropic_api_key: 'inherited-key', + Anthropic_Custom_Headers: 'Authorization: inherited', + claude_code_child_session: '1', + CLAUDE_CODE_SESSION_ID: 'inherited-session', + SAFE_VALUE: 'preserved' + } + } + ) + + expect(env).toEqual({ + ANTHROPIC_AUTH_TOKEN: 'configured-token', + Claude_Code_Session_Id: 'configured-session', + SAFE_VALUE: 'preserved' + }) + }) + + it('can strip child-session stamps reintroduced by a full SDK launch overlay', () => { + expect( + buildClaudeChildProcessEnv( + { + CLAUDE_CODE_CHILD_SESSION: 'configured-child-session', + CLAUDE_CODE_SESSION_ID: 'configured-session', + CLAUDE_CODE_BRIDGE_SESSION_ID: 'configured-bridge-session' + }, + { + scrubConfiguredChildSessionStamps: true, + inheritedEnv: { + CLAUDE_CODE_CHILD_SESSION: 'inherited-child-session', + SAFE_VALUE: 'preserved' + } + } + ) + ).toEqual({ SAFE_VALUE: 'preserved' }) + }) +}) diff --git a/src/main/claude/claude-child-process-environment.ts b/src/main/claude/claude-child-process-environment.ts new file mode 100644 index 00000000000..58f00c5c1b8 --- /dev/null +++ b/src/main/claude/claude-child-process-environment.ts @@ -0,0 +1,69 @@ +import { CLAUDE_AUTH_ENV_VARS, applyClaudeEnvPatch } from '../claude-accounts/environment' + +const CLAUDE_CHILD_SESSION_STAMP_ENV_KEYS = [ + 'CLAUDE_CODE_CHILD_SESSION', + 'CLAUDE_CODE_SESSION_ID', + 'CLAUDE_CODE_BRIDGE_SESSION_ID' +] as const + +function cloneProcessEnv(source: NodeJS.ProcessEnv): Record { + const env: Record = {} + for (const [key, value] of Object.entries(source)) { + if (value !== undefined) { + env[key] = value + } + } + return env +} + +function stripClaudeChildSessionStamps( + env: Record, + platform: NodeJS.Platform +): Record { + for (const key of CLAUDE_CHILD_SESSION_STAMP_ENV_KEYS) { + for (const envKey of Object.keys(env)) { + if (envKey === key || (platform === 'win32' && envKey.toUpperCase() === key)) { + delete env[envKey] + } + } + } + return env +} + +export function buildClaudeChildProcessEnv( + configuredEnv: Record = {}, + options: { + inheritedEnv?: NodeJS.ProcessEnv + platform?: NodeJS.Platform + scrubConfiguredChildSessionStamps?: boolean + } = {} +): Record { + const inheritedEnv = options.inheritedEnv ?? process.env + const platform = options.platform ?? process.platform + const env = applyClaudeEnvPatch( + cloneProcessEnv(inheritedEnv), + {}, + { + stripAuthEnv: true, + platform + } + ) + if (platform === 'win32') { + const authKeys = new Set(CLAUDE_AUTH_ENV_VARS.map((key) => key.toUpperCase())) + for (const [key, value] of Object.entries(env)) { + const normalized = key.toUpperCase() + if ( + authKeys.has(normalized) || + (normalized === 'ANTHROPIC_CUSTOM_HEADERS' && + /authorization|x-api-key|api-key|bearer/i.test(value)) + ) { + delete env[key] + } + } + } + if (options.scrubConfiguredChildSessionStamps) { + return stripClaudeChildSessionStamps({ ...env, ...configuredEnv }, platform) + } + stripClaudeChildSessionStamps(env, platform) + return { ...env, ...configuredEnv } +} diff --git a/src/main/claude/claude-child-root-termination.ts b/src/main/claude/claude-child-root-termination.ts new file mode 100644 index 00000000000..bed422532e1 --- /dev/null +++ b/src/main/claude/claude-child-root-termination.ts @@ -0,0 +1,54 @@ +import type { SpawnedProcess } from '../../shared/child-process/run-process' +import type { PosixProcessIdentity } from '../pty-descendant-termination' +import type { + WindowsDescendantSnapshot, + WindowsProcessIdentity +} from '../windows-descendant-exit-verification' + +export type ClaudeRootIdentity = PosixProcessIdentity | WindowsProcessIdentity + +type RootTerminationInput = { + child: Pick + exited: () => boolean +} + +/** + * Kills the root through the handle Node owns rather than through its pid, which + * is why no identity probe gates it: libuv drops that handle in the same turn it + * reaps, so the signal either reaches the process Orca spawned or reaches + * nothing. A probe here could only let an unreadable process table cost the tree + * the one fallback that still works once every table read has failed. + * + * False means no signal was sent, because the root had already left. + */ +export function terminateClaudeRoot(input: RootTerminationInput): boolean { + return input.exited() ? false : input.child.kill('SIGKILL') +} + +type WindowsRootTerminationInput = { + snapshot: WindowsDescendantSnapshot | null + exited: () => boolean + verifyRoot: (root: WindowsProcessIdentity) => Promise + terminateTree: (root: WindowsProcessIdentity) => Promise + killRoot: () => boolean +} + +/** + * `taskkill /T /F` addresses a bare pid, so a dead root's pid may already belong + * to a stranger whose whole tree it would take down: that one is identity-gated. + * The direct root kill after it runs however the probe decided. + */ +export async function terminateClaudeWindowsRoot( + input: WindowsRootTerminationInput +): Promise<{ rootVerified: boolean }> { + const { snapshot, exited, verifyRoot, terminateTree, killRoot } = input + let rootVerified = false + if (!exited() && snapshot) { + rootVerified = await verifyRoot(snapshot.root).catch(() => false) + if (rootVerified && !exited()) { + await terminateTree(snapshot.root).catch(() => {}) + } + } + killRoot() + return { rootVerified } +} diff --git a/src/main/claude/claude-child-tree-snapshot.ts b/src/main/claude/claude-child-tree-snapshot.ts new file mode 100644 index 00000000000..e0955648b02 --- /dev/null +++ b/src/main/claude/claude-child-tree-snapshot.ts @@ -0,0 +1,128 @@ +import type { DescendantSnapshot } from '../pty-descendant-termination' +import type { WindowsDescendantSnapshot } from '../windows-descendant-exit-verification' + +/** One platform's descendant tree, tagged so neither verifier can be handed the other's rows. */ +export type ClaudeCapturedTree = + | { platform: 'posix'; tree: DescendantSnapshot } + | { platform: 'win32'; tree: WindowsDescendantSnapshot } + +/** + * Process-table reads are not atomic: a refresh can omit a still-live row, but + * it can also observe a new process after the old row exited. Retain rows absent + * from the refresh, but reject a PID whose identity changed between reads. + */ +function mergeRowsByPid( + previous: readonly Row[], + next: readonly Row[], + sameIdentity: (previous: Row, next: Row) => boolean, + previousBoundary: (row: Row) => number, + nextBoundary: (row: Row) => number, + refreshBoundary: number +): { rows: Row[]; capturedAtMsByPid?: Readonly> } | null { + const merged = new Map() + const capturedAtMsByPid: Record = {} + for (const row of previous) { + const prior = merged.get(row.pid) + if (prior && !sameIdentity(prior, row)) { + return null + } + merged.set(row.pid, row) + capturedAtMsByPid[String(row.pid)] = previousBoundary(row) + } + for (const row of next) { + const prior = merged.get(row.pid) + if (prior && !sameIdentity(prior, row)) { + return null + } + if (!prior) { + capturedAtMsByPid[String(row.pid)] = nextBoundary(row) + } + merged.set(row.pid, row) + } + const boundaries = Object.values(capturedAtMsByPid) + const needsBoundaryMap = + new Set(boundaries).size > 1 || boundaries.some((boundary) => boundary !== refreshBoundary) + return { + rows: [...merged.values()], + ...(needsBoundaryMap ? { capturedAtMsByPid } : {}) + } +} + +export function mergeClaudeCapturedTrees( + previous: ClaudeCapturedTree, + next: ClaudeCapturedTree +): ClaudeCapturedTree | null { + if (previous.platform !== next.platform) { + return null + } + if (previous.platform === 'posix' && next.platform === 'posix') { + if (previous.tree.rootPgid !== next.tree.rootPgid) { + return null + } + // A refresh cannot repair an earlier capture that lacked root identity; + // retaining those rows would permit a later numeric-pid kill without proof. + if (!previous.tree.root || !next.tree.root) { + return null + } + if ( + previous.tree.root.pid !== next.tree.root.pid || + previous.tree.root.startedAt !== next.tree.root.startedAt + ) { + return null + } + const descendants = mergeRowsByPid( + previous.tree.descendants, + next.tree.descendants, + (left, right) => left.pgid === right.pgid && left.startedAt === right.startedAt, + (row) => previous.tree.capturedAtMsByPid?.[String(row.pid)] ?? previous.tree.capturedAtMs, + (row) => next.tree.capturedAtMsByPid?.[String(row.pid)] ?? next.tree.capturedAtMs, + next.tree.capturedAtMs + ) + if (!descendants) { + return null + } + return { + platform: 'posix', + tree: { + ...next.tree, + // Retained rows keep their earlier boundary; new rows use the refresh + // boundary. The scalar remains the latest scan for legacy consumers. + descendants: descendants.rows, + ...(descendants.capturedAtMsByPid + ? { capturedAtMsByPid: descendants.capturedAtMsByPid } + : {}) + } + } + } + if (previous.platform === 'win32' && next.platform === 'win32') { + if ( + previous.tree.root.pid !== next.tree.root.pid || + previous.tree.root.creationTimeMs !== next.tree.root.creationTimeMs + ) { + return null + } + const descendants = mergeRowsByPid( + previous.tree.descendants, + next.tree.descendants, + (left, right) => left.creationTimeMs === right.creationTimeMs, + (row) => previous.tree.capturedAtMsByPid?.[String(row.pid)] ?? previous.tree.capturedAtMs, + (row) => next.tree.capturedAtMsByPid?.[String(row.pid)] ?? next.tree.capturedAtMs, + next.tree.capturedAtMs + ) + if (!descendants) { + return null + } + return { + platform: 'win32', + tree: { + ...next.tree, + descendants: descendants.rows, + ...(descendants.capturedAtMsByPid + ? { capturedAtMsByPid: descendants.capturedAtMsByPid } + : {}), + unidentifiedCount: Math.max(previous.tree.unidentifiedCount, next.tree.unidentifiedCount) + } + } + } + return null +} diff --git a/src/main/claude/claude-command-lifecycle-frames.test.ts b/src/main/claude/claude-command-lifecycle-frames.test.ts new file mode 100644 index 00000000000..8126a546ed8 --- /dev/null +++ b/src/main/claude/claude-command-lifecycle-frames.test.ts @@ -0,0 +1,115 @@ +import { describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity +} from '../../shared/agent-session-journal-types' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' + +function sinkState() { + const items: { identity: AgentJournalItemIdentity; body: AgentJournalItemBody }[] = [] + const sink: StructuredAgentSessionEventSink = { + appendItem: (identity, body) => items.push({ identity, body }), + appendTombstone: () => {}, + publish: vi.fn() + } + return { sink, items } +} + +function providerFrameKinds(items: { body: AgentJournalItemBody }[]): string[] { + return items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame ? [item.body.providerFrame.kind] : [] + ) +} + +/** + * The queue-bookkeeping frame Claude Code 2.1.258 emits for every uuid-stamped + * command: `command_uuid` plus a state, and no content of its own. Shape and + * states taken from the CLI's own emission sites. + */ +function commandLifecycle(state: 'started' | 'completed' | 'cancelled', uuid: string) { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'command_lifecycle', + command_uuid: 'command-1', + state, + uuid, + session_id: 'claude-session' + } + } +} + +function userTurn(uuid: string, text: string) { + return { + type: 'message' as const, + sessionId: 'orca-session', + startsTurn: true as const, + message: { + type: 'user', + uuid, + session_id: 'claude-session', + parent_tool_use_id: null, + isReplay: true, + message: { role: 'user', content: text } + } + } +} + +function assistantReply(uuid: string, text: string) { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'assistant', + uuid, + session_id: 'claude-session', + parent_tool_use_id: null, + message: { role: 'assistant', content: [{ type: 'text', text }] } + } + } +} + +describe('Claude command_lifecycle frames', () => { + it('keeps queue bookkeeping off the transcript for a whole turn', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userTurn('user-1', 'Reply with exactly PROBE_OK and nothing else.')) + translator.handle(commandLifecycle('started', 'lifecycle-1')) + translator.handle(assistantReply('assistant-1', 'PROBE_OK')) + translator.handle(commandLifecycle('completed', 'lifecycle-2')) + translator.handle(commandLifecycle('completed', 'lifecycle-3')) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { + type: 'result', + subtype: 'success', + uuid: 'result-1', + session_id: 'claude-session', + is_error: false, + result: 'PROBE_OK', + terminal_reason: 'completed' + } + }) + + expect(providerFrameKinds(state.items)).toEqual([]) + // The turn's real content is untouched. + expect( + state.items.flatMap((item) => + item.body.kind === 'message' && item.body.role === 'assistant' ? [item.body.blocks] : [] + ) + ).toEqual([[{ type: 'text', text: 'PROBE_OK' }]]) + }) + + it('keeps a cancelled command off the transcript too', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(commandLifecycle('cancelled', 'lifecycle-4')) + + expect(providerFrameKinds(state.items)).toEqual([]) + }) +}) diff --git a/src/main/claude/claude-config-dir-pin.test.ts b/src/main/claude/claude-config-dir-pin.test.ts new file mode 100644 index 00000000000..c7be40a6f38 --- /dev/null +++ b/src/main/claude/claude-config-dir-pin.test.ts @@ -0,0 +1,34 @@ +import { homedir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { claudeConfigDirEnvPatch, defaultClaudeConfigDir } from './claude-config-dir-pin' + +describe('claude config dir pin', () => { + it('does not pin the CLI default home, so the macOS Keychain stays reachable', () => { + expect(claudeConfigDirEnvPatch(join(homedir(), '.claude'), { env: {} })).toEqual({}) + expect(claudeConfigDirEnvPatch(`${join(homedir(), '.claude')}/`, { env: {} })).toEqual({}) + expect(claudeConfigDirEnvPatch(' ', { env: {} })).toEqual({}) + }) + + it('pins a managed account home the CLI would not find on its own', () => { + expect(claudeConfigDirEnvPatch('/accounts/claude/managed', { env: {} })).toEqual({ + CLAUDE_CONFIG_DIR: '/accounts/claude/managed' + }) + }) + + it('treats an inherited CLAUDE_CONFIG_DIR as the default the CLI already resolves', () => { + const env = { CLAUDE_CONFIG_DIR: '/inherited/home' } + expect(defaultClaudeConfigDir(env)).toBe('/inherited/home') + expect(claudeConfigDirEnvPatch('/inherited/home', { env })).toEqual({}) + expect(claudeConfigDirEnvPatch('/other/home', { env })).toEqual({ + CLAUDE_CONFIG_DIR: '/other/home' + }) + }) + + it('compares Windows homes case-insensitively', () => { + const env = { CLAUDE_CONFIG_DIR: 'C:\\Users\\Work\\.claude' } + expect(claudeConfigDirEnvPatch('c:\\users\\work\\.claude', { env, platform: 'win32' })).toEqual( + {} + ) + }) +}) diff --git a/src/main/claude/claude-config-dir-pin.ts b/src/main/claude/claude-config-dir-pin.ts new file mode 100644 index 00000000000..d5cc0b8b186 --- /dev/null +++ b/src/main/claude/claude-config-dir-pin.ts @@ -0,0 +1,37 @@ +import { homedir } from 'node:os' +import { join, resolve } from 'node:path' + +/** The config dir the Claude CLI resolves for itself when nothing pins one. */ +export function defaultClaudeConfigDir(env: NodeJS.ProcessEnv = process.env): string { + return env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') +} + +function samePath(a: string, b: string, platform: NodeJS.Platform): boolean { + const left = resolve(a) + const right = resolve(b) + return platform === 'win32' ? left.toLowerCase() === right.toLowerCase() : left === right +} + +/** + * An explicit CLAUDE_CONFIG_DIR moves the Claude CLI off the default Keychain item onto + * one derived from the pinned path, so a claude.ai OAuth login stops working even when + * the pin names the CLI's own default. Pin only a home the CLI would not find on its + * own — the same rule the legacy PTY path applies via `ClaudeRuntimePathResolver`. + * + * The pinned value is the account home verbatim: the CLI keys its credential lookup on + * the literal string, so re-spelling an equivalent path (absolute vs `~`, trailing + * separator) selects a different identity. Normalization here is for the equality test + * only and must never reach the env. + */ +export function claudeConfigDirEnvPatch( + accountHome: string, + options: { env?: NodeJS.ProcessEnv; platform?: NodeJS.Platform } = {} +): { CLAUDE_CONFIG_DIR?: string } { + const env = options.env ?? process.env + const platform = options.platform ?? process.platform + const resolved = accountHome.trim() + if (!resolved || samePath(resolved, defaultClaudeConfigDir(env), platform)) { + return {} + } + return { CLAUDE_CONFIG_DIR: resolved } +} diff --git a/src/main/claude/claude-descendant-escalation-boundary.test.ts b/src/main/claude/claude-descendant-escalation-boundary.test.ts new file mode 100644 index 00000000000..55459df0c7f --- /dev/null +++ b/src/main/claude/claude-descendant-escalation-boundary.test.ts @@ -0,0 +1,124 @@ +import { describe, expect, it, vi } from 'vitest' +import { terminateDescendantSnapshotWithVerdict } from '../pty-descendant-exit-verification' +import { + collectDescendantRows, + type DescendantSnapshot, + type ProcessTableRow +} from '../pty-descendant-termination' +import { createClaudeChildTreeReaper } from './claude-agent-sdk-exit-proof' + +const ROOT_PID = 500 +const ORCA_PGID = 400 +const ROOT_STARTED_AT = 'Thu Sep 3 18:04:50 2026' +/** The second both close-time walks land in. */ +const WALK_SECOND = 'Thu Sep 3 18:05:04 2026' +const WALK_MS = Date.parse(WALK_SECOND) +const EARLIER_SECOND = 'Thu Sep 3 18:05:03 2026' + +/** The measured split: `s20` at :03.946 died, `s21` at :04.042 leaked. */ +const EARLIER_BORN = [700, 701, 702] +const WALK_SECOND_BORN = [721, 722, 723, 724] + +type Cohort = { pids: number[]; startedAt: string } + +const LIVE_TREE: Cohort[] = [ + { pids: EARLIER_BORN, startedAt: EARLIER_SECOND }, + { pids: WALK_SECOND_BORN, startedAt: WALK_SECOND } +] + +function rowsFor(cohorts: Cohort[]): ProcessTableRow[] { + return [ + { pid: ROOT_PID, ppid: 1, pgid: ORCA_PGID, startedAt: ROOT_STARTED_AT }, + ...cohorts.flatMap((cohort) => + cohort.pids.map((pid) => ({ + pid, + ppid: ROOT_PID, + pgid: ORCA_PGID, + startedAt: cohort.startedAt + })) + ) + ] +} + +/** A real ppid walk from the root, exactly as production captures one. */ +function walk(capturedAtMs: number, cohorts: Cohort[] = LIVE_TREE): DescendantSnapshot { + return collectDescendantRows(ROOT_PID, rowsFor(cohorts), capturedAtMs) +} + +function killedPids(calls: [number, NodeJS.Signals][]): number[] { + return calls.flatMap(([pid, signal]) => (signal === 'SIGKILL' ? [pid] : [])).sort((a, b) => a - b) +} + +function signalledPids(calls: [number, NodeJS.Signals][]): number[] { + return calls.flatMap(([pid, signal]) => (signal === 'SIGTERM' ? [pid] : [])).sort((a, b) => a - b) +} + +/** + * Drives the real reaper and the real verifier against a process table where + * every descendant traps SIGTERM, so only a forced sweep can end them. The root + * is alive for both walks and gone by the sweep, which is the measured teardown. + */ +async function sweep( + captures: DescendantSnapshot[], + liveTree: Cohort[] = LIVE_TREE +): Promise<[number, NodeJS.Signals][]> { + const calls: [number, NodeJS.Signals][] = [] + const captureDescendants = vi.fn() + for (const capture of captures) { + captureDescendants.mockResolvedValueOnce(capture) + } + const tree = createClaudeChildTreeReaper( + { pid: ROOT_PID, kill: vi.fn(() => true) }, + { + platform: 'linux', + exited: () => false, + captureDescendants, + terminateDescendants: (snapshot) => + terminateDescendantSnapshotWithVerdict(snapshot, { + requireIdentityBeforeSignal: true, + graceMs: 0, + verifyMs: 120, + sendSignal: (pid, signal) => calls.push([pid, signal]), + readTable: async () => ({ rows: rowsFor(liveTree), capturedAtMs: Date.now() }) + }) + } + ) + // The close ladder's shape: arm, then re-walk the live root at the boundary. + await tree.capture() + await tree.refresh?.() + await tree.reap() + return calls +} + +describe('Claude descendant forced-sweep fence', () => { + it('escalates a descendant forked in the same second as both close walks', async () => { + // Both walks land inside second :04, one ps duration apart, and the root is + // gone before a third could run. A descendant born at :04.042 is no less + // ours than its sibling born 96ms earlier at :03.946. + const calls = await sweep([walk(WALK_MS + 42), walk(WALK_MS + 140)]) + + expect(signalledPids(calls)).toEqual([...EARLIER_BORN, ...WALK_SECOND_BORN]) + expect(killedPids(calls)).toEqual([...EARLIER_BORN, ...WALK_SECOND_BORN]) + }) + + it('still escalates descendants born before the walk that first saw them', async () => { + const onlyEarlier = [{ pids: EARLIER_BORN, startedAt: EARLIER_SECOND }] + const calls = await sweep([walk(WALK_MS + 42, onlyEarlier)], onlyEarlier) + + expect(killedPids(calls)).toEqual(EARLIER_BORN) + }) + + it('withholds the sweep from a row no walk re-derived, on its start second alone', async () => { + // 900 was seen once, in its own birth second, and the refresh did not find + // it. The merge retains the row, but nothing re-proved it belongs to us, so + // the second-resolution fence is all there is and it still says no. + const retained = { pids: [900], startedAt: WALK_SECOND } + const firstWalk = walk(WALK_MS + 42, [...LIVE_TREE, retained]) + const refresh = walk(WALK_MS + 140) + + const calls = await sweep([firstWalk, refresh], [...LIVE_TREE, retained]) + + expect(signalledPids(calls)).toEqual([...EARLIER_BORN, ...WALK_SECOND_BORN, 900]) + expect(killedPids(calls)).toEqual([...EARLIER_BORN, ...WALK_SECOND_BORN]) + }) +}) diff --git a/src/main/claude/claude-stream-json-connection-close.test.ts b/src/main/claude/claude-stream-json-connection-close.test.ts new file mode 100644 index 00000000000..e824139a776 --- /dev/null +++ b/src/main/claude/claude-stream-json-connection-close.test.ts @@ -0,0 +1,126 @@ +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import type { ChildProcessWithoutNullStreams } from 'node:child_process' +import { describe, expect, it, vi } from 'vitest' +import type { query } from '@anthropic-ai/claude-agent-sdk' +import { + openClaudeStreamJsonConnection, + type ClaudeStreamJsonLaunch +} from './claude-stream-json-connection' + +const mocks = vi.hoisted(() => { + const refresh = vi.fn() + const proveClaudeChildExit = vi.fn() + const tree = { + capture: vi.fn(async () => {}), + refresh: (...args: unknown[]) => refresh(...args), + reap: vi.fn(async () => 'exited' as const), + treeVerdict: 'unverifiable' as const + } + return { proveClaudeChildExit, refresh, tree } +}) + +vi.mock('./claude-agent-sdk-exit-proof', () => ({ + createClaudeChildTreeReaper: vi.fn(() => mocks.tree), + proveClaudeChildExit: (...args: unknown[]) => mocks.proveClaudeChildExit(...args) +})) + +function fakeChild(): ChildProcessWithoutNullStreams { + const child = new EventEmitter() + return Object.assign(child, { + pid: 424242, + stdin: new PassThrough(), + stdout: new PassThrough(), + stderr: new PassThrough(), + kill: vi.fn() + }) as unknown as ChildProcessWithoutNullStreams +} + +describe('Claude stream-json close ordering', () => { + it('waits for the live tree refresh before ending stdin', async () => { + const refreshDone = Promise.withResolvers() + mocks.refresh.mockReturnValueOnce(refreshDone.promise) + mocks.proveClaudeChildExit.mockResolvedValueOnce(true) + const child = fakeChild() + const launch: ClaudeStreamJsonLaunch = { + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo' + } + const queryImpl = ((params: Parameters[0]) => { + if (!params.options) { + throw new Error('missing SDK options') + } + params.options.spawnClaudeCodeProcess?.({ + command: 'claude', + args: [], + env: {}, + signal: new AbortController().signal + }) + void (async () => { + for await (const _message of params.prompt) { + // The SDK owns the transport write; the close test only needs its EOF boundary. + } + child.stdin.end() + })() + return (async function* () {})() + }) as typeof query + const connection = await openClaudeStreamJsonConnection(launch, {}, () => child, queryImpl) + + const closing = connection.close() + await new Promise((resolve) => setImmediate(resolve)) + expect(child.stdin.writableEnded).toBe(false) + + refreshDone.resolve() + await expect(closing).resolves.toBe(true) + expect(child.stdin.writableEnded).toBe(true) + }) + + it('requests a fresh close boundary after an output capture starts', async () => { + mocks.refresh.mockReset() + mocks.proveClaudeChildExit.mockReset() + const outputCapture = Promise.withResolvers() + const closeCapture = Promise.withResolvers() + mocks.refresh + .mockReturnValueOnce(outputCapture.promise) + .mockReturnValueOnce(closeCapture.promise) + mocks.proveClaudeChildExit.mockResolvedValueOnce(true) + const child = fakeChild() + const launch: ClaudeStreamJsonLaunch = { + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo' + } + const queryImpl = ((params: Parameters[0]) => { + params.options?.spawnClaudeCodeProcess?.({ + command: 'claude', + args: [], + env: {}, + signal: new AbortController().signal + }) + void (async () => { + for await (const _message of params.prompt) { + // The SDK owns the transport write; the close test only needs its EOF boundary. + } + child.stdin.end() + })() + return (async function* () {})() + }) as typeof query + const connection = await openClaudeStreamJsonConnection(launch, {}, () => child, queryImpl) + + child.stderr.emit('data', 'output') + await vi.waitFor(() => expect(mocks.refresh).toHaveBeenCalledTimes(1)) + const closing = connection.close() + await Promise.resolve() + + expect(mocks.refresh).toHaveBeenCalledTimes(2) + expect(child.stdin.writableEnded).toBe(false) + + outputCapture.resolve() + await Promise.resolve() + expect(child.stdin.writableEnded).toBe(false) + closeCapture.resolve() + await expect(closing).resolves.toBe(true) + expect(child.stdin.writableEnded).toBe(true) + }) +}) diff --git a/src/main/claude/claude-stream-json-connection.test.ts b/src/main/claude/claude-stream-json-connection.test.ts new file mode 100644 index 00000000000..c4f1f9a6fca --- /dev/null +++ b/src/main/claude/claude-stream-json-connection.test.ts @@ -0,0 +1,768 @@ +import { execFileSync } from 'node:child_process' +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { spawnProcess, type SpawnedProcess } from '../../shared/child-process/run-process' +import { hasLiveClaudePtys } from '../claude-accounts/live-pty-gate' +import type { ProcessSpec } from '../../shared/child-process/process-spec' +import { query, type CanUseTool, type Options } from '@anthropic-ai/claude-agent-sdk' +import { + openClaudeStreamJsonConnection, + type ClaudeStreamJsonConnection, + type ClaudeStreamJsonLaunch +} from './claude-stream-json-connection' +import { openAgentSessionJournal } from '../native-chat/agent-session-journal/journal-store-factory' +import { createDeferredStructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { claudeAuthDiagnostic } from './claude-structured-init-proof' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' +import { readClaudeStructuredSessionOptions } from './claude-structured-session-options' +import type { ClaudeSession } from './claude-structured-session-state' +import { CLAUDE_STRUCTURED_BASE_OPTIONS } from './claude-structured-launch-resolution' + +// These drive the real SDK against the scripted fake CLI, so every assertion is +// about the environment, argv and frames a real child actually saw. +const FAKE_CLI = join(__dirname, '__fixtures__', 'claude-agent-sdk-scripted-cli.mjs') +const SESSION_ID = '5348c19f-6a54-4c2e-9c68-9c2b1a3d4e5f' +const HOLD_OPEN = { delayMs: 10_000 } + +type ScriptedCliReport = { + argv: string[] + controlRequests: { request_id: string; request: { subtype: string } }[] + controlResponses: { response: { request_id: string; response?: unknown } }[] + userMessages: Record[] + descendantPid: number | null +} + +const scratchDirs: string[] = [] +const openConnections: ClaudeStreamJsonConnection[] = [] + +afterEach(async () => { + for (const connection of openConnections.splice(0)) { + await connection.close() + } + for (const dir of scratchDirs.splice(0)) { + rmSync(dir, { recursive: true, force: true }) + } + spawned.splice(0) + spawnedChildren.splice(0) + vi.unstubAllEnvs() +}) + +function scriptScenario( + steps: Record[], + controlResponses: Record = {} +) { + const dir = mkdtempSync(join(tmpdir(), 'claude-sdk-connection-')) + scratchDirs.push(dir) + const scenarioPath = join(dir, 'scenario.json') + const reportPath = join(dir, 'report.json') + writeFileSync(scenarioPath, JSON.stringify({ steps, controlResponses })) + return { + cwd: dir, + env: { + PATH: process.env.PATH ?? '', + ORCA_SDK_CONTRACT_SCENARIO_PATH: scenarioPath, + ORCA_SDK_CONTRACT_REPORT_PATH: reportPath + }, + readReport: () => JSON.parse(readFileSync(reportPath, 'utf8')) as ScriptedCliReport + } +} + +function launchFor( + scenario: { cwd: string; env: Record }, + env: Record = {} +): ClaudeStreamJsonLaunch { + return { + pathToClaudeCodeExecutable: FAKE_CLI, + options: { ...CLAUDE_STRUCTURED_BASE_OPTIONS, sessionId: SESSION_ID }, + cwd: scenario.cwd, + env: { ...scenario.env, ...env } + } +} + +/** The derived child environment, captured where Orca actually hands it to the OS. */ +const spawned: ProcessSpec[] = [] +/** The retained child, so a test can end it the way a crashing CLI would. */ +const spawnedChildren: SpawnedProcess[] = [] + +async function open( + launch: ClaudeStreamJsonLaunch, + handlers: Parameters[1] = {}, + queryImpl?: typeof query +): Promise { + const connection = await openClaudeStreamJsonConnection( + launch, + handlers, + (spec) => { + spawned.push(spec) + const child = spawnProcess(spec) + spawnedChildren.push(child) + return child + }, + queryImpl + ) + openConnections.push(connection) + return connection +} + +function childEnv(): Record { + return (spawned.at(-1)?.env ?? {}) as Record +} + +async function until(read: () => T | null | undefined, label: string): Promise { + for (let attempt = 0; attempt < 400; attempt++) { + const value = read() + if (value !== null && value !== undefined) { + return value + } + await new Promise((resolve) => setTimeout(resolve, 25)) + } + throw new Error(`timed out waiting for ${label}`) +} + +function readReportSafely(scenario: { readReport: () => ScriptedCliReport }) { + try { + return scenario.readReport() + } catch { + return null + } +} + +function processState(pid: number): 'running' | 'exited' { + try { + const state = execFileSync('ps', ['-o', 'state=', '-p', String(pid)], { + encoding: 'utf8', + env: { ...process.env, LANG: 'C', LC_ALL: 'C' } + }).trim() + return state.startsWith('Z') ? 'exited' : 'running' + } catch (error) { + if ((error as { status?: number }).status === 1) { + return 'exited' + } + throw error + } +} + +describe('Claude stream-json connection', () => { + it('passes the Claude Code system-prompt preset through to SDK query', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + let captured: Options | undefined + await open(launchFor(scenario), {}, (params) => { + captured = params.options + return query(params) + }) + + expect(captured?.systemPrompt).toEqual({ type: 'preset', preset: 'claude_code' }) + }) + + it('hands the child a derived environment, the resolved CLI path, and keeps the pid', async () => { + vi.stubEnv('ANTHROPIC_API_KEY', 'sk-ant-SHELL-LEAK') + vi.stubEnv('CLAUDE_CODE_CHILD_SESSION', '1') + vi.stubEnv('NODE_OPTIONS', '--require=/tmp/inject.js') + // An inherited value wins over the SDK's default, so clear it to pin the default. + vi.stubEnv('CLAUDE_CODE_ENTRYPOINT', undefined) + vi.stubEnv('ORCA_CONNECTION_MARKER', 'inherited') + const scenario = scriptScenario([HOLD_OPEN]) + const connection = await open( + launchFor(scenario, { + CLAUDE_CONFIG_DIR: '/accounts/managed/home', + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ORCA_AGENT_SESSION_SPAWN_TOKEN: 'spawn-9', + CLAUDE_CODE_CHILD_SESSION: 'configured-child-session', + CLAUDE_CODE_SESSION_ID: 'configured-session', + CLAUDE_CODE_BRIDGE_SESSION_ID: 'configured-bridge-session' + }) + ) + + // Ownership proof: the pid is a real live process, not a value the SDK reported. + expect(connection.pid).toEqual(expect.any(Number)) + expect(() => process.kill(connection.pid as number, 0)).not.toThrow() + const env = childEnv() + // The managed home is pinned verbatim: the CLI keys credential lookup on the literal string. + expect(env.CLAUDE_CONFIG_DIR).toBe('/accounts/managed/home') + expect(env.ANTHROPIC_AUTH_TOKEN).toBe('configured-token') + expect(env.ORCA_AGENT_SESSION_SPAWN_TOKEN).toBe('spawn-9') + expect(env.ORCA_CONNECTION_MARKER).toBe('inherited') + expect(env.ANTHROPIC_API_KEY).toBeUndefined() + expect(env.CLAUDE_CODE_CHILD_SESSION).toBeUndefined() + expect(env.CLAUDE_CODE_SESSION_ID).toBeUndefined() + expect(env.CLAUDE_CODE_BRIDGE_SESSION_ID).toBeUndefined() + // Two SDK mutations of the child env, pinned so a bump cannot change them unseen. + expect(env.CLAUDE_CODE_ENTRYPOINT).toBe('sdk-ts') + expect(env.NODE_OPTIONS).toBeUndefined() + // The bundled binary is excluded from the install, so the resolved path is mandatory. + const report = await until(() => readReportSafely(scenario), 'the scripted CLI report') + expect(report.argv[0]).toBe(FAKE_CLI) + // The .mjs fixture makes the SDK run it under node; a real CLI path is the program + // itself. Either way the resolved path is what Orca's spawner is asked to execute. + expect([spawned.at(-1)?.program, ...(spawned.at(-1)?.args ?? [])]).toContain(FAKE_CLI) + expect(report.argv).toContain('--replay-user-messages') + expect(report.argv).toContain(`--session-id=${SESSION_ID}`) + }) + + it('leaves the default CLI home unpinned so macOS Keychain OAuth keeps working', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + await open(launchFor(scenario)) + + await until(() => readReportSafely(scenario), 'the scripted CLI report') + expect(childEnv().CLAUDE_CONFIG_DIR).toBeUndefined() + }) + + it('settles a send only once the frame reached the child, and replays reach onMessage', async () => { + const replay = { + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'hello' }] }, + parent_tool_use_id: null, + isReplay: true, + session_id: SESSION_ID, + uuid: 'uuid-replay-1' + } + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: replay }, HOLD_OPEN]) + const messages: Record[] = [] + const connection = await open(launchFor(scenario), { + onMessage: (message) => messages.push(message) + }) + + await connection.send({ + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'hello' }] }, + parent_tool_use_id: null, + session_id: SESSION_ID + }) + // The report exists from the child's first line of work, so poll for the frame + // itself: `send` settles on the SDK's completed write, and the child still has + // to read that line before it can record it. + const report = await until( + () => (readReportSafely(scenario)?.userMessages.length ? readReportSafely(scenario) : null), + 'the user frame recorded by the child' + ) + expect(report.userMessages).toHaveLength(1) + + await until(() => messages.find((message) => message.uuid === 'uuid-replay-1'), 'the replay') + // The replay is delivered verbatim, so the dispatch acknowledgement still binds on it. + expect(messages.find((message) => message.uuid === 'uuid-replay-1')).toEqual(replay) + }) + + it('rejects a send the SDK pulled but could not write to a terminated child', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, HOLD_OPEN]) + const connection = await open(launchFor(scenario)) + const child = spawnedChildren.at(-1) + + // Same tick as the send, so the liveness guard still passes and the frame + // reaches the SDK's input pump: its `transport.write` is what fails, which is + // the window a child crashing mid-send actually opens. + child?.kill('SIGKILL') + const sent = connection.send({ + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'hello' }] }, + parent_tool_use_id: null, + session_id: SESSION_ID + }) + + await expect(sent).rejects.toThrow() + expect(readReportSafely(scenario)?.userMessages ?? []).toHaveLength(0) + }) + + it('delivers an unmodeled frame verbatim so the provider-fallback row survives', async () => { + const unknown = { + type: 'frame_kind_from_the_future', + session_id: SESSION_ID, + uuid: 'uuid-unknown-1', + payload: { nested: { flags: ['a', 'b'] } } + } + const scenario = scriptScenario([{ emit: unknown }, HOLD_OPEN]) + const messages: Record[] = [] + await open(launchFor(scenario), { onMessage: (message) => messages.push(message) }) + + await until(() => messages.find((message) => message.uuid === 'uuid-unknown-1'), 'the frame') + expect(messages.find((message) => message.uuid === 'uuid-unknown-1')).toEqual(unknown) + }) + + it('commits the real partial-message cadence as one assistant item through the translator', async () => { + // The frame order and per-frame uuids are the ones Claude Code 2.1.258 emits + // under --include-partial-messages: every stream_event and the block's final + // assistant frame each carry their own uuid; only message.id ties them. + const stream = (uuid: string, event: Record) => ({ + type: 'stream_event', + uuid, + session_id: SESSION_ID, + parent_tool_use_id: null, + event + }) + const frames = [ + stream('uuid-message-start', { + type: 'message_start', + message: { id: 'msg_01', role: 'assistant', content: [] } + }), + stream('uuid-block-start', { + type: 'content_block_start', + index: 0, + content_block: { type: 'text', text: '' } + }), + stream('uuid-delta-1', { + type: 'content_block_delta', + index: 0, + delta: { type: 'text_delta', text: 'ST' } + }), + stream('uuid-delta-2', { + type: 'content_block_delta', + index: 0, + delta: { type: 'text_delta', text: 'REAMOK_ELEC_64E632' } + }), + { + type: 'assistant', + uuid: 'uuid-assistant-final', + session_id: SESSION_ID, + parent_tool_use_id: null, + message: { + id: 'msg_01', + role: 'assistant', + content: [{ type: 'text', text: 'STREAMOK_ELEC_64E632' }], + stop_reason: null + } + }, + stream('uuid-block-stop', { type: 'content_block_stop', index: 0 }), + stream('uuid-message-delta', { type: 'message_delta', delta: { stop_reason: 'end_turn' } }), + stream('uuid-message-stop', { type: 'message_stop' }), + { + type: 'result', + subtype: 'success', + is_error: false, + duration_ms: 1, + duration_api_ms: 1, + num_turns: 1, + result: 'STREAMOK_ELEC_64E632', + stop_reason: 'end_turn', + session_id: SESSION_ID, + uuid: 'uuid-result' + } + ] + const scenario = scriptScenario([...frames.map((frame) => ({ emit: frame })), HOLD_OPEN]) + const journal = await openAgentSessionJournal({ + identity: { + sessionId: 'session-1', + workspaceId: 'workspace-1', + hostId: 'host-1', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: SESSION_ID, leafUuid: 'leaf-1' } + }, + journalDir: join(scenario.cwd, 'journal'), + now: () => 1_700_000_000_000, + mintEpoch: () => 'epoch-1' + }) + const deferred = createDeferredStructuredAgentSessionEventSink() + deferred.bind({ journal, fence: 1, publish: vi.fn() }) + const translator = createClaudeJournalTranslator({ sink: deferred.sink }) + let settled = false + await open(launchFor(scenario), { + onMessage: (message) => { + translator.handle({ type: 'message', sessionId: 'session-1', message }) + settled ||= message.type === 'result' + } + }) + + await until(() => (settled ? true : null), 'the result frame') + await deferred.drained() + const items = journal.snapshot().items + const assistant = items.filter( + (item) => item.body.kind === 'message' && item.body.role === 'assistant' + ) + expect(assistant.map((item) => item.body)).toEqual([ + { + kind: 'message', + role: 'assistant', + blocks: [{ type: 'text', text: 'STREAMOK_ELEC_64E632' }] + } + ]) + expect(assistant.map((item) => item.itemId)).toEqual([`claude:${SESSION_ID}:uuid-block-start`]) + expect( + items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame ? [item.body.providerFrame.kind] : [] + ) + ).toEqual([]) + // The journal owns a SQLite connection now; afterEach removes this temp root and an open + // handle blocks that on Windows. + await journal.close() + }) + + it('feeds an inbound permission request to canUseTool and writes its answer back on the same id', async () => { + const scenario = scriptScenario([ + { + emit: { + type: 'control_request', + request_id: 'perm-421', + request: { + subtype: 'can_use_tool', + tool_name: 'Bash', + input: { command: 'ls' }, + tool_use_id: 'toolu_1', + permission_suggestions: [{ type: 'addRules' }] + } + } + }, + { awaitControlResponse: 'perm-421' }, + HOLD_OPEN + ]) + const seen: { toolName: string; requestId: string; toolUseID: string; suggestions: unknown }[] = + [] + const canUseTool: CanUseTool = (toolName, _input, options) => { + seen.push({ + toolName, + requestId: options.requestId, + toolUseID: options.toolUseID, + suggestions: options.suggestions + }) + return Promise.resolve({ behavior: 'deny', message: 'No', toolUseID: options.toolUseID }) + } + await open(launchFor(scenario), { canUseTool }) + + await until(() => (seen.length > 0 ? seen : null), 'the inbound permission request') + expect(seen).toEqual([ + { + toolName: 'Bash', + requestId: 'perm-421', + toolUseID: 'toolu_1', + suggestions: [{ type: 'addRules' }] + } + ]) + const written = await until( + () => + readReportSafely(scenario)?.controlResponses.find( + (frame) => frame.response.request_id === 'perm-421' + ), + 'the permission answer' + ) + expect(written.response.response).toMatchObject({ behavior: 'deny', message: 'No' }) + }) + + it('drives Orca control methods onto the SDK and times out with the init proof message', async () => { + const scenario = scriptScenario([HOLD_OPEN], { + initialize: { models: [{ value: 'sonnet' }], account: { tokenSource: 'oauth' } }, + get_settings: { env: { ANTHROPIC_BASE_URL: 'https://settings.example.test' } } + }) + const connection = await open(launchFor(scenario)) + + await expect(connection.initializationResult()).resolves.toMatchObject({ + models: [{ value: 'sonnet' }] + }) + await expect(connection.getSettings()).resolves.toEqual({ + env: { ANTHROPIC_BASE_URL: 'https://settings.example.test' } + }) + await expect(connection.setModel('opus')).resolves.toBeUndefined() + const requests = await until( + () => + readReportSafely(scenario)?.controlRequests.find( + (frame) => frame.request.subtype === 'set_model' + ), + 'the set_model control request' + ) + expect(requests.request.subtype).toBe('set_model') + }) + + it('reads supportedModels from the catalog the running CLI reported', async () => { + const scenario = scriptScenario([HOLD_OPEN], { + initialize: { + models: [ + { value: 'default', resolvedModel: 'claude-opus-5' }, + { + value: 'opus', + displayName: 'Opus 5', + description: 'The live row, not the seed', + resolvedModel: 'claude-opus-5', + supportsEffort: true, + supportedEffortLevels: ['low', 'high'] + } + ] + } + }) + const connection = await open(launchFor(scenario)) + + await expect(connection.supportedModels()).resolves.toMatchObject([ + { value: 'default', resolvedModel: 'claude-opus-5' }, + { value: 'opus', displayName: 'Opus 5', supportedEffortLevels: ['low', 'high'] } + ]) + }) + + it('serves the picker the live catalog rather than falling back to the static seed', async () => { + const scenario = scriptScenario([HOLD_OPEN], { + initialize: { + models: [ + { value: 'default', resolvedModel: 'claude-opus-5' }, + { + value: 'opus', + displayName: 'Opus 5', + description: 'The live row, not the seed', + resolvedModel: 'claude-opus-5', + supportsEffort: true, + supportedEffortLevels: ['low', 'high'] + } + ] + } + }) + const connection = await open(launchFor(scenario)) + const session = { + connection, + options: new Map(), + reportedOptions: {} + } as unknown as ClaudeSession + + const options = await readClaudeStructuredSessionOptions(session, 5_000) + + // The seed carries neither this description nor a two-level effort list, so + // both can only have come from the child. + expect(options.models).toContainEqual({ + id: 'opus', + label: 'Opus 5', + description: 'The live row, not the seed', + isDefault: true, + efforts: [ + { value: 'low', label: 'Low' }, + { value: 'high', label: 'High' } + ] + }) + expect(options.current.model).toBe('opus') + }) + + it('feeds the auth diagnostic from the settings the running child reports', async () => { + for (const key of ['ANTHROPIC_BASE_URL', 'ANTHROPIC_AUTH_TOKEN', 'ANTHROPIC_API_KEY']) { + vi.stubEnv(key, undefined) + } + const scenario = scriptScenario([HOLD_OPEN], { + get_settings: { + env: { + ANTHROPIC_BASE_URL: 'https://settings.example.test', + ANTHROPIC_AUTH_TOKEN: 'secret' + } + } + }) + const connection = await open(launchFor(scenario)) + const init = { providerSessionId: SESSION_ID, uuid: null, model: null, message: {} } + + // With no ambient auth, every true below can only have come from the CLI's settings. + expect(claudeAuthDiagnostic(init, null)).toMatchObject({ + baseUrlConfigured: false, + authTokenConfigured: false + }) + const diagnostic = claudeAuthDiagnostic(init, await connection.getSettings()) + expect(diagnostic).toMatchObject({ + baseUrlConfigured: true, + authTokenConfigured: true, + apiKeyConfigured: false + }) + expect(JSON.stringify(diagnostic)).not.toContain('secret') + }) + + it('reports an unauthenticated start through the init deadline instead of hanging', async () => { + // The scripted CLI never answers, which is the shape of a silently unauthenticated CLI. + const scenario = scriptScenario([HOLD_OPEN]) + const connection = await open({ + ...launchFor(scenario), + env: { ...launchFor(scenario).env, ORCA_SDK_CONTRACT_IGNORE_CONTROL_REQUESTS: '1' } + }) + + await expect(connection.initializationResult({ timeoutMs: 200 })).rejects.toThrow( + 'claude initialize request timed out' + ) + }) + + it('reports a self-exit with its status and stderr, and leaves its tree unverifiable', async () => { + const scenario = scriptScenario([{ stderr: 'claude: not signed in\n' }, { exit: 1 }]) + let exit: Error | null = null + const connection = await open(launchFor(scenario), { + onExit: (error) => { + exit = error + } + }) + + await until(() => exit, 'the exit error') + // The status and stderr are the only diagnostic a refused start leaves behind. + expect((exit as unknown as Error).message).toMatch(/exited \(code 1\): claude: not signed in/) + expect(connection.closed).toBe(true) + // The root's exit is first-hand, but it left before a descendant snapshot + // could be armed, so close() has no tree proof to offer and says so. + await expect(connection.close()).resolves.toBe(false) + expect(connection.exitVerdict).toEqual({ root: 'exited', tree: 'unverifiable' }) + }) + + it.runIf(process.platform !== 'win32')( + 'proves a natural SDK exit and cleans up its descendant before recovery', + async () => { + const scenario = scriptScenario([ + { stderr: 'claude: natural exit\n' }, + { delayMs: 500 }, + { exit: 1 } + ]) + let exit: Error | null = null + const connection = await open( + { + ...launchFor(scenario), + env: { ...launchFor(scenario).env, ORCA_SDK_CONTRACT_DESCENDANT: '1' } + }, + { onExit: (error) => (exit = error) } + ) + const report = await until(() => { + const current = readReportSafely(scenario) + return current?.descendantPid ? current : null + }, 'the descendant report') + await until(() => exit, 'the natural exit error') + try { + await expect(connection.close()).resolves.toBe(true) + expect(connection.exitVerdict).toEqual({ root: 'exited', tree: 'exited' }) + expect(processState(report.descendantPid as number)).toBe('exited') + } finally { + try { + process.kill(report.descendantPid as number, 'SIGKILL') + } catch { + // Already gone. + } + } + }, + 20_000 + ) + + it('settles a spawn error followed by close as processless and closes idempotently', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + const missingCli = join(scenario.cwd, 'claude-that-does-not-exist') + let fault: Error | null = null + let exit: Error | null = null + const connection = await open( + { ...launchFor(scenario), pathToClaudeCodeExecutable: missingCli }, + { + onFault: (error) => { + fault = error + }, + onExit: (error) => { + exit = error + } + } + ) + + await until( + () => (connection.exitVerdict.root === 'processless' ? connection.exitVerdict : null), + 'the processless spawn settlement' + ) + expect(connection.pid).toBeUndefined() + expect(fault).toBeInstanceOf(Error) + expect(exit).toBeNull() + await expect(Promise.all([connection.close(), connection.close()])).resolves.toEqual([ + true, + true + ]) + await expect(connection.close()).resolves.toBe(true) + expect(connection.exitVerdict).toEqual({ root: 'processless', tree: 'exited' }) + }) + + it('does not treat a child error event as first-hand root exit proof', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + let exit: Error | null = null + const connection = await open(launchFor(scenario), { + onExit: (error) => { + exit = error + } + }) + const child = spawnedChildren.at(-1) + expect(child).toBeDefined() + + child?.emit('error', new Error('child transport fault')) + + expect(exit).toBeNull() + expect(connection.exitVerdict.root).toBe('live') + await until(() => exit, 'the distinct child exit') + expect(connection.exitVerdict.root).toBe('exited') + }) + + it('proves the exit of a child that ignores a graceful shutdown', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + const connection = await open({ + ...launchFor(scenario), + env: { ...launchFor(scenario).env, ORCA_SDK_CONTRACT_IGNORE_SIGTERM: '1' } + }) + + // Keep the lstart capture boundary outside the child's displayed start second. + await new Promise((resolve) => setTimeout(resolve, 1_100)) + await expect(connection.close()).resolves.toBe(true) + }, 20_000) +}) + +// A structured Claude child owns the account's credentials while it runs, exactly as +// a Claude PTY does. The gate is what makes runtime-auth-sync defer the managed OAuth +// refresh instead of rotating the single-use token out from under a live session, and +// structured sessions used to be invisible to it. +describe('the managed-auth live gate', () => { + it('holds while a structured child runs and releases when it ends', async () => { + // The gate is a process-wide singleton and a sibling test's release lands on its + // child's 'close' event, which can settle after that test's close() resolved. + await until(() => (hasLiveClaudePtys() ? null : true), 'a drained auth gate') + const scenario = scriptScenario([ + { emit: { type: 'system', subtype: 'init', session_id: SESSION_ID, uuid: 'init-1' } }, + { wait: HOLD_OPEN } + ]) + const connection = await open(launchFor(scenario)) + + expect(hasLiveClaudePtys()).toBe(true) + + await connection.close() + + await until(() => (hasLiveClaudePtys() ? null : true), 'the auth gate to drain') + expect(hasLiveClaudePtys()).toBe(false) + }, 30_000) + + it('releases when the child dies on its own rather than through close()', async () => { + await until(() => (hasLiveClaudePtys() ? null : true), 'a drained auth gate') + const scenario = scriptScenario([ + { emit: { type: 'system', subtype: 'init', session_id: SESSION_ID, uuid: 'init-1' } }, + { wait: HOLD_OPEN } + ]) + await open(launchFor(scenario)) + expect(hasLiveClaudePtys()).toBe(true) + + spawnedChildren.at(-1)?.kill('SIGKILL') + + await until(() => (hasLiveClaudePtys() ? null : true), 'the auth gate to drain') + expect(hasLiveClaudePtys()).toBe(false) + }, 30_000) + + // The gate entry is deliberately unpersisted, so confirmSeededClaudeLivePtys can never + // reconcile a stray one: a leak here defers the managed OAuth refresh for the life of + // the process. Entering the gate only after the release handlers are attached makes + // that unreachable regardless of what the setup in between does. + it('leaks no gate entry when setup throws between spawn and handler attachment', async () => { + await until(() => (hasLiveClaudePtys() ? null : true), 'a drained auth gate') + const scenario = scriptScenario([ + { emit: { type: 'system', subtype: 'init', session_id: SESSION_ID, uuid: 'init-1' } }, + { wait: HOLD_OPEN } + ]) + let started: SpawnedProcess | null = null + + try { + await expect( + openClaudeStreamJsonConnection(launchFor(scenario), {}, (spec) => { + const child = spawnProcess(spec) + started = child + const attach = child.stderr.on.bind(child.stderr) + // Measured attach order: the SDK binds stderr 'data' from inside query(), + // before the child is even assigned. The SECOND bind is this connection's own + // armTreeOnOutput — the first statement that runs after the child exists and + // before its 'exit'/'close' release handlers. Throwing on the first is + // vacuous: it escapes before any gate entry could have happened. + let dataAttaches = 0 + child.stderr.on = ((event: string, listener: (...args: unknown[]) => void) => { + if (event === 'data') { + dataAttaches += 1 + if (dataAttaches === 2) { + throw new Error('stderr listener attach failed') + } + } + return attach(event, listener) + }) as typeof child.stderr.on + return child + }) + ).rejects.toThrow('stderr listener attach failed') + + expect(hasLiveClaudePtys()).toBe(false) + } finally { + ;(started as SpawnedProcess | null)?.kill('SIGKILL') + } + }, 30_000) +}) diff --git a/src/main/claude/claude-stream-json-connection.ts b/src/main/claude/claude-stream-json-connection.ts new file mode 100644 index 00000000000..dd6bbc8a5eb --- /dev/null +++ b/src/main/claude/claude-stream-json-connection.ts @@ -0,0 +1,283 @@ +import { randomUUID } from 'node:crypto' +import type * as ClaudeAgentSdk from '@anthropic-ai/claude-agent-sdk' +import type { CanUseTool, OnUserDialog, SDKUserMessage } from '@anthropic-ai/claude-agent-sdk' +import { spawnProcess } from '../../shared/child-process/run-process' +import { + markClaudeStructuredChildExited, + markClaudeStructuredChildSpawned +} from '../claude-accounts/live-pty-gate' +import { buildClaudeChildProcessEnv } from './claude-child-process-environment' +import { + ClaudeControlRequestError, + createClaudeControlSurface, + type ClaudeControlSurface +} from './claude-agent-sdk-control-requests' +import { createClaudeChildTreeReaper, proveClaudeChildExit } from './claude-agent-sdk-exit-proof' +import type { DescendantTreeVerdict } from '../pty-descendant-exit-verification' +import { createClaudeCodeProcessSpawn } from './claude-agent-sdk-process-spawn' +import { createClaudeUserMessageQueue } from './claude-agent-sdk-user-message-queue' +import type { ClaudeStructuredSdkOptions } from './claude-structured-launch-resolution' + +export { ClaudeControlRequestError } + +/** + * The SDK is loaded at the structured-Claude boundary rather than by this module's + * import. The ordinary runtime's class graph statically reaches this file, and the + * SDK sets `process.env.NoDefaultCurrentDirectoryInExePath` at import time — a + * Windows executable-search change that a user who never leaves the terminal/TUI + * path never opted into, and a missing SDK would fail runtime startup. Memoized, + * so a session pays the import once per process rather than once per connection. + */ +let claudeAgentSdk: Promise | null = null + +function loadClaudeAgentSdk(): Promise { + claudeAgentSdk ??= import('@anthropic-ai/claude-agent-sdk') + return claudeAgentSdk +} + +export type ClaudeStreamJsonLaunch = { + /** Orca's resolved user CLI; the SDK falls back to a bundled binary that is not installed. */ + pathToClaudeCodeExecutable: string + options: ClaudeStructuredSdkOptions + cwd: string + env?: Record +} + +export type ClaudeStreamJsonConnectionHandlers = { + onMessage?: (message: Record) => void + /** + * The SDK owns inbound permission control: it hands `can_use_tool` to this callback with + * a stable requestId and an abort signal, dedups duplicate delivery, and matches the + * response by request_id itself. Setting it makes the SDK pass `--permission-prompt-tool + * stdio` automatically; it must not be paired with `permissionPromptToolName`. + */ + canUseTool?: CanUseTool + /** `request_user_dialog` control; the CLI only emits kinds declared in `supportedDialogKinds`. */ + onUserDialog?: OnUserDialog + /** A transport/process fault that is not itself first-hand root exit proof. */ + onFault?: (error: Error) => void + onExit?: (error: Error) => void +} + +/** + * Two questions with their own evidence. The root's verdict is first-hand: Orca's + * own child handle reported exit, or reported error then close before it ever had + * a pid. The tree's comes from bounded descendant verification, and `unverifiable` + * is never collapsed into either neighbour. + */ +export type ClaudeChildExitVerdict = { + root: 'exited' | 'live' | 'processless' + tree: DescendantTreeVerdict +} + +export type ClaudeStreamJsonConnection = ClaudeControlSurface & { + readonly pid: number | undefined + readonly closed: boolean + /** What the ladder has observed so far; read after a `close()` that returned false. */ + readonly exitVerdict: ClaudeChildExitVerdict + send: (message: Record) => Promise + /** Resolves true after processless settlement, or root exit plus observed tree exit. */ + close: () => Promise +} + +type ExitStatus = { code: number | null; signal: NodeJS.Signals | null } + +function exitError(stderrTail: string, status: ExitStatus | null, cause?: Error): Error { + const detail = stderrTail.trim() + // The status is the diagnostic a signed-out or refused start leaves behind; + // it has to survive every wrapper between here and the user. + const how = + status?.signal !== null && status?.signal !== undefined + ? ` (signal ${status.signal})` + : status?.code !== null && status?.code !== undefined + ? ` (code ${status.code})` + : '' + const message = `claude stream-json exited${how}${detail ? `: ${detail}` : ''}` + return cause ? new Error(message, { cause }) : new Error(message) +} + +export async function openClaudeStreamJsonConnection( + launch: ClaudeStreamJsonLaunch, + handlers: ClaudeStreamJsonConnectionHandlers = {}, + spawnImpl: typeof spawnProcess = spawnProcess, + queryImpl?: typeof ClaudeAgentSdk.query +): Promise { + const { query } = await loadClaudeAgentSdk() + const spawner = createClaudeCodeProcessSpawn(spawnImpl) + const inbox = createClaudeUserMessageQueue() + const session = (queryImpl ?? query)({ + prompt: inbox.messages, + options: { + ...launch.options, + cwd: launch.cwd, + // Why env is never omitted: the SDK inherits process.env when it is, which is + // exactly the ambient ANTHROPIC_* auth leak this lane already shipped once. + env: buildClaudeChildProcessEnv(launch.env, { scrubConfiguredChildSessionStamps: true }), + pathToClaudeCodeExecutable: launch.pathToClaudeCodeExecutable, + spawnClaudeCodeProcess: spawner.spawn, + ...(handlers.canUseTool ? { canUseTool: handlers.canUseTool } : {}), + ...(handlers.onUserDialog ? { onUserDialog: handlers.onUserDialog } : {}) + } + }) + const child = spawner.child + if (!child) { + throw new Error('the claude agent SDK returned without spawning a child') + } + // This child owns the account's credentials for as long as it runs, exactly as a + // Claude PTY does — hold the OAuth-refresh gate so a managed refresh cannot rotate + // the single-use token out from under it mid-turn. Entered below, once a release + // path exists. + const authGateKey = randomUUID() + const releaseAuthGate = (): void => markClaudeStructuredChildExited(authGateKey) + let exited = false + let exitStatus: ExitStatus | null = null + let closing = false + let processless = false + let prePidSpawnError = false + let terminalError: Error | null = null + let faultReported = false + let exitReported = false + let closePromise: Promise | null = null + // One reaper per child: every close attempt and error-path reap shares its proof. + const rootSettled = (): boolean => exited || processless + const tree = createClaudeChildTreeReaper(child, { exited: rootSettled }) + + // Arm lazily on actual child output instead of issuing a process-table scan for + // every session at startup. A natural SDK exit can race a later close, while + // output-triggered observation still catches the usual live-child window. + let outputObservationArmed = false + const armTreeOnOutput = (): void => { + if (outputObservationArmed) { + return + } + outputObservationArmed = true + void (tree.refresh?.() ?? tree.capture()) + } + child.stderr.on('data', armTreeOnOutput) + // The SDK may synchronously spawn the CLI and consume an early stderr chunk + // before this connection can attach its listener; the bounded tail preserves + // that observation for the same lazy arm. + if (spawner.stderrTail.length > 0) { + armTreeOnOutput() + } + + let settleExit = (): void => {} + const exitPromise = new Promise((resolve) => { + settleExit = resolve + }) + const markExited = (): void => { + exited = true + releaseAuthGate() + settleExit() + } + child.on('exit', (code, signal) => { + exitStatus = { code, signal } + markExited() + handleUnexpectedEnd() + }) + + const handleUnexpectedEnd = (cause?: Error): void => { + terminalError ??= exitError(spawner.stderrTail, exitStatus, cause) + inbox.fail(terminalError) + if (!closing && !faultReported) { + faultReported = true + handlers.onFault?.(terminalError) + } + if (!closing && exited && !exitReported) { + exitReported = true + handlers.onExit?.(terminalError) + } + } + + void (async () => { + for await (const message of session) { + handlers.onMessage?.(message as unknown as Record) + } + })().catch((error: unknown) => { + // The SDK ends its generator in error when the child dies or the transport + // fails; a transport failure with a live child still has to reap the tree. + if (!closing && !exited) { + void tree.reap() + } + handleUnexpectedEnd(error instanceof Error ? error : new Error(String(error))) + }) + + child.on('error', (error) => { + if (spawner.pid === undefined) { + prePidSpawnError = true + } + if (!closing && !exited) { + void tree.reap() + } + handleUnexpectedEnd(error) + }) + child.on('close', () => { + // Covers the spawn-failure path too, where no 'exit' ever arrives. + releaseAuthGate() + if (prePidSpawnError && spawner.pid === undefined) { + processless = true + settleExit() + } + handleUnexpectedEnd() + }) + child.stdin.on('error', (error) => { + if (!closing) { + void tree.reap() + handleUnexpectedEnd(error) + } + }) + // Why here and not at spawn: a structured gate entry is deliberately unpersisted, so + // confirmSeededClaudeLivePtys can never reconcile a stray one and a leak defers the + // managed OAuth refresh for the life of the process. Entering only after 'exit' and + // 'close' are attached makes that unreachable — any later throw still leaves a + // listener that releases. Nothing between spawn and here can yield, so the child + // cannot end before the gate is entered. + markClaudeStructuredChildSpawned(authGateKey) + + const send = (message: Record): Promise => { + if (closing || exited || terminalError || child.stdin.destroyed || !child.stdin.writable) { + return Promise.reject(terminalError ?? new Error('claude stream-json connection is closed')) + } + return inbox.push(message as unknown as SDKUserMessage) + } + + const close = (): Promise => { + closePromise ??= (async () => { + closing = true + // Arm the descendant proof before ending stdin. The SDK may exit the root + // immediately; a post-exit walk cannot recover descendants that reparented. + await (tree.refresh?.() ?? tree.capture()) + inbox.end() + const proven = await proveClaudeChildExit({ + child, + exitPromise, + exited: rootSettled, + tree + }) + inbox.fail(new Error('claude stream-json connection closed')) + if (!proven) { + closePromise = null + } + return proven + })() + return closePromise + } + + return { + ...createClaudeControlSurface(session), + get pid() { + return spawner.pid + }, + get closed() { + return closing || exited || terminalError !== null + }, + get exitVerdict() { + return { + root: processless ? 'processless' : exited ? 'exited' : 'live', + tree: tree.treeVerdict + } as const + }, + send, + close + } +} diff --git a/src/main/claude/claude-streamed-block-identity.ts b/src/main/claude/claude-streamed-block-identity.ts new file mode 100644 index 00000000000..5cbf6674159 --- /dev/null +++ b/src/main/claude/claude-streamed-block-identity.ts @@ -0,0 +1,110 @@ +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import { claudeRecord, claudeText } from './claude-structured-item-translation' + +// Under --include-partial-messages every stream_event frame carries its own +// uuid, and the block's final `assistant` frame carries yet another; only +// `message.id` ties them together. The block's first stream frame mints the +// journal identity, and the final frame lands on it in block order instead of +// appending a duplicate under its own uuid. + +export type ClaudeStreamedTextDelta = { identity: AgentJournalItemIdentity; text: string } + +type StreamedMessage = { + messageId: string | null + blocks: Map + /** Streamed text blocks whose final assistant frame has not arrived, in block order. */ + awaitingFinal: AgentJournalItemIdentity[] +} + +export type ClaudeStreamedBlockRegistry = { + /** Text a stream_event frame appends to its block, or null when it carries none. */ + observe: (frame: Record) => ClaudeStreamedTextDelta | null + /** The streamed identity a final assistant frame reconciles onto, if its block streamed. */ + reconcile: (frame: { + sessionId: string + parentToolUseId: string | null + messageId: string | null + }) => AgentJournalItemIdentity | null + clear: () => void +} + +function scopeKey(sessionId: string, parentToolUseId: string | null): string { + return `${sessionId}/${parentToolUseId ?? ''}` +} + +export function createClaudeStreamedBlockRegistry(): ClaudeStreamedBlockRegistry { + const messages = new Map() + + const messageFor = (scope: string): StreamedMessage => { + let streamed = messages.get(scope) + if (!streamed) { + streamed = { messageId: null, blocks: new Map(), awaitingFinal: [] } + messages.set(scope, streamed) + } + return streamed + } + + const mint = ( + streamed: StreamedMessage, + sessionId: string, + index: number, + uuid: string + ): AgentJournalItemIdentity => { + const identity: AgentJournalItemIdentity = { provider: 'claude', sessionId, uuid } + streamed.blocks.set(index, identity) + streamed.awaitingFinal.push(identity) + return identity + } + + return { + observe: (frame) => { + const event = claudeRecord(frame.event) + const sessionId = claudeText(frame.session_id) + const uuid = claudeText(frame.uuid) + if (frame.type !== 'stream_event' || !event || !sessionId || !uuid) { + return null + } + const scope = scopeKey(sessionId, claudeText(frame.parent_tool_use_id)) + if (event.type === 'message_start') { + messages.set(scope, { + messageId: claudeText(claudeRecord(event.message)?.id), + blocks: new Map(), + awaitingFinal: [] + }) + return null + } + const index = typeof event.index === 'number' ? event.index : 0 + if (event.type === 'content_block_start') { + const block = claudeRecord(event.content_block) + if (block?.type !== 'text') { + return null + } + const identity = mint(messageFor(scope), sessionId, index, uuid) + const text = claudeText(block.text) + return text ? { identity, text } : null + } + if (event.type !== 'content_block_delta') { + return null + } + const delta = claudeRecord(event.delta) + const text = delta?.type === 'text_delta' ? claudeText(delta.text) : null + if (!text) { + return null + } + const streamed = messageFor(scope) + const identity = streamed.blocks.get(index) ?? mint(streamed, sessionId, index, uuid) + return { identity, text } + }, + reconcile: (frame) => { + const streamed = messages.get(scopeKey(frame.sessionId, frame.parentToolUseId)) + if ( + !streamed || + (frame.messageId && streamed.messageId && frame.messageId !== streamed.messageId) + ) { + return null + } + return streamed.awaitingFinal.shift() ?? null + }, + clear: () => messages.clear() + } +} diff --git a/src/main/claude/claude-streamed-text-checkpoints.test.ts b/src/main/claude/claude-streamed-text-checkpoints.test.ts new file mode 100644 index 00000000000..00a0bc0edd6 --- /dev/null +++ b/src/main/claude/claude-streamed-text-checkpoints.test.ts @@ -0,0 +1,93 @@ +import { describe, expect, it } from 'vitest' +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import { createClaudeStreamedTextCheckpoints } from './claude-streamed-text-checkpoints' + +function identityOf(uuid: string): AgentJournalItemIdentity { + return { provider: 'claude', sessionId: 'claude-session', uuid } +} + +function checkpoints() { + const rows: { uuid: string; text: string }[] = [] + let scheduled: (() => void) | null = null + const store = createClaudeStreamedTextCheckpoints({ + persist: (identity, text) => { + rows.push({ uuid: 'uuid' in identity ? identity.uuid : '', text }) + }, + schedule: (run) => { + scheduled = run + return () => { + scheduled = null + } + } + }) + return { + store, + rows, + runWindow: () => { + const run = scheduled as (() => void) | null + run?.() + } + } +} + +describe('claude streamed text checkpoints', () => { + it('rewrites a block row with the full text accumulated so far', () => { + const { store, rows, runWindow } = checkpoints() + + store.append(identityOf('block-1'), 'hel') + store.append(identityOf('block-1'), 'lo') + runWindow() + + expect(rows).toEqual([{ uuid: 'block-1', text: 'hello' }]) + expect(store.pending).toBe(1) + }) + + it('drops every block still awaiting its final frame at settlement', () => { + const { store, rows, runWindow } = checkpoints() + + store.append(identityOf('block-1'), 'partial answer') + runWindow() + store.settle() + + expect(store.pending).toBe(0) + // The row written before settlement stays; nothing is rewritten afterwards. + store.flush() + expect(rows).toEqual([{ uuid: 'block-1', text: 'partial answer' }]) + }) + + it('keeps a block whose final frame arrived out of the settlement sweep', () => { + const { store } = checkpoints() + + store.append(identityOf('block-1'), 'one') + store.append(identityOf('block-2'), 'two') + store.forget('claude:claude-session:block-1') + + expect(store.pending).toBe(1) + store.settle() + expect(store.pending).toBe(0) + }) + + it('flushes text the widening checkpoint interval has not written yet', () => { + const { store, rows } = checkpoints() + + store.append(identityOf('block-1'), 'x') + store.flush() + + expect(rows).toEqual([{ uuid: 'block-1', text: 'x' }]) + // Already at the row's length: a second flush has nothing to write. + store.flush() + expect(rows).toHaveLength(1) + }) + + it('stops persisting once disposed', () => { + const { store, rows, runWindow } = checkpoints() + + store.append(identityOf('block-1'), 'text') + store.dispose() + runWindow() + store.flush() + + expect(rows).toEqual([]) + expect(store.pending).toBe(0) + }) +}) diff --git a/src/main/claude/claude-streamed-text-checkpoints.ts b/src/main/claude/claude-streamed-text-checkpoints.ts new file mode 100644 index 00000000000..348ecd99558 --- /dev/null +++ b/src/main/claude/claude-streamed-text-checkpoints.ts @@ -0,0 +1,105 @@ +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { + createAgentSessionDeltaCoalescer, + type AgentSessionDeltaCoalescerDeps +} from '../native-chat/agent-session-wire/agent-session-delta-coalescer' + +export type ClaudeStreamedTextCheckpointDeps = { + /** Rewrites the block's journal row with the text accumulated so far. */ + persist: (identity: AgentJournalItemIdentity, text: string) => void + coalesceMs?: number + schedule?: AgentSessionDeltaCoalescerDeps['schedule'] +} + +export type ClaudeStreamedTextCheckpoints = { + /** Accumulate a delta; the row is rewritten on the coalescer's own cadence. */ + append: (identity: AgentJournalItemIdentity, text: string) => void + /** Write every block whose row is behind the text received for it. */ + flush: () => void + /** Drop one block's state, for a block whose final frame has now landed. */ + forget: (key: string) => void + /** + * Drop every block still awaiting its final frame, at turn settlement. Their + * text is already journaled by the flush that precedes settlement; keeping it + * live would grow with every interrupted turn for the life of the session. + */ + settle: () => void + /** Blocks still awaiting a final frame. A settled turn must leave none. */ + readonly pending: number + dispose: () => void +} + +/** + * Growth of a streamed block's row between its deltas and its final frame. + * + * The row is rewritten on a widening interval rather than per delta: a 200-line + * reply would otherwise rewrite the same journal row once per token. + */ +export function createClaudeStreamedTextCheckpoints( + deps: ClaudeStreamedTextCheckpointDeps +): ClaudeStreamedTextCheckpoints { + const identities = new Map() + const latestText = new Map() + const checkpointLengths = new Map() + + const persist = (key: string, text: string, force: boolean): void => { + latestText.set(key, text) + const checkpointLength = checkpointLengths.get(key) ?? 0 + const nextLength = Math.max(checkpointLength + 32, Math.ceil(checkpointLength * 1.125)) + if (!force && checkpointLength > 0 && text.length < nextLength) { + return + } + const identity = identities.get(key) + if (!identity) { + return + } + checkpointLengths.set(key, text.length) + deps.persist(identity, text) + } + + const coalescer = createAgentSessionDeltaCoalescer({ + ...(deps.coalesceMs === undefined ? {} : { windowMs: deps.coalesceMs }), + ...(deps.schedule ? { schedule: deps.schedule } : {}), + emit: (key, text) => persist(key, text, false) + }) + + const drop = (key: string): void => { + coalescer.forget(key) + identities.delete(key) + latestText.delete(key) + checkpointLengths.delete(key) + } + + return { + append: (identity, text) => { + const key = agentJournalItemKey(identity) + identities.set(key, identity) + coalescer.append(key, text) + }, + flush: () => { + coalescer.flushAll() + for (const [key, text] of latestText) { + if (checkpointLengths.get(key) !== text.length) { + persist(key, text, true) + } + } + }, + forget: drop, + settle: () => { + // Map iteration tolerates deletion of the entry just visited. + for (const key of identities.keys()) { + drop(key) + } + }, + get pending() { + return identities.size + }, + dispose: () => { + coalescer.dispose() + identities.clear() + latestText.clear() + checkpointLengths.clear() + } + } +} diff --git a/src/main/claude/claude-structured-acquisition-release.ts b/src/main/claude/claude-structured-acquisition-release.ts new file mode 100644 index 00000000000..6be06c86e93 --- /dev/null +++ b/src/main/claude/claude-structured-acquisition-release.ts @@ -0,0 +1,44 @@ +import { + closeClaudeSession, + claudeAcquisitionCleanupError +} from './claude-structured-session-close' +import type { + ClaudeAcquisitionRegistry, + ClaudeSession, + ClaudeSessionExit, + ClaudeStructuredSessionAdapterDeps +} from './claude-structured-session-state' + +/** + * Cleanup for an acquisition the host could not commit or prove. A session that + * a first-hand exit already removed is not an absence to report as proven: the + * ladder on its connection still answers, and that answer is classified exactly + * as a start-time failure would be. + */ +export async function releaseClaudeAcquisition(input: { + sessionId: string + sessions: Map + acquisitions: ClaudeAcquisitionRegistry + exits: Map + onExitProven?: (sessionId: string, exit: ClaudeSessionExit) => Promise + persistHandle?: ClaudeStructuredSessionAdapterDeps['persistHandle'] + onEvent?: ClaudeStructuredSessionAdapterDeps['onEvent'] + onBackgroundTasksChanged?: ClaudeStructuredSessionAdapterDeps['onBackgroundTasksChanged'] +}): Promise { + const exit = input.exits.get(input.sessionId) + if (!exit || input.sessions.has(input.sessionId) || input.acquisitions.get(input.sessionId)) { + return closeClaudeSession(input) + } + const firstProof = exit.closePromise ? await exit.closePromise : false + // A failed exit-path proof is retained as evidence, not as a terminal result; + // a release retry must drive a fresh tree verification on the same connection. + const retriedProof = firstProof || (await exit.connection.close()) + if (retriedProof) { + await input.onExitProven?.(input.sessionId, exit) + // Keep the first-hand exit evidence indexed until the tree proof succeeds; + // a failed close must be retryable and cannot look like an absent session. + input.exits.delete(input.sessionId) + return true + } + throw claudeAcquisitionCleanupError(exit.connection, exit.error) +} diff --git a/src/main/claude/claude-structured-auth-parity.test.ts b/src/main/claude/claude-structured-auth-parity.test.ts new file mode 100644 index 00000000000..ddc69366aad --- /dev/null +++ b/src/main/claude/claude-structured-auth-parity.test.ts @@ -0,0 +1,235 @@ +import { afterEach, describe, expect, it } from 'vitest' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import { beginClaudeAuthSwitch, endClaudeAuthSwitch } from '../claude-accounts/live-pty-gate' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE +} from '../claude-accounts/environment' +import type { AgentSessionRecordStore } from '../runtime/agent-session-record-store' +import { createClaudeStructuredLaunchResolver } from './claude-structured-launch-resolution' +import { ClaudeStructuredSessionAdapter } from './claude-structured-session-adapter' +import { + PROVIDER_SESSION_ID, + adapterFor, + fakeClaude, + identityFor +} from './claude-structured-session-test-support' + +const SESSION_ID = 'orca-session-auth' +const IDENTITY = { sessionId: SESSION_ID } as Parameters< + ReturnType +>[0]['identity'] + +function record(): AgentSessionRecord { + return { + sessionId: SESSION_ID, + provider: 'claude', + location: { + executionHostId: LOCAL_EXECUTION_HOST_ID, + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'folder' + }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/home/work/.claude' }, + providerHandleChain: [] + } as unknown as AgentSessionRecord +} + +function resolverFor(options: { + stripAuthEnv: boolean + overlay?: Record + authSwitchSettleTimeoutMs?: number +}): ReturnType { + return createClaudeStructuredLaunchResolver({ + store: { getRecord: () => record() } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: options.stripAuthEnv }), + authSwitchSettleTimeoutMs: options.authSwitchSettleTimeoutMs ?? 20, + ...(options.overlay ? { resolveEnv: () => options.overlay as Record } : {}) + }) +} + +/** + * An adapter driven by the REAL launch resolver, not the stub in the shared test + * support — the stub has no auth guard at all, so a teardown-window test built on it + * would pass whatever the guard did. + */ +function realResolverAdapter( + claude: ReturnType, + authSwitchSettleTimeoutMs: number +): ClaudeStructuredSessionAdapter { + const resumable = { + ...record(), + providerHandleChain: [ + { handle: { provider: 'claude', sessionId: PROVIDER_SESSION_ID, leafUuid: null } } + ] + } as unknown as AgentSessionRecord + return new ClaudeStructuredSessionAdapter({ + resolveLaunch: createClaudeStructuredLaunchResolver({ + store: { getRecord: () => resumable } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + authSwitchSettleTimeoutMs + }), + openConnection: claude.openConnection, + readProcessStartTime: async () => 1_700_000_000_000, + now: () => 1_700_000_000_500, + persistHandle: async () => {} + }) +} + +function withAmbientAuth(value: string, run: () => Promise): Promise { + const restore = process.env.ANTHROPIC_API_KEY + process.env.ANTHROPIC_API_KEY = value + return run().finally(() => { + if (restore === undefined) { + delete process.env.ANTHROPIC_API_KEY + } else { + process.env.ANTHROPIC_API_KEY = restore + } + }) +} + +describe('claude structured auth parity with the terminal preflight', () => { + afterEach(() => { + endClaudeAuthSwitch() + }) + + // Task 1 — the terminal preflight refuses this at spawn-env.ts:25 and + // runtime/spawn-preflight.ts:139; the structured path used to let the override win. + it('refuses an explicit Anthropic auth override while a managed account is pinned', async () => { + await expect( + resolverFor({ stripAuthEnv: true, overlay: { ANTHROPIC_API_KEY: 'sk-ant-CONFIGURED' } })({ + identity: IDENTITY + }) + ).rejects.toThrow(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + }) + + it('refuses an auth-like ANTHROPIC_CUSTOM_HEADERS override while a managed account is pinned', async () => { + await expect( + resolverFor({ + stripAuthEnv: true, + overlay: { ANTHROPIC_CUSTOM_HEADERS: 'Authorization: Bearer sk-ant-CONFIGURED' } + })({ identity: IDENTITY }) + ).rejects.toThrow(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + }) + + it('still admits a non-auth env overlay under a managed account', async () => { + const launch = await resolverFor({ + stripAuthEnv: true, + overlay: { ANTHROPIC_BASE_URL: 'https://gateway.example.test' } + })({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_BASE_URL).toBe('https://gateway.example.test') + }) + + // Task 2 — legacy computes stripAuthEnv at runtime-auth-preparation.ts:72, so a + // system-auth user's own shell key is their sign-in and must survive. + it('passes an ambient Anthropic key through when no managed account is active', async () => { + await withAmbientAuth('sk-ant-SHELL', async () => { + const launch = await resolverFor({ stripAuthEnv: false })({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBe('sk-ant-SHELL') + }) + }) + + it('lets an explicit overlay override the ambient key when no managed account is active', async () => { + await withAmbientAuth('sk-ant-SHELL', async () => { + const launch = await resolverFor({ + stripAuthEnv: false, + overlay: { ANTHROPIC_API_KEY: 'sk-ant-CONFIGURED' } + })({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBe('sk-ant-CONFIGURED') + }) + }) + + it('still strips the ambient Anthropic key when a managed account is pinned', async () => { + await withAmbientAuth('sk-ant-SHELL', async () => { + const launch = await resolverFor({ stripAuthEnv: true })({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBeUndefined() + }) + }) + + // Task 3 — the terminal preflight guards this at four sites; the structured path had none. + it('refuses launch resolution when an account switch never settles', async () => { + beginClaudeAuthSwitch() + + await expect( + resolverFor({ stripAuthEnv: true, authSwitchSettleTimeoutMs: 20 })({ identity: IDENTITY }) + ).rejects.toThrow(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) + }) + + it('waits a settling account switch out rather than refusing a resolved launch', async () => { + beginClaudeAuthSwitch() + setTimeout(() => endClaudeAuthSwitch(), 20) + + const launch = await resolverFor({ + stripAuthEnv: true, + authSwitchSettleTimeoutMs: 5_000 + })({ identity: IDENTITY }) + + expect(launch.claudeConfigDir).toBe('/home/work/.claude') + }) + + it('refuses an acquire before it tears the previous session down', async () => { + const claude = fakeClaude() + const adapter = adapterFor(claude) + beginClaudeAuthSwitch() + + await expect( + adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + ).rejects.toThrow(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) + // Nothing was spawned, so the refusal must not have opened a connection. + expect(claude.connections).toHaveLength(0) + }) + + // The teardown between the entry guard and launch resolution closes the live child + // and proves its tree — seconds, not milliseconds. A switch that begins inside it + // has already cost the user their session, so refusing there produces exactly the + // outcome the entry guard advertises against: a dead chat and no replacement. + it('replaces the session when a switch begins inside the acquire teardown', async () => { + const claude = fakeClaude() + const adapter = realResolverAdapter(claude, 5_000) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + const live = claude.connections[0]! + const closeWithSwitch = live.close + live.close = async () => { + beginClaudeAuthSwitch() + setTimeout(() => endClaudeAuthSwitch(), 20) + return closeWithSwitch() + } + + await expect( + adapter.acquire({ identity: identityFor(), fence: 8, spawnToken: 'spawn-10' }) + ).resolves.toMatchObject({ process: { spawnToken: 'spawn-10' } }) + expect(live.closed).toBe(true) + // The replacement child exists: the user's chat came back. + expect(claude.connections).toHaveLength(2) + expect(claude.connections[1]!.closed).toBe(false) + await adapter.closeAll() + }) + + it('still refuses a mid-teardown switch that never settles, leaving nothing half-open', async () => { + const claude = fakeClaude() + const adapter = realResolverAdapter(claude, 20) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + const live = claude.connections[0]! + const closeWithSwitch = live.close + live.close = async () => { + beginClaudeAuthSwitch() + return closeWithSwitch() + } + + await expect( + adapter.acquire({ identity: identityFor(), fence: 8, spawnToken: 'spawn-10' }) + ).rejects.toThrow(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) + // No replacement child was opened, so nothing is left running unowned. + expect(claude.connections).toHaveLength(1) + await adapter.closeAll() + }) +}) diff --git a/src/main/claude/claude-structured-content-parts.test.ts b/src/main/claude/claude-structured-content-parts.test.ts new file mode 100644 index 00000000000..d2142150937 --- /dev/null +++ b/src/main/claude/claude-structured-content-parts.test.ts @@ -0,0 +1,98 @@ +import { describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity +} from '../../shared/agent-session-journal-types' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' + +function sinkState() { + const items: { identity: AgentJournalItemIdentity; body: AgentJournalItemBody }[] = [] + const sink: StructuredAgentSessionEventSink = { + appendItem: (identity, body) => items.push({ identity, body }), + appendTombstone: () => {}, + publish: vi.fn() + } + return { sink, items } +} + +function providerRows(items: { body: AgentJournalItemBody }[]) { + return items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame + ? [{ kind: item.body.providerFrame.kind, text: item.body.text }] + : [] + ) +} + +function userMessageWith(part: unknown) { + return { + type: 'message' as const, + sessionId: 'orca-session', + startsTurn: true as const, + message: { + type: 'user', + uuid: 'user-1', + session_id: 'claude-session', + parent_tool_use_id: null, + isReplay: true, + message: { role: 'user', content: [{ type: 'text', text: 'look at this' }, part] } + } + } +} + +/** Exactly what claudeDispatchMessageContent sends for a local attachment. */ +const BASE64_IMAGE = { + type: 'image', + source: { type: 'base64', media_type: 'image/png', data: 'iVBORw0KGgoAAAANSUhEUg==' } +} + +describe('Claude message content parts', () => { + it('does not leak a wire kind for a locally attached image', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userMessageWith(BASE64_IMAGE)) + + expect(providerRows(state.items)).toEqual([]) + }) + + it('still renders an image the CLI sends by url', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + userMessageWith({ type: 'image', source: { type: 'url', url: 'https://x.test/a.png' } }) + ) + + expect(providerRows(state.items)).toEqual([]) + expect( + state.items.flatMap((item) => (item.body.kind === 'message' ? item.body.blocks : [])) + ).toContainEqual({ type: 'image-ref', url: 'https://x.test/a.png' }) + }) + + it('says what is true for a content part it cannot render, not the wire kind', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userMessageWith({ type: 'some_future_part', payload: { a: 1 } })) + + const rows = providerRows(state.items) + expect(rows).toHaveLength(1) + // The kind stays on the row for debugging, behind the disclosure. + expect(rows[0].kind).toBe('message:user:content:some_future_part') + // ...but the visible text is a sentence, not the opcode. + expect(rows[0].text).not.toContain('message:user:content') + expect(rows[0].text.toLowerCase()).toContain('claude') + }) + + it('prefers a readable sentence the part carries over the placeholder', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + userMessageWith({ type: 'some_future_part', message: 'the server refused the upload' }) + ) + + expect(providerRows(state.items)[0].text).toBe('the server refused the upload') + }) +}) diff --git a/src/main/claude/claude-structured-control-actions.test.ts b/src/main/claude/claude-structured-control-actions.test.ts new file mode 100644 index 00000000000..a90cb7908ba --- /dev/null +++ b/src/main/claude/claude-structured-control-actions.test.ts @@ -0,0 +1,161 @@ +import { describe, expect, it, vi } from 'vitest' +import { + cancelClaudeTurn, + answerClaudePrompt, + stopClaudeBackgroundTasks +} from './claude-structured-control-actions' +import { ClaudeControlRequestError } from './claude-stream-json-connection' +import { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import type { ClaudeSession } from './claude-structured-session-state' +import { ClaudeBackgroundTaskTracker } from './claude-background-task-tracker' + +type InterruptResult = Awaited> + +function sessionWith(input: { + capabilities?: string[] + interrupt: (options?: { cancelQueued?: boolean; timeoutMs?: number }) => Promise + cancelAsyncMessage?: (uuid: string) => Promise + prompts?: ClaudePromptRegistry +}): { + session: ClaudeSession + interrupt: ReturnType + cancelAsyncMessage: ReturnType +} { + const interrupt = vi.fn(input.interrupt) + const cancelAsyncMessage = vi.fn(input.cancelAsyncMessage ?? (async () => {})) + const session = { + capabilities: input.capabilities ?? [], + prompts: input.prompts ?? new ClaudePromptRegistry(), + connection: { interrupt, cancelAsyncMessage } + } as unknown as ClaudeSession + return { session, interrupt, cancelAsyncMessage } +} + +describe('cancelClaudeTurn', () => { + it('interrupts without a receipt on an older CLI and reports the turn cancelled', async () => { + const { session, interrupt, cancelAsyncMessage } = sessionWith({ + interrupt: async () => undefined + }) + + await expect(cancelClaudeTurn(session, 5_000)).resolves.toEqual({ cancelled: true }) + expect(interrupt).toHaveBeenCalledWith({ timeoutMs: 5_000 }) + expect(cancelAsyncMessage).not.toHaveBeenCalled() + }) + + it('withdraws every still-queued message a plain interrupt receipt reports', async () => { + const { session, interrupt, cancelAsyncMessage } = sessionWith({ + capabilities: ['interrupt_receipt_v1'], + interrupt: async () => ({ still_queued: ['queued-1', 'queued-2'] }) + }) + + await expect(cancelClaudeTurn(session, 5_000)).resolves.toEqual({ cancelled: true }) + // No cancel_queued capability, so the queue is swept one uuid at a time. + expect(interrupt).toHaveBeenCalledWith({ timeoutMs: 5_000 }) + expect(cancelAsyncMessage.mock.calls.map((call) => call[0])).toEqual(['queued-1', 'queued-2']) + }) + + it('sends cancel_queued and never sweeps when the CLI advertises the capability', async () => { + const { session, interrupt, cancelAsyncMessage } = sessionWith({ + capabilities: ['interrupt_receipt_v1', 'interrupt_cancel_queued_v1'], + interrupt: async () => ({ still_queued: [], cancelled: ['queued-1'] }) + }) + + await expect(cancelClaudeTurn(session, 5_000)).resolves.toEqual({ cancelled: true }) + expect(interrupt).toHaveBeenCalledWith({ cancelQueued: true, timeoutMs: 5_000 }) + expect(cancelAsyncMessage).not.toHaveBeenCalled() + }) + + it('reports a not-running interrupt as not cancelled without throwing', async () => { + const { session } = sessionWith({ + interrupt: async () => { + throw new ClaudeControlRequestError('interrupt', 'not running') + } + }) + + await expect(cancelClaudeTurn(session, 5_000)).resolves.toEqual({ cancelled: false }) + }) + + it('propagates a transport failure such as an interrupt timeout', async () => { + const { session } = sessionWith({ + interrupt: async () => { + throw new Error('claude interrupt request timed out') + } + }) + + await expect(cancelClaudeTurn(session, 5_000)).rejects.toThrow('timed out') + }) +}) + +describe('answerClaudePrompt', () => { + it('settles the pending prompt callback and forgets it', async () => { + const prompts = new ClaudePromptRegistry() + const settle = vi.fn() + const prompt = prompts.register({ + requestId: 'perm-1', + toolName: 'Bash', + toolUseId: 'tool-1', + input: { command: 'ls' }, + suggestions: [], + settle + })! + prompts.bindJournalItemId('journal-1', prompt.promptKey) + const { session } = sessionWith({ interrupt: async () => undefined, prompts }) + + await answerClaudePrompt(session, { itemId: 'journal-1', kind: 'approval', optionId: 'allow' }) + + expect(settle).toHaveBeenCalledWith( + expect.objectContaining({ behavior: 'allow', toolUseID: 'tool-1' }) + ) + expect(prompts.find('journal-1')).toBeNull() + }) + + it('refuses an answer for a prompt Claude is no longer waiting on', async () => { + const { session } = sessionWith({ interrupt: async () => undefined }) + await expect( + answerClaudePrompt(session, { itemId: 'missing', kind: 'approval', optionId: 'allow' }) + ).rejects.toThrow(/no longer waiting/) + }) +}) + +describe('stopClaudeBackgroundTasks', () => { + it('stops each live SDK task id and never depends on an active turn id', async () => { + const backgroundTasks = new ClaudeBackgroundTaskTracker() + backgroundTasks.observe({ + type: 'system', + subtype: 'background_tasks_changed', + tasks: [ + { task_id: 'task-agent', task_type: 'local_agent', description: 'agent' }, + { task_id: 'task-bash', task_type: 'local_bash', description: 'bash' } + ] + }) + const stopTask = vi.fn(async (_taskId: string, _options?: { timeoutMs?: number }) => {}) + const session = { backgroundTasks, connection: { stopTask } } as unknown as ClaudeSession + + await expect(stopClaudeBackgroundTasks(session, 5_000)).resolves.toEqual({ cancelled: true }) + expect(stopTask.mock.calls).toEqual([ + ['task-agent', { timeoutMs: 5_000 }], + ['task-bash', { timeoutMs: 5_000 }] + ]) + }) + + it('stops issuing requests when ownership changes between tasks', async () => { + const backgroundTasks = new ClaudeBackgroundTaskTracker() + for (const taskId of ['task-1', 'task-2']) { + backgroundTasks.observe({ + type: 'system', + subtype: 'task_started', + task_id: taskId, + task_type: 'local_agent', + is_backgrounded: true + }) + } + let current = true + const stopTask = vi.fn(async (_taskId: string) => { + current = false + }) + const session = { backgroundTasks, connection: { stopTask } } as unknown as ClaudeSession + + await stopClaudeBackgroundTasks(session, undefined, () => current) + expect(stopTask).toHaveBeenCalledTimes(1) + }) +}) diff --git a/src/main/claude/claude-structured-control-actions.ts b/src/main/claude/claude-structured-control-actions.ts new file mode 100644 index 00000000000..d216304c311 --- /dev/null +++ b/src/main/claude/claude-structured-control-actions.ts @@ -0,0 +1,83 @@ +import { applyClaudePromptAnswer } from './claude-structured-prompt-replies' +import { ClaudeControlRequestError } from './claude-stream-json-connection' +import type { ClaudeSession } from './claude-structured-session-state' + +const INTERRUPT_CANCEL_QUEUED_CAPABILITY = 'interrupt_cancel_queued_v1' + +export type ClaudeTurnCancellationGuard = () => boolean + +/** + * Interrupt the running turn, then make sure no queued async user message survives to spawn a + * later unexpected turn. On a CLI advertising `interrupt_cancel_queued_v1` one round trip + * cancels the queue alongside the abort; otherwise the interrupt receipt lists `still_queued` + * uuids, and each is withdrawn best-effort with `cancel_async_message`. Older CLIs resolve no + * receipt, so there is nothing to sweep. + */ +export async function cancelClaudeTurn( + session: ClaudeSession, + timeoutMs: number | undefined, + isCurrent: ClaudeTurnCancellationGuard = () => true +): Promise<{ cancelled: boolean }> { + // The SDK interrupt is session-scoped. Re-check the caller's turn/fence + // immediately before issuing it so a delayed request cannot stop a later turn. + if (!isCurrent()) { + return { cancelled: false } + } + const cancelQueued = session.capabilities.includes(INTERRUPT_CANCEL_QUEUED_CAPABILITY) + try { + const receipt = await session.connection.interrupt({ + ...(cancelQueued ? { cancelQueued: true } : {}), + timeoutMs + }) + if (!cancelQueued) { + for (const uuid of receipt?.still_queued ?? []) { + await session.connection.cancelAsyncMessage(uuid, { timeoutMs }).catch(() => {}) + } + } + return { cancelled: true } + } catch (error) { + if (error instanceof ClaudeControlRequestError) { + return { cancelled: false } + } + throw error + } +} + +export async function stopClaudeBackgroundTasks( + session: ClaudeSession, + timeoutMs: number | undefined, + isCurrent: ClaudeTurnCancellationGuard = () => true +): Promise<{ cancelled: boolean }> { + const taskIds = session.backgroundTasks.stoppableTaskIds + let cancelled = false + for (const taskId of taskIds) { + if (!isCurrent()) { + break + } + try { + await session.connection.stopTask(taskId, { timeoutMs }) + cancelled = true + } catch (error) { + if (!(error instanceof ClaudeControlRequestError)) { + throw error + } + } + } + return { cancelled } +} + +export async function answerClaudePrompt( + session: ClaudeSession, + input: { itemId: string; kind: 'approval' | 'question'; optionId: string } +): Promise { + const found = session.prompts.find(input.itemId) + if (!found || found.prompt.kind !== input.kind) { + throw new Error(`claude is no longer waiting on ${input.itemId}`) + } + const response = applyClaudePromptAnswer(found, input.optionId) + if (response === null) { + return + } + session.prompts.forget(found.prompt) + found.prompt.settle(response) +} diff --git a/src/main/claude/claude-structured-dispatch-content.ts b/src/main/claude/claude-structured-dispatch-content.ts new file mode 100644 index 00000000000..71f180bc3ac --- /dev/null +++ b/src/main/claude/claude-structured-dispatch-content.ts @@ -0,0 +1,165 @@ +import { createHash } from 'node:crypto' +import { open } from 'node:fs/promises' +import { extname } from 'node:path' +import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types' +import type { NativeChatBlock } from '../../shared/native-chat-types' + +const MAX_IMAGE_BYTES = 5 * 1024 * 1024 +const MAX_IMAGE_COUNT = 20 +const MAX_TOTAL_IMAGE_BYTES = 20 * 1024 * 1024 +const MAX_REPLAY_CONTENT_KEY_BYTES = 256 + +type ImageBudget = { + count: number + localBytes: number +} + +export async function readClaudeImage(path: string, openImpl: typeof open = open): Promise { + const file = await openImpl(path, 'r') + try { + const invalidImage = (): Error => + new Error(`Claude image must be a non-empty file no larger than ${MAX_IMAGE_BYTES} bytes`) + const info = await file.stat() + if (!info.isFile()) { + throw new Error('Claude image must be a file') + } + if (info.size > MAX_IMAGE_BYTES) { + throw invalidImage() + } + const buffer = Buffer.allocUnsafe(info.size + 1) + let bytesRead = 0 + while (bytesRead < buffer.length) { + const result = await file.read(buffer, bytesRead, buffer.length - bytesRead, bytesRead) + if (result.bytesRead === 0) { + break + } + bytesRead += result.bytesRead + } + // A file can grow after the initial stat and after the final read returns + // zero. Prove the descriptor's size matches what was copied before sending. + const finalInfo = await file.stat() + if (bytesRead === 0 || bytesRead > MAX_IMAGE_BYTES || finalInfo.size !== bytesRead) { + throw invalidImage() + } + return buffer.subarray(0, bytesRead) + } finally { + await file.close() + } +} + +const IMAGE_MIME_BY_EXTENSION: Record = { + '.gif': 'image/gif', + '.jpeg': 'image/jpeg', + '.jpg': 'image/jpeg', + '.png': 'image/png', + '.webp': 'image/webp' +} + +async function imageContent( + block: Extract, + budget: ImageBudget +): Promise { + budget.count += 1 + if (budget.count > MAX_IMAGE_COUNT) { + throw new Error(`Claude messages support at most ${MAX_IMAGE_COUNT} images`) + } + if (block.url) { + return { type: 'image', source: { type: 'url', url: block.url } } + } + if (!block.path) { + throw new Error('image reference has neither a path nor a URL') + } + const data = await readClaudeImage(block.path) + budget.localBytes += data.byteLength + if (budget.localBytes > MAX_TOTAL_IMAGE_BYTES) { + throw new Error(`Claude images must total no more than ${MAX_TOTAL_IMAGE_BYTES} bytes`) + } + const mediaType = IMAGE_MIME_BY_EXTENSION[extname(block.path).toLowerCase()] + if (!mediaType) { + throw new Error(`Claude does not support the image type ${extname(block.path)}`) + } + return { + type: 'image', + source: { + type: 'base64', + media_type: mediaType, + data: data.toString('base64') + } + } +} + +export async function claudeDispatchMessageContent( + body: AgentJournalMessageItem +): Promise { + if (body.role !== 'user') { + throw new Error('Claude dispatch accepts only user messages') + } + const content: unknown[] = [] + const imageBudget: ImageBudget = { count: 0, localBytes: 0 } + for (const block of body.blocks as NativeChatBlock[]) { + if (block.type === 'text' && block.text.length > 0) { + content.push({ type: 'text', text: block.text }) + } else if (block.type === 'image-ref') { + content.push(await imageContent(block, imageBudget)) + } + } + if (content.length === 0) { + throw new Error('Claude dispatch requires text or an image') + } + return content +} + +/** + * Keep waiter metadata bounded even when a dispatch contains large base64 images. + * The digest is only diagnostic: replay acknowledgement must use provider identity. + */ +export function claudeDispatchContentKey(content: readonly unknown[]): string { + const digest = createHash('sha256') + const summary = content + .map((part) => { + const record = + typeof part === 'object' && part !== null && !Array.isArray(part) + ? (part as Record) + : null + const type = typeof record?.type === 'string' ? record.type : 'unknown' + if (type === 'text') { + return `text:${typeof record?.text === 'string' ? record.text.length : 0}` + } + const source = + typeof record?.source === 'object' && record.source !== null + ? (record.source as Record) + : null + if (type === 'image' && source?.type === 'base64') { + return `image:${typeof source.media_type === 'string' ? source.media_type : ''}:${typeof source.data === 'string' ? source.data.length : 0}` + } + return type + }) + .join(',') + for (const [index, part] of content.entries()) { + const record = + typeof part === 'object' && part !== null && !Array.isArray(part) + ? (part as Record) + : null + const type = typeof record?.type === 'string' ? record.type : 'unknown' + digest.update(`${index}:${type}:`) + if (type === 'text' && typeof record?.text === 'string') { + digest.update(record.text) + continue + } + const source = + typeof record?.source === 'object' && record.source !== null + ? (record.source as Record) + : null + if (type === 'image' && source?.type === 'base64') { + digest.update(typeof source.media_type === 'string' ? source.media_type : '') + digest.update(':') + if (typeof source.data === 'string') { + digest.update(source.data) + } + continue + } + digest.update(JSON.stringify(part)) + } + const key = `v1:${summary.slice(0, 128)}:${digest.digest('hex')}` + return key.slice(0, MAX_REPLAY_CONTENT_KEY_BYTES) +} diff --git a/src/main/claude/claude-structured-dispatch.test.ts b/src/main/claude/claude-structured-dispatch.test.ts new file mode 100644 index 00000000000..d66a64f82eb --- /dev/null +++ b/src/main/claude/claude-structured-dispatch.test.ts @@ -0,0 +1,600 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it, vi } from 'vitest' +import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types' +import { dispatchClaudeTurn, resolveClaudeReplayWaiter } from './claude-structured-dispatch' +import { readClaudeImage } from './claude-structured-dispatch-content' +import type { ClaudeSession } from './claude-structured-session-state' +import { ClaudeBackgroundTaskTracker } from './claude-background-task-tracker' + +function sessionFor(send = vi.fn().mockResolvedValue(undefined)): ClaudeSession { + return { + connection: { send } as unknown as ClaudeSession['connection'], + providerSessionId: 'provider-session', + claudeConfigDir: '/accounts/claude', + leafUuid: null, + fence: 1, + acquisitionGeneration: 'generation-1', + prompts: {} as ClaudeSession['prompts'], + dispatchWaiters: [], + retiredDispatchWaiters: [], + replayContentFallbackBlocked: false, + backgroundTasks: new ClaudeBackgroundTaskTracker(), + dispatchSequence: 0, + optionMutationSequence: 0, + options: new Map(), + reportedOptions: {}, + reportedModelMutation: 0, + confirmedOptions: new Set(), + restoreSkippedOptions: new Set(), + capabilities: [], + events: undefined, + translator: null + } +} + +function userMessage(blocks: AgentJournalMessageItem['blocks']): AgentJournalMessageItem { + return { kind: 'message', role: 'user', blocks } +} + +function userReplayFrame(uuid: string, text: string): Record { + return { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid, + message: { role: 'user', content: [{ type: 'text', text }] } + } +} + +describe('Claude structured dispatch image limits', () => { + it('recovers the active identity when a timed-out replay arrives late', async () => { + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const sentUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid + await expect(dispatched).resolves.toMatchObject({ state: 'unknown' }) + + expect(resolveClaudeReplayWaiter(session, userReplayFrame(sentUuid!, 'one'))).toBe(true) + expect(session.activeTurnId).toBe(sentUuid) + expect(session.activeTurnSequence).toBe(session.dispatchSequence) + }) + + it('never lets a late replay for dispatch A resolve dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const firstUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'two' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid + + expect(resolveClaudeReplayWaiter(session, userReplayFrame(firstUuid!, 'one'))).toBe(false) + expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) + expect(resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid!, 'two'))).toBe(true) + await expect(second).resolves.toMatchObject({ providerIdentity: { uuid: secondUuid } }) + }) + + it('does not let an identical late replay for dispatch A resolve active dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = session.dispatchWaiters[0]!.sentUuid + + expect(resolveClaudeReplayWaiter(session, userReplayFrame('provider-a', 'same prompt'))).toBe( + false + ) + expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) + + resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid, 'same prompt')) + await expect(second).resolves.toMatchObject({ providerIdentity: { uuid: secondUuid } }) + }) + + it('does not let a fresh-UUID replay for an evicted dispatch resolve active dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + const firstUuid = session.retiredDispatchWaiters[0]!.sentUuid + + const fillerDispatches = await Promise.all( + Array.from({ length: 64 }, (_, index) => + dispatchClaudeTurn( + session, + { + clientMessageId: `filler-${index}`, + body: userMessage([{ type: 'text', text: 'same prompt' }]) + }, + 5 + ) + ) + ) + expect(fillerDispatches.every((outcome) => outcome.state === 'unknown')).toBe(true) + expect(session.retiredDispatchWaiters).toHaveLength(64) + expect(session.replayContentFallbackBlocked).toBe(true) + expect(session.retiredDispatchWaiters.some((waiter) => waiter.sentUuid === firstUuid)).toBe( + false + ) + + while (session.retiredDispatchWaiters.length > 0) { + const sentUuid = session.retiredDispatchWaiters[0]!.sentUuid + resolveClaudeReplayWaiter(session, userReplayFrame(sentUuid, 'same prompt')) + } + expect(session.retiredDispatchWaiters).toHaveLength(0) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = session.dispatchWaiters[0]!.sentUuid + + expect( + resolveClaudeReplayWaiter(session, userReplayFrame('provider-a-late', 'same prompt')) + ).toBe(false) + expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) + + resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid, 'same prompt')) + await expect(second).resolves.toMatchObject({ providerIdentity: { uuid: secondUuid } }) + }) + + it('does not let a fresh-UUID result for an evicted slash dispatch resolve active dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + const firstUuid = session.retiredDispatchWaiters[0]!.sentUuid + + const fillerDispatches = await Promise.all( + Array.from({ length: 64 }, (_, index) => + dispatchClaudeTurn( + session, + { + clientMessageId: `filler-${index}`, + body: userMessage([{ type: 'text', text: '/permissions' }]) + }, + 5 + ) + ) + ) + expect(fillerDispatches.every((outcome) => outcome.state === 'unknown')).toBe(true) + expect(session.retiredDispatchWaiters).toHaveLength(64) + expect(session.replayContentFallbackBlocked).toBe(true) + expect(session.retiredDispatchWaiters.some((waiter) => waiter.sentUuid === firstUuid)).toBe( + false + ) + + while (session.retiredDispatchWaiters.length > 0) { + const sentUuid = session.retiredDispatchWaiters[0]!.sentUuid + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: `result-${sentUuid}`, + user_message_uuid: sentUuid + }) + ).toBe(false) + } + expect(session.retiredDispatchWaiters).toHaveLength(0) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = session.dispatchWaiters[0]!.sentUuid + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'result-a-late' + }) + ).toBe(false) + expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'result-b', + user_message_uuid: secondUuid + }) + ).toBe(false) + await expect(second).resolves.toMatchObject({ + providerIdentity: { uuid: 'result-b' } + }) + }) + + it('does not let a legacy result for timed-out ordinary dispatch A resolve slash dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'ordinary' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'legacy-result-a' + }) + ).toBe(false) + await expect(second).resolves.toMatchObject({ state: 'unknown' }) + }) + + it('removes only its own waiter when a later send fails', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const firstWaiter = session.dispatchWaiters[0] + session.connection.send = vi.fn().mockRejectedValue(new Error('broken pipe')) + + await expect( + dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'two' }]) }, + 100 + ) + ).resolves.toMatchObject({ state: 'unknown', reason: 'broken pipe' }) + expect(session.dispatchWaiters).toEqual([firstWaiter]) + + const firstUuid = (firstWaiter as { sentUuid?: string }).sentUuid + resolveClaudeReplayWaiter(session, userReplayFrame(firstUuid!, 'one')) + await expect(first).resolves.toMatchObject({ providerIdentity: { uuid: firstUuid } }) + }) + + it('keeps a replay accepted before its send reports failure', async () => { + let session!: ClaudeSession + const send = vi.fn(async (message: Record) => { + resolveClaudeReplayWaiter(session, { ...message, uuid: 'turn-race' }) + throw new Error('write raced provider acknowledgement') + }) + session = sessionFor(send) + + await expect( + dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, + 100 + ) + ).resolves.toMatchObject({ state: 'accepted', providerIdentity: { uuid: 'turn-race' } }) + expect(session.dispatchWaiters).toHaveLength(0) + }) + + it('accepts a slash command from its result receipt when Claude omits the user replay', async () => { + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'command-result-uuid' + }) + ).toBe(false) + + await expect(dispatched).resolves.toEqual({ + state: 'accepted', + providerIdentity: { + provider: 'claude', + sessionId: 'provider-session', + uuid: 'command-result-uuid' + } + }) + }) + + it('correlates a later slash-command result by user_message_uuid despite a timed-out slash waiter', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = session.dispatchWaiters[0]!.sentUuid + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'result-b', + user_message_uuid: secondUuid + }) + ).toBe(false) + await expect(second).resolves.toMatchObject({ + state: 'accepted', + providerIdentity: { uuid: 'result-b' } + }) + }) + + it('does not mistake a normal turn result for its missing user replay', async () => { + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'hello' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + session_id: 'provider-session', + uuid: 'unrelated-result-uuid' + }) + ).toBe(false) + expect(session.dispatchWaiters).toHaveLength(1) + expect( + resolveClaudeReplayWaiter(session, { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid: 'user-replay-uuid', + message: { + role: 'user', + content: [{ type: 'text', text: 'hello' }] + } + }) + ).toBe(true) + + await expect(dispatched).resolves.toMatchObject({ + state: 'accepted', + providerIdentity: { uuid: 'user-replay-uuid' } + }) + }) + + it('ignores a top-level tool-result user frame while waiting for a slash command replay', async () => { + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + + resolveClaudeReplayWaiter(session, { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid: 'tool-result-uuid', + message: { + role: 'user', + content: [{ type: 'tool_result', tool_use_id: 'tool-1', content: 'done' }] + } + }) + expect(session.dispatchWaiters).toHaveLength(1) + + resolveClaudeReplayWaiter(session, { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid: 'user-replay-uuid', + message: { + role: 'user', + content: [{ type: 'text', text: '/permissions' }] + } + }) + + await expect(dispatched).resolves.toEqual({ + state: 'accepted', + providerIdentity: { + provider: 'claude', + sessionId: 'provider-session', + uuid: 'user-replay-uuid' + } + }) + }) + + it('rejects more than twenty URL images before sending', async () => { + const session = sessionFor() + const body = userMessage( + Array.from({ length: 21 }, (_, index) => ({ + type: 'image-ref' as const, + url: `https://example.test/${index}.png` + })) + ) + + await expect( + dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }, 1) + ).resolves.toEqual({ state: 'rejected', reason: 'Claude messages support at most 20 images' }) + expect(session.connection.send).not.toHaveBeenCalled() + }) + + it('rejects local images whose aggregate size exceeds twenty MiB', async () => { + const directory = await mkdtemp(join(tmpdir(), 'orca-claude-images-')) + try { + const paths = await Promise.all( + Array.from({ length: 5 }, async (_, index) => { + const path = join(directory, `${index}.png`) + await writeFile(path, Buffer.alloc(5 * 1024 * 1024)) + return path + }) + ) + const session = sessionFor() + const body = userMessage(paths.map((path) => ({ type: 'image-ref' as const, path }))) + + await expect( + dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }, 1) + ).resolves.toEqual({ + state: 'rejected', + reason: `Claude images must total no more than ${20 * 1024 * 1024} bytes` + }) + expect(session.connection.send).not.toHaveBeenCalled() + } finally { + await rm(directory, { recursive: true, force: true }) + } + }) + + it('rejects a local image by actual bytes read beyond the per-image cap', async () => { + const directory = await mkdtemp(join(tmpdir(), 'orca-claude-image-')) + try { + const path = join(directory, 'oversized.png') + await writeFile(path, Buffer.alloc(5 * 1024 * 1024 + 1)) + const session = sessionFor() + const body = userMessage([{ type: 'image-ref', path }]) + + await expect( + dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }, 1) + ).resolves.toEqual({ + state: 'rejected', + reason: `Claude image must be a non-empty file no larger than ${5 * 1024 * 1024} bytes` + }) + expect(session.connection.send).not.toHaveBeenCalled() + } finally { + await rm(directory, { recursive: true, force: true }) + } + }) + + it('allocates local image reads from the file size, not the maximum cap', async () => { + const directory = await mkdtemp(join(tmpdir(), 'orca-claude-image-')) + const allocUnsafe = vi.spyOn(Buffer, 'allocUnsafe') + try { + const path = join(directory, 'small.png') + await writeFile(path, Buffer.alloc(64)) + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'image-ref', path }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const sentUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid + resolveClaudeReplayWaiter(session, { + ...userReplayFrame(sentUuid!, ''), + message: { + role: 'user', + content: [ + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: '' } } + ] + } + }) + await expect(dispatched).resolves.toMatchObject({ state: 'accepted' }) + expect(allocUnsafe).toHaveBeenCalled() + expect(allocUnsafe.mock.calls.some(([size]) => size === 64 + 1)).toBe(true) + expect(allocUnsafe.mock.calls.some(([size]) => size >= 5 * 1024 * 1024)).toBe(false) + } finally { + allocUnsafe.mockRestore() + await rm(directory, { recursive: true, force: true }) + } + }) + + it('bounds retained waiter identity bytes when image dispatches time out', async () => { + const directory = await mkdtemp(join(tmpdir(), 'orca-claude-image-')) + try { + const path = join(directory, 'large.png') + await writeFile(path, Buffer.alloc(64 * 1024)) + const session = sessionFor() + const body = userMessage([{ type: 'image-ref', path }]) + await Promise.all( + Array.from({ length: 64 }, (_, index) => + dispatchClaudeTurn(session, { clientMessageId: `client-${index}`, body }, 1) + ) + ) + + expect(session.retiredDispatchWaiters).toHaveLength(64) + const retainedKeyBytes = session.retiredDispatchWaiters.reduce( + (total, waiter) => total + waiter.replayContentKey.length, + 0 + ) + expect(retainedKeyBytes).toBeLessThan(64 * 512) + expect( + session.retiredDispatchWaiters.every((waiter) => waiter.replayContentKey.length < 512) + ).toBe(true) + } finally { + await rm(directory, { recursive: true, force: true }) + } + }) + + it('rejects a local image when it grows after the initial stat', async () => { + const stat = vi + .fn() + .mockResolvedValueOnce({ isFile: () => true, size: 64 }) + .mockResolvedValueOnce({ isFile: () => true, size: 128 }) + const read = vi.fn(async (buffer: Buffer, offset: number) => { + if (read.mock.calls.length === 1) { + buffer.fill(1, offset, offset + 64) + return { bytesRead: 64, buffer } + } + return { bytesRead: 0, buffer } + }) + const open = vi.fn().mockResolvedValue({ + stat, + read, + close: vi.fn().mockResolvedValue(undefined) + } as never) + await expect(readClaudeImage('/controlled/growing.png', open)).rejects.toThrow( + `Claude image must be a non-empty file no larger than ${5 * 1024 * 1024} bytes` + ) + }) +}) diff --git a/src/main/claude/claude-structured-dispatch.ts b/src/main/claude/claude-structured-dispatch.ts new file mode 100644 index 00000000000..96271e41d71 --- /dev/null +++ b/src/main/claude/claude-structured-dispatch.ts @@ -0,0 +1,264 @@ +import { randomUUID } from 'node:crypto' +import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types' +import type { AgentSessionDispatchOutcome } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { + claudeHasReplayContent, + readClaudeMessageEnvelope +} from './claude-structured-item-translation' +import type { ClaudeDispatchWaiter, ClaudeSession } from './claude-structured-session-state' +import { readClaudeFrameString } from './claude-structured-init-proof' +import { + claudeDispatchContentKey, + claudeDispatchMessageContent +} from './claude-structured-dispatch-content' + +const MAX_RETIRED_DISPATCH_WAITERS = 64 + +export function resolveClaudeReplayWaiter( + session: ClaudeSession, + message: Record +): boolean { + const envelope = readClaudeMessageEnvelope(message) + const isUserReplay = + envelope?.role === 'user' && + message.parent_tool_use_id === null && + claudeHasReplayContent(envelope) + const isCompletedCommand = message.type === 'result' + if ( + (!isUserReplay && !isCompletedCommand) || + readClaudeFrameString(message, 'session_id') !== session.providerSessionId + ) { + return false + } + const uuid = readClaudeFrameString(message, 'uuid') + if (!uuid) { + return false + } + + // Newer SDK frames carry the client uuid that caused a turn. A correlation + // value is authoritative: never fall back to queue order or content, since + // identical prompts may be in flight across a timeout boundary. + const userMessageUuid = readClaudeFrameString(message, 'user_message_uuid') + if (userMessageUuid) { + const exact = session.dispatchWaiters.find( + (candidate) => candidate.sentUuid === userMessageUuid + ) + if (exact) { + settleWaiter(session, exact, uuid) + return isUserReplay && exact.dispatchSequence === session.dispatchSequence + } + const retired = session.retiredDispatchWaiters.find( + (candidate) => candidate.sentUuid === userMessageUuid + ) + if (retired) { + forgetRetiredWaiter(session, retired) + return recoverLateIdentity(session, retired, uuid, isUserReplay) + } + return false + } + + const exact = session.dispatchWaiters.find((candidate) => candidate.sentUuid === uuid) + if (exact) { + settleWaiter(session, exact, uuid) + return isUserReplay && exact.dispatchSequence === session.dispatchSequence + } + const retired = session.retiredDispatchWaiters.find((candidate) => candidate.sentUuid === uuid) + if (retired) { + forgetRetiredWaiter(session, retired) + return recoverLateIdentity(session, retired, uuid, isUserReplay) + } + + if (isUserReplay) { + // Compatibility CLIs may mint a new replay uuid instead of echoing the + // client uuid. Content is an acceptable join only when it is the sole + // candidate on one side of the timeout boundary; with active and retired + // candidates present, identical prompts are intentionally left unknown. + const replayContentKey = claudeDispatchContentKey(envelope.content) + if (!session.replayContentFallbackBlocked && session.retiredDispatchWaiters.length === 0) { + const compatible = session.dispatchWaiters.filter( + (candidate) => candidate.replayContentKey === replayContentKey + ) + if (compatible.length === 1) { + settleWaiter(session, compatible[0]!, uuid) + return compatible[0]!.dispatchSequence === session.dispatchSequence + } + } else if (!session.replayContentFallbackBlocked && session.dispatchWaiters.length === 0) { + const lateCompatible = session.retiredDispatchWaiters.filter( + (candidate) => candidate.replayContentKey === replayContentKey + ) + if (lateCompatible.length === 1) { + const [candidate] = lateCompatible + forgetRetiredWaiter(session, candidate!) + return recoverLateIdentity(session, candidate!, uuid, true) + } + } + return false + } + const current = session.dispatchWaiters[0] + if (isCompletedCommand && !current?.acceptsResult) { + return false + } + // A legacy result has no dispatch correlation. Any retired waiter makes queue order ambiguous, + // even when the retired dispatch was an ordinary turn rather than a slash command. + if (isCompletedCommand && session.retiredDispatchWaiters.length > 0) { + return false + } + // Once an eviction occurred, a fresh result uuid cannot be joined to a waiter by queue order. + if (isCompletedCommand && session.replayContentFallbackBlocked) { + return false + } + const waiter = uuid ? session.dispatchWaiters.shift() : undefined + if (waiter && uuid) { + clearTimeout(waiter.timer) + waiter.settledUuid = uuid + waiter.resolve(uuid) + return isUserReplay + } + return false +} + +function settleWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter, uuid: string): void { + const index = session.dispatchWaiters.indexOf(waiter) + if (index !== -1) { + session.dispatchWaiters.splice(index, 1) + } + clearTimeout(waiter.timer) + waiter.settledUuid = uuid + waiter.resolve(uuid) +} + +function forgetRetiredWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter): void { + const index = session.retiredDispatchWaiters.indexOf(waiter) + if (index !== -1) { + session.retiredDispatchWaiters.splice(index, 1) + } +} + +function recoverLateIdentity( + session: ClaudeSession, + waiter: ClaudeDispatchWaiter, + uuid: string, + isUserReplay: boolean +): boolean { + if (!isUserReplay && !waiter.acceptsResult) { + return false + } + if (waiter.dispatchSequence === session.dispatchSequence) { + session.activeTurnId = uuid + session.activeTurnSequence = waiter.dispatchSequence + } + return isUserReplay && waiter.dispatchSequence === session.dispatchSequence +} + +function waitForReplay( + session: ClaudeSession, + timeoutMs: number, + acceptsResult: boolean, + sentUuid: string, + replayContentKey: string +): { waiter: ClaudeDispatchWaiter; promise: Promise } { + let waiter!: ClaudeDispatchWaiter + const promise = new Promise((resolve) => { + waiter = { + acceptsResult, + sentUuid, + dispatchSequence: session.dispatchSequence, + replayContentKey, + resolve, + timer: setTimeout(() => { + const index = session.dispatchWaiters.indexOf(waiter) + if (index !== -1) { + session.dispatchWaiters.splice(index, 1) + } + retireWaiter(session, waiter) + resolve(null) + }, timeoutMs) + } + waiter.timer.unref?.() + session.dispatchWaiters.push(waiter) + }) + return { waiter, promise } +} + +function retireWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter): void { + const index = session.dispatchWaiters.indexOf(waiter) + if (index !== -1) { + session.dispatchWaiters.splice(index, 1) + } + clearTimeout(waiter.timer) + if (!waiter.retired) { + waiter.retired = true + session.retiredDispatchWaiters.push(waiter) + if (session.retiredDispatchWaiters.length > MAX_RETIRED_DISPATCH_WAITERS) { + session.replayContentFallbackBlocked = true + session.retiredDispatchWaiters.splice( + 0, + session.retiredDispatchWaiters.length - MAX_RETIRED_DISPATCH_WAITERS + ) + } + } +} + +export async function dispatchClaudeTurn( + session: ClaudeSession, + input: { clientMessageId: string; body: AgentJournalMessageItem }, + timeoutMs: number +): Promise { + let content: unknown[] + try { + content = await claudeDispatchMessageContent(input.body) + } catch (error) { + return { state: 'rejected', reason: (error as Error).message } + } + const dispatchSequence = ++session.dispatchSequence + const acceptsResult = input.body.blocks.some( + (block) => block.type === 'text' && block.text.trimStart().startsWith('/') + ) + const sentUuid = randomUUID() + const replay = waitForReplay( + session, + timeoutMs, + acceptsResult, + sentUuid, + claudeDispatchContentKey(content) + ) + const replayed = replay.promise + try { + await session.connection.send({ + type: 'user', + uuid: sentUuid, + message: { role: 'user', content }, + parent_tool_use_id: null, + session_id: session.providerSessionId + }) + } catch (error) { + const waiter = replay.waiter + if (waiter.settledUuid) { + const uuid = await replayed + if (uuid) { + session.activeTurnId = uuid + session.activeTurnSequence = dispatchSequence + return { + state: 'accepted', + providerIdentity: { provider: 'claude', sessionId: session.providerSessionId, uuid } + } + } + } + if (!waiter.retired) { + retireWaiter(session, waiter) + waiter.resolve(null) + } + return { state: 'unknown', reason: (error as Error).message } + } + const uuid = await replayed + if (uuid) { + session.activeTurnId = uuid + session.activeTurnSequence = dispatchSequence + } + return uuid + ? { + state: 'accepted', + providerIdentity: { provider: 'claude', sessionId: session.providerSessionId, uuid } + } + : { state: 'unknown', reason: 'claude accepted a message but did not replay its uuid in time' } +} diff --git a/src/main/claude/claude-structured-effort-reporting.test.ts b/src/main/claude/claude-structured-effort-reporting.test.ts new file mode 100644 index 00000000000..be022d86956 --- /dev/null +++ b/src/main/claude/claude-structured-effort-reporting.test.ts @@ -0,0 +1,257 @@ +import { describe, expect, it } from 'vitest' +import { AgentSessionOptionRejectedError } from '../native-chat/agent-session-wire/structured-agent-session-option-error' +import { + restoreClaudeStructuredSessionOptions, + setClaudeStructuredOption +} from './claude-structured-options' +import { readClaudeSettingsEffort } from './claude-structured-session-options' +import type { ClaudeSession } from './claude-structured-session-state' +import type { ClaudeStructuredSessionEvent } from './claude-structured-session-adapter' +import { acquired, fakeClaude } from './claude-structured-session-test-support' + +/** Verbatim from Claude Code 2.1.258's get_settings response. */ +const REAL_SETTINGS = { + applied: { model: 'claude-opus-5[1m]', effort: 'high', advisor: null, ultracode: false }, + effective: { model: 'claude-opus-5[1m]', effortLevel: 'high', env: {} }, + sources: {} +} + +function sessionWith( + reported: string | null, + calls: string[] = [], + listed?: { model: string; catalog: readonly Record[] } +) { + return { + session: { + options: new Map(listed ? [['model', listed.model]] : []), + reportedOptions: {} as { model?: string; effort?: string }, + optionMutationSequence: 0, + reportedModelMutation: 0, + confirmedOptions: new Set(), + restoreSkippedOptions: new Set(), + connection: { + supportedModels: async () => { + calls.push('list_models') + return [...(listed?.catalog ?? [])] + }, + setModel: async (model: string) => { + calls.push(`set_model:${model}`) + }, + applyFlagSettings: async (settings: { effortLevel?: string }) => { + // The measured behaviour: an unknown effort is accepted and ignored. + calls.push(`apply:${settings.effortLevel}`) + }, + getSettings: async () => { + calls.push('get_settings') + return reported === null + ? { applied: {}, effective: {}, sources: {} } + : { applied: { effort: reported }, effective: { effortLevel: reported }, sources: {} } + } + } + } as unknown as ClaudeSession, + calls + } +} + +describe('Claude effort reporting', () => { + it('reads the effort get_settings reports', () => { + expect(readClaudeSettingsEffort(REAL_SETTINGS)).toBe('high') + }) + + it.each([ + [ + 'the provider stops reporting it', + { applied: { effort: 'high' }, effective: {}, sources: {} } + ], + ['the payload carries no effective block', { applied: { effort: 'high' } }], + ['the request failed outright', null] + ])('reports no effort when %s', (_case, settings) => { + // Never defaulted: an effort nothing measured would be worse than a blank + // pill, and this is the assertion that goes red if the key is renamed. + expect(readClaudeSettingsEffort(settings)).toBeNull() + }) + + it('publishes the effort from get_settings, which system/init never carries', async () => { + const claude = fakeClaude({ settings: REAL_SETTINGS }) + const adapter = await acquired(claude) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { effort: 'high' } + }) + }) + + it('leaves the effort unreported when the session never learns one', async () => { + const claude = fakeClaude({ settings: { applied: {}, effective: {}, sources: {} } }) + const adapter = await acquired(claude) + + const options = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(options.current.effort).toBeUndefined() + expect(options.current.model).toBeTruthy() + }) + + it('keeps the init fixture free of an effort the real frame never sends', async () => { + const events: ClaudeStructuredSessionEvent[] = [] + await acquired(fakeClaude(), {}, events) + const init = events.flatMap((event) => + event.type === 'message' && event.message.subtype === 'init' ? [event.message] : [] + ) + + expect(init).toHaveLength(1) + expect(init[0]).toHaveProperty('model') + // The regression that hid this defect: a fixture inventing `effortLevel` + // kept every gate green over a value that is always empty in production. + expect(Object.keys(init[0])).not.toContain('effortLevel') + }) +}) + +describe('Claude effort readback', () => { + it('records an effort the child did not adopt without vouching for it', async () => { + const { session, calls } = sessionWith('high') + + // The disagreement stops the confirmation, not the write: no other client + // vetoes here, and the pre-flight catalog guard already refuses the levels + // the model cannot run. + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'bogus-effort-xyz' }, undefined) + ).resolves.toEqual({ effort: 'bogus-effort-xyz' }) + expect(session.confirmedOptions.has('effort')).toBe(false) + // The child's own answer is kept rather than discarded with the refusal. + expect(session.reportedOptions.effort).toBe('high') + expect(calls).toEqual(['apply:bogus-effort-xyz', 'get_settings']) + }) + + it('records an effort the child confirms', async () => { + const { session } = sessionWith('low') + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'low' }, undefined) + ).resolves.toEqual({ effort: 'low' }) + }) + + it('records the request when the readback is unavailable', async () => { + // No evidence of a refusal is not evidence of one; the apply itself succeeded. + const { session } = sessionWith(null) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'low' }, undefined) + ).resolves.toEqual({ effort: 'low' }) + }) +}) + +describe('Claude effort against the model that must run it', () => { + const HAIKU = { value: 'haiku', resolvedModel: 'claude-haiku-4-5-20251001', displayName: 'Haiku' } + const SONNET = { + value: 'sonnet', + resolvedModel: 'claude-sonnet-5', + displayName: 'Sonnet', + supportsEffort: true, + supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'] + } + + it('refuses an effort the current model advertises no control for', async () => { + const { session, calls } = sessionWith('high', [], { model: 'haiku', catalog: [HAIKU, SONNET] }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).rejects.toBeInstanceOf(AgentSessionOptionRejectedError) + // Measured on Claude Code 2.1.260: apply_flag_settings stores `high` on a + // haiku session and get_settings reads it straight back, so a send here is + // never undone. The refusal has to land before the write. + expect(calls).toEqual(['list_models']) + expect(session.options.has('effort')).toBe(false) + }) + + it('refuses a level outside the ones the current model advertises', async () => { + const { session } = sessionWith('high', [], { + model: 'sonnet', + catalog: [{ ...SONNET, supportedEffortLevels: ['low', 'medium'] }] + }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).rejects.toBeInstanceOf(AgentSessionOptionRejectedError) + }) + + it('sends an effort the current model advertises', async () => { + const { session, calls } = sessionWith('high', [], { + model: 'sonnet', + catalog: [HAIKU, SONNET] + }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'high' }) + expect(calls).toEqual(['list_models', 'apply:high', 'get_settings']) + expect(session.confirmedOptions.has('effort')).toBe(true) + }) + + it('sends `max`, which the readback cannot report, when the model advertises it', async () => { + // UNREPORTED_EFFORTS still governs: no get_settings, so no false disagreement. + const { session, calls } = sessionWith('high', [], { model: 'sonnet', catalog: [SONNET] }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'max' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'max' }) + expect(calls).toEqual(['list_models', 'apply:max']) + expect(session.confirmedOptions.has('effort')).toBe(false) + }) + + it('sends the effort when the model is not in the catalog the CLI listed', async () => { + // An unlisted model is an unknown one, not one that refuses effort. + const { session, calls } = sessionWith('high', [], { model: 'sonnet', catalog: [HAIKU] }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'high' }) + expect(calls).toEqual(['list_models', 'apply:high', 'get_settings']) + }) + + it('sends the effort when list_models is unavailable', async () => { + const calls: string[] = [] + const { session } = sessionWith('high', calls, { model: 'sonnet', catalog: [] }) + session.connection.supportedModels = async () => { + calls.push('list_models') + throw new Error('this CLI predates list_models') + } + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'high' }) + expect(calls).toEqual(['list_models', 'apply:high', 'get_settings']) + }) + + it('matches the model the init frame reported, not just the id the user picked', async () => { + const { session } = sessionWith('high', [], { model: 'sonnet', catalog: [HAIKU, SONNET] }) + session.options.delete('model') + session.reportedOptions.model = 'claude-haiku-4-5-20251001' + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).rejects.toBeInstanceOf(AgentSessionOptionRejectedError) + }) + + it('keeps a disagreeing effort through restore instead of skipping it', async () => { + const calls: string[] = [] + const { session } = sessionWith('high', calls, { model: 'sonnet', catalog: [SONNET] }) + session.options.set('effort', 'low') + + await restoreClaudeStructuredSessionOptions(session, undefined) + + expect(session.options.get('effort')).toBe('low') + expect(session.restoreSkippedOptions.has('effort')).toBe(false) + expect(session.confirmedOptions.has('effort')).toBe(false) + }) + + it('drops a stale effort on restore instead of replaying it onto the new model', async () => { + const calls: string[] = [] + const { session } = sessionWith('high', calls, { model: 'sonnet', catalog: [HAIKU, SONNET] }) + session.options.set('model', 'haiku') + session.options.set('effort', 'high') + + await restoreClaudeStructuredSessionOptions(session, undefined) + + expect(session.options.has('effort')).toBe(false) + expect(session.restoreSkippedOptions.has('effort')).toBe(true) + expect(calls.filter((call) => call.startsWith('apply:'))).toEqual([]) + }) +}) diff --git a/src/main/claude/claude-structured-inbound-control.test.ts b/src/main/claude/claude-structured-inbound-control.test.ts new file mode 100644 index 00000000000..07be4bbb516 --- /dev/null +++ b/src/main/claude/claude-structured-inbound-control.test.ts @@ -0,0 +1,164 @@ +import { describe, expect, it, vi } from 'vitest' +import type { CanUseTool } from '@anthropic-ai/claude-agent-sdk' +import { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import { + buildClaudePermissionCallbacks, + CLAUDE_BLOCKING_CONTROL_CALLBACKS, + CLAUDE_CAN_USE_TOOL_SUBTYPE, + CLAUDE_REQUEST_USER_DIALOG_SUBTYPE +} from './claude-structured-inbound-control' + +type CanUseToolOptions = Parameters[2] + +function permissionOptions( + requestId: string, + toolUseID: string, + signal: AbortSignal, + suggestions?: unknown[] +): CanUseToolOptions { + return { + requestId, + toolUseID, + signal, + ...(suggestions ? { suggestions } : {}) + } as unknown as CanUseToolOptions +} + +function callbacksFor() { + const prompts = new ClaudePromptRegistry() + const emit = vi.fn() + const { canUseTool, onUserDialog } = buildClaudePermissionCallbacks({ + sessionId: 'session-1', + prompts, + emit + }) + return { prompts, emit, canUseTool, onUserDialog } +} + +describe('Claude permission callbacks', () => { + it('registers a decodable can_use_tool as a durable prompt and settles it from the registry', async () => { + const control = callbacksFor() + const answered = control.canUseTool( + 'Bash', + { command: 'git status' }, + permissionOptions('perm-1', 'tool-1', new AbortController().signal, [{ type: 'addRules' }]) + ) + + expect(control.emit).toHaveBeenCalledWith( + expect.objectContaining({ + type: 'prompt', + sessionId: 'session-1', + prompt: expect.objectContaining({ promptKey: 'perm-1', toolName: 'Bash', kind: 'approval' }) + }) + ) + const found = control.prompts.find('perm-1') + expect(found?.prompt.suggestions).toEqual([{ type: 'addRules' }]) + // The prompt's settle is the SDK callback's own resolve — answering resolves this promise. + found?.prompt.settle({ behavior: 'allow', toolUseID: 'tool-1' }) + await expect(answered).resolves.toEqual({ behavior: 'allow', toolUseID: 'tool-1' }) + }) + + it('denies a malformed permission request without registering a prompt', async () => { + const control = callbacksFor() + const answered = control.canUseTool( + '', + {}, + permissionOptions('perm-2', 'tool-2', new AbortController().signal) + ) + + await expect(answered).resolves.toEqual({ + behavior: 'deny', + message: 'Orca could not decode this permission request.', + toolUseID: 'tool-2' + }) + expect(control.prompts.find('perm-2')).toBeNull() + expect(control.emit).not.toHaveBeenCalled() + }) + + it('settles a pending prompt with null and forgets it when the abort signal fires', async () => { + const control = callbacksFor() + const controller = new AbortController() + const answered = control.canUseTool( + 'Bash', + { command: 'ls' }, + permissionOptions('perm-3', 'tool-3', controller.signal) + ) + expect(control.prompts.find('perm-3')).not.toBeNull() + + controller.abort() + + await expect(answered).resolves.toBeNull() + expect(control.emit).toHaveBeenLastCalledWith( + expect.objectContaining({ type: 'prompt-cancelled', promptKey: 'perm-3' }) + ) + // Forgotten: a late answer can no longer find the prompt to authorize the wrong tool. + expect(control.prompts.find('perm-3')).toBeNull() + }) + + it('cancels a request whose abort raced ahead of delivery without emitting a prompt', async () => { + const control = callbacksFor() + const controller = new AbortController() + controller.abort() + + const answered = control.canUseTool( + 'Bash', + { command: 'ls' }, + permissionOptions('perm-4', 'tool-4', controller.signal) + ) + + await expect(answered).resolves.toBeNull() + expect(control.prompts.find('perm-4')).toBeNull() + expect(control.emit).toHaveBeenCalledTimes(1) + expect(control.emit).toHaveBeenCalledWith( + expect.objectContaining({ type: 'prompt-cancelled', promptKey: 'perm-4' }) + ) + }) + + it('settles every in-flight prompt with null when the registry is cleared', async () => { + const control = callbacksFor() + const first = control.canUseTool( + 'Bash', + { command: 'a' }, + permissionOptions('perm-5', 'tool-5', new AbortController().signal) + ) + const second = control.canUseTool( + 'Bash', + { command: 'b' }, + permissionOptions('perm-6', 'tool-6', new AbortController().signal) + ) + + // What session close does: settle each pending callback so no promise dangles. + for (const prompt of control.prompts.clear()) { + prompt.settle(null) + } + + await expect(first).resolves.toBeNull() + await expect(second).resolves.toBeNull() + }) + + it('answers a user dialog deny-safe', async () => { + const control = callbacksFor() + await expect( + control.onUserDialog( + { dialogKind: 'refusal_fallback_prompt', payload: {} }, + { signal: new AbortController().signal, requestId: 'dialog-1' } + ) + ).resolves.toEqual({ behavior: 'cancelled' }) + }) + + it('enumerates every blocking control request and wires a callback for each', () => { + // The stable surface of controls a turn can block on. Adding one here without wiring its + // callback below fails this test rather than silently leaving a control unhandled. + expect(new Set(Object.keys(CLAUDE_BLOCKING_CONTROL_CALLBACKS))).toEqual( + new Set([CLAUDE_CAN_USE_TOOL_SUBTYPE, CLAUDE_REQUEST_USER_DIALOG_SUBTYPE]) + ) + const callbacks = buildClaudePermissionCallbacks({ + sessionId: 'session-1', + prompts: new ClaudePromptRegistry(), + emit: vi.fn() + }) as unknown as Record + for (const callbackName of Object.values(CLAUDE_BLOCKING_CONTROL_CALLBACKS)) { + expect(typeof callbacks[callbackName], `${callbackName} must be wired`).toBe('function') + } + }) +}) diff --git a/src/main/claude/claude-structured-inbound-control.ts b/src/main/claude/claude-structured-inbound-control.ts new file mode 100644 index 00000000000..343e76d4ea5 --- /dev/null +++ b/src/main/claude/claude-structured-inbound-control.ts @@ -0,0 +1,91 @@ +import type { CanUseTool, OnUserDialog, PermissionResult } from '@anthropic-ai/claude-agent-sdk' +import type { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import type { ClaudeStructuredSessionEvent } from './claude-structured-session-state' + +export const CLAUDE_CAN_USE_TOOL_SUBTYPE = 'can_use_tool' +export const CLAUDE_REQUEST_USER_DIALOG_SUBTYPE = 'request_user_dialog' + +/** + * The blocking control requests Orca answers, each mapped to the SDK consumer callback that + * answers it. This is the stable surface a real turn can block on: `can_use_tool` through + * `canUseTool` and `request_user_dialog` through `onUserDialog`. Every other control-request + * subtype the SDK routes (elicitation, oauth/host token refresh, mcp_message, hook_callback) + * is either not surfaced to this consumer or fails closed inside the SDK; adding a new + * blocking control Orca must answer means adding its callback here, and the catalog test + * fails if a named callback is missing. + */ +export const CLAUDE_BLOCKING_CONTROL_CALLBACKS = { + [CLAUDE_CAN_USE_TOOL_SUBTYPE]: 'canUseTool', + [CLAUDE_REQUEST_USER_DIALOG_SUBTYPE]: 'onUserDialog' +} as const + +export type ClaudeBlockingControlSubtype = keyof typeof CLAUDE_BLOCKING_CONTROL_CALLBACKS + +export type ClaudePermissionCallbackDeps = { + sessionId: string + prompts: ClaudePromptRegistry + emit: (event: ClaudeStructuredSessionEvent) => void +} + +function denySafeResult(toolUseId: string | undefined): PermissionResult { + return { + behavior: 'deny', + message: 'Orca could not decode this permission request.', + ...(toolUseId ? { toolUseID: toolUseId } : {}) + } +} + +/** + * Build the SDK permission callbacks from the durable prompt registry. + * + * A decodable `can_use_tool` becomes a durable prompt whose `settle` resolves this callback; + * a malformed one is denied without registering. The SDK's abort signal fires on + * `control_cancel_request` (a cancelled turn), which forgets the prompt and settles it with + * `null` — never authorizing a tool. A late answer after abort finds no prompt and is refused + * by `answerClaudePrompt`. `onUserDialog` is deny-safe; the CLI only emits dialog kinds Orca + * declares in `supportedDialogKinds`, which is empty. + */ +export function buildClaudePermissionCallbacks(deps: ClaudePermissionCallbackDeps): { + canUseTool: CanUseTool + onUserDialog: OnUserDialog +} { + const canUseTool: CanUseTool = (toolName, input, options) => + new Promise((resolve) => { + const prompt = deps.prompts.register({ + requestId: options.requestId, + toolName, + toolUseId: options.toolUseID, + input, + suggestions: options.suggestions ?? [], + settle: resolve as (response: Record | null) => void + }) + if (!prompt) { + resolve(denySafeResult(options.toolUseID)) + return + } + const cancel = (): void => { + if (deps.prompts.forgetIfPending(prompt)) { + deps.emit({ + type: 'prompt-cancelled', + sessionId: deps.sessionId, + promptKey: prompt.promptKey + }) + // Null is the SDK's "no response written" sentinel: a cancelled request must not + // be answered, only forgotten. + resolve(null) + } + } + if (options.signal.aborted) { + // No abort event can still fire, so registering a listener would park the callback + // forever behind a prompt nothing will answer. + cancel() + return + } + options.signal.addEventListener('abort', cancel, { once: true }) + deps.emit({ type: 'prompt', sessionId: deps.sessionId, prompt }) + }) + + const onUserDialog: OnUserDialog = () => Promise.resolve({ behavior: 'cancelled' }) + + return { canUseTool, onUserDialog } +} diff --git a/src/main/claude/claude-structured-init-deadline.ts b/src/main/claude/claude-structured-init-deadline.ts new file mode 100644 index 00000000000..f3acd6c3af9 --- /dev/null +++ b/src/main/claude/claude-structured-init-deadline.ts @@ -0,0 +1,68 @@ +import type { ClaudeInitObservation } from './claude-structured-init-proof' +import { claudeInitializationAuthError } from './claude-structured-init-proof' +import type { ClaudeStreamJsonConnection } from './claude-stream-json-connection' +import { AgentSessionAcquisitionRefusal } from '../native-chat/agent-session-wire/structured-agent-session-adapter' + +export type ClaudeInitDeadline = { + promise: Promise + resolve: (init: ClaudeInitObservation) => void + reject: (error: Error) => void + start: () => void + clear: () => void +} + +export function claudeInitTimeoutError( + sessionId: string, + timeoutMs: number +): AgentSessionAcquisitionRefusal { + return new AgentSessionAcquisitionRefusal( + `Claude did not finish starting session ${sessionId} within ${Math.ceil(timeoutMs / 1000)} seconds. Verify the selected Claude account is signed in and CLAUDE_CONFIG_DIR contains valid credentials, then retry; no SessionStart or system/init proof arrived.` + ) +} + +export async function requestClaudeInitialization( + connection: ClaudeStreamJsonConnection, + sessionId: string, + timeoutMs: number +): Promise { + try { + const result = await connection.initializationResult({ timeoutMs }) + const authError = claudeInitializationAuthError(result) + if (authError) { + throw authError + } + return result + } catch (error) { + if (error instanceof Error && error.message === 'claude initialize request timed out') { + throw claudeInitTimeoutError(sessionId, timeoutMs) + } + throw error + } +} + +export function createClaudeInitDeadline(sessionId: string, timeoutMs: number): ClaudeInitDeadline { + let resolve = (_init: ClaudeInitObservation): void => {} + let reject = (_error: Error): void => {} + const promise = new Promise((resolvePromise, rejectPromise) => { + resolve = resolvePromise + reject = rejectPromise + }) + void promise.catch(() => {}) + let timer: ReturnType | null = null + + return { + promise, + resolve, + reject, + start: () => { + timer = setTimeout(() => reject(claudeInitTimeoutError(sessionId, timeoutMs)), timeoutMs) + timer.unref?.() + }, + clear: () => { + if (timer) { + clearTimeout(timer) + timer = null + } + } + } +} diff --git a/src/main/claude/claude-structured-init-proof.ts b/src/main/claude/claude-structured-init-proof.ts new file mode 100644 index 00000000000..c29cb2d4715 --- /dev/null +++ b/src/main/claude/claude-structured-init-proof.ts @@ -0,0 +1,88 @@ +import { CLAUDE_DEFAULT_SETTING_SOURCES } from './claude-structured-launch-resolution' +import type { ClaudeAuthDiagnostic } from './claude-structured-session-state' +import { AgentSessionAcquisitionRefusal } from '../native-chat/agent-session-wire/structured-agent-session-adapter' + +export type ClaudeInitObservation = { + providerSessionId: string + uuid: string | null + /** The resolved model id the CLI reports it is running; only `system/init` carries it. */ + model: string | null + message: Record +} + +function isRecord(value: unknown): value is Record { + return typeof value === 'object' && value !== null && !Array.isArray(value) +} + +export function readClaudeFrameString(source: Record, key: string): string | null { + const value = source[key] + return typeof value === 'string' && value.length > 0 ? value : null +} + +export function readClaudeInit(message: Record): ClaudeInitObservation | null { + const hookName = readClaudeFrameString(message, 'hook_name') + const isInit = message.type === 'system' && message.subtype === 'init' + const isSessionStart = + message.type === 'system' && + (message.subtype === 'hook_started' || message.subtype === 'hook_response') && + hookName?.startsWith('SessionStart:') === true + if (!isInit && !isSessionStart) { + return null + } + const providerSessionId = readClaudeFrameString(message, 'session_id') + return providerSessionId + ? { + providerSessionId, + uuid: isInit ? readClaudeFrameString(message, 'uuid') : null, + model: isInit ? readClaudeFrameString(message, 'model') : null, + message + } + : null +} + +export function readClaudeModels(initialization: unknown): unknown[] { + return isRecord(initialization) && Array.isArray(initialization.models) + ? initialization.models + : [] +} + +/** CLI capabilities advertised on the initialize result or the yielded system/init frame. */ +export function readClaudeCapabilities( + init: ClaudeInitObservation, + initialization: unknown +): string[] { + const fromResult = isRecord(initialization) ? initialization.capabilities : undefined + const fromFrame = init.message.capabilities + const source = Array.isArray(fromResult) ? fromResult : Array.isArray(fromFrame) ? fromFrame : [] + return source.filter((value): value is string => typeof value === 'string') +} + +export function claudeInitializationAuthError( + initialization: unknown +): AgentSessionAcquisitionRefusal | null { + const account = + isRecord(initialization) && isRecord(initialization.account) ? initialization.account : null + return readClaudeFrameString(account ?? {}, 'tokenSource') === 'none' + ? new AgentSessionAcquisitionRefusal( + 'Claude is not signed in for the selected account. Sign in with the Claude CLI for this CLAUDE_CONFIG_DIR, then retry.' + ) + : null +} + +export function claudeAuthDiagnostic( + init: ClaudeInitObservation, + settings: unknown +): ClaudeAuthDiagnostic { + const env = isRecord(settings) && isRecord(settings.env) ? settings.env : {} + const apiKeySource = readClaudeFrameString(init.message, 'apiKeySource') + const configured = (key: string): boolean => + (typeof env[key] === 'string' && (env[key] as string).trim().length > 0) || + Boolean(process.env[key]?.trim()) + return { + apiKeySourceConfigured: apiKeySource !== null && apiKeySource !== 'none', + baseUrlConfigured: configured('ANTHROPIC_BASE_URL'), + authTokenConfigured: configured('ANTHROPIC_AUTH_TOKEN'), + apiKeyConfigured: configured('ANTHROPIC_API_KEY'), + settingSources: CLAUDE_DEFAULT_SETTING_SOURCES + } +} diff --git a/src/main/claude/claude-structured-item-translation.ts b/src/main/claude/claude-structured-item-translation.ts new file mode 100644 index 00000000000..d86093ee0a5 --- /dev/null +++ b/src/main/claude/claude-structured-item-translation.ts @@ -0,0 +1,179 @@ +import type { + AgentJournalItemBody, + AgentJournalItemIdentity, + AgentJournalMessageItem +} from '../../shared/agent-session-journal-types' +import type { NativeChatBlock } from '../../shared/native-chat-types' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' + +export type ClaudeMessageEnvelope = { + sessionId: string + uuid: string + role: 'assistant' | 'user' + content: unknown[] + /** Messages API id shared by every frame of one streamed assistant message. */ + messageId: string | null + parentToolUseId: string | null +} + +export type ClaudeToolUse = { id: string; name: string; input: unknown } +export type ClaudeToolResult = { toolUseId: string; output: string; failed: boolean } + +export function claudeRecord(value: unknown): Record | null { + return typeof value === 'object' && value !== null && !Array.isArray(value) + ? (value as Record) + : null +} + +export function claudeText(value: unknown): string | null { + return typeof value === 'string' && value.length > 0 ? value : null +} + +export function readClaudeMessageEnvelope( + frame: Record +): ClaudeMessageEnvelope | null { + if (frame.type !== 'assistant' && frame.type !== 'user') { + return null + } + const message = claudeRecord(frame.message) + const sessionId = claudeText(frame.session_id) + const uuid = claudeText(frame.uuid) + const role = message?.role + return sessionId && uuid && (role === 'assistant' || role === 'user') + ? { + sessionId, + uuid, + role, + content: messageContent(message?.content), + messageId: claudeText(message?.id), + parentToolUseId: claudeText(frame.parent_tool_use_id) + } + : null +} + +// A user replay may carry its text as a bare string (MessageParam), not blocks. +function messageContent(content: unknown): unknown[] { + if (Array.isArray(content)) { + return content + } + const text = claudeText(content) + return text ? [{ type: 'text', text }] : [] +} + +export function claudeMessageIdentity( + envelope: Pick +): AgentJournalItemIdentity { + return { provider: 'claude', sessionId: envelope.sessionId, uuid: envelope.uuid } +} + +function messageBlocks(envelope: ClaudeMessageEnvelope): NativeChatBlock[] { + const blocks: NativeChatBlock[] = [] + for (const value of envelope.content) { + const part = claudeRecord(value) + const text = claudeText(part?.text) + if (part?.type === 'text' && text) { + blocks.push({ type: 'text', text }) + continue + } + const source = claudeRecord(part?.source) + const url = claudeText(source?.url) + if (part?.type === 'image' && source?.type === 'url' && url) { + blocks.push({ type: 'image-ref', url }) + } + } + return blocks +} + +export function claudeMessageBody(envelope: ClaudeMessageEnvelope): AgentJournalMessageItem | null { + const blocks = messageBlocks(envelope) + return blocks.length > 0 ? { kind: 'message', role: envelope.role, blocks } : null +} + +export function claudeHasReplayContent(envelope: ClaudeMessageEnvelope): boolean { + return envelope.content.some((value) => { + const part = claudeRecord(value) + return part !== null && part.type !== 'tool_result' + }) +} + +export function claudeToolUses(envelope: ClaudeMessageEnvelope): ClaudeToolUse[] { + return envelope.content.flatMap((value) => { + const part = claudeRecord(value) + const id = claudeText(part?.id) + const name = claudeText(part?.name) + return part?.type === 'tool_use' && id && name ? [{ id, name, input: part.input ?? null }] : [] + }) +} + +function resultText(value: unknown): string { + if (typeof value === 'string') { + return value + } + if (!Array.isArray(value)) { + return value === undefined ? '' : JSON.stringify(value) + } + return value + .flatMap((entry) => { + if (typeof entry === 'string') { + return [entry] + } + const part = claudeRecord(entry) + return part?.type === 'text' && typeof part.text === 'string' ? [part.text] : [] + }) + .join('\n') +} + +export function claudeToolResults(envelope: ClaudeMessageEnvelope): ClaudeToolResult[] { + return envelope.content.flatMap((value) => { + const part = claudeRecord(value) + const toolUseId = claudeText(part?.tool_use_id) + return part?.type === 'tool_result' && toolUseId + ? [ + { + toolUseId, + output: resultText(part.content), + failed: part.is_error === true + } + ] + : [] + }) +} + +export function claudeThinkingText(envelope: ClaudeMessageEnvelope): string | null { + const parts = envelope.content.flatMap((value) => { + const part = claudeRecord(value) + const thinking = claudeText(part?.thinking) + return part?.type === 'thinking' && thinking ? [thinking] : [] + }) + return parts.length > 0 ? parts.join('\n') : null +} + +export function claudeToolBody(input: { + tool: ClaudeToolUse + result?: ClaudeToolResult +}): AgentJournalItemBody { + return { + kind: 'tool-call', + name: input.tool.name, + input: input.tool.input, + state: input.result ? (input.result.failed ? 'failed' : 'completed') : 'running', + ...(input.result + ? { output: boundInlineText(input.result.output, DEFAULT_JOURNAL_PAYLOAD_LIMITS).bounded } + : {}) + } +} + +export function claudeStreamingMessageBody(text: string): AgentJournalMessageItem { + return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text }] } +} + +export function claudeToolIdentity(sessionId: string, toolUseId: string): AgentJournalItemIdentity { + return { provider: 'orca', clientMessageId: `claude-tool:${sessionId}:${toolUseId}` } +} + +export function claudeThinkingIdentity(sessionId: string, uuid: string): AgentJournalItemIdentity { + return { provider: 'orca', clientMessageId: `claude-thinking:${sessionId}:${uuid}` } +} diff --git a/src/main/claude/claude-structured-journal-translation.test.ts b/src/main/claude/claude-structured-journal-translation.test.ts new file mode 100644 index 00000000000..f403313dae8 --- /dev/null +++ b/src/main/claude/claude-structured-journal-translation.test.ts @@ -0,0 +1,811 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity, + AgentJournalRenderItem, + AgentSessionJournalIdentity +} from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { activeStructuredAgentSessionTurnId } from '../../shared/structured-agent-session-projection' +import { openAgentSessionJournal } from '../native-chat/agent-session-journal/journal-store-factory' +import { + createDeferredStructuredAgentSessionEventSink, + type StructuredAgentSessionEventSink +} from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' +import type { ClaudePendingPrompt } from './claude-structured-prompt-replies' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' + +function sinkState() { + const items: { identity: AgentJournalItemIdentity; body: AgentJournalItemBody }[] = [] + const tombstones: AgentJournalItemIdentity[] = [] + const sink: StructuredAgentSessionEventSink = { + appendItem: (identity, body) => items.push({ identity, body }), + appendTombstone: (identity) => tombstones.push(identity), + publish: vi.fn() + } + return { sink, items, tombstones } +} + +function message( + type: 'assistant' | 'user', + uuid: string, + content: unknown[], + parentToolUseId: string | null = null +) { + return { + type: 'message' as const, + sessionId: 'orca-session', + ...(type === 'user' && parentToolUseId === null ? { startsTurn: true as const } : {}), + message: { + type, + uuid, + session_id: 'claude-session', + parent_tool_use_id: parentToolUseId, + message: { role: type, content } + } + } +} + +// Frames below follow the Claude Code 2.1.258 / SDK 0.3.251 partial-message +// cadence captured from the real CLI: every stream_event carries its own uuid, +// the final assistant frame for a block carries yet another, and only +// message.id ties them together. +function streamEvent(uuid: string, event: Record) { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'stream_event', + uuid, + session_id: 'claude-session', + parent_tool_use_id: null, + event + } + } +} + +function resultFrame(subtype: string, fields: Record) { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'result', + subtype, + duration_ms: 1200, + duration_api_ms: 1100, + num_turns: 1, + session_id: 'claude-session', + uuid: `result-${subtype}`, + ...fields + } + } +} + +/** One streamed text turn in wire order: message_start, the block's start frame, + * one delta per chunk, the block's final assistant frame, the stop frames and + * the success result. */ +function streamedTextTurn(input: { + messageId: string + startUuid: string + finalUuid: string + chunks: string[] +}) { + const text = input.chunks.join('') + return { + start: [ + streamEvent(`${input.messageId}-message-start`, { + type: 'message_start', + message: { id: input.messageId, role: 'assistant', content: [] } + }), + streamEvent(input.startUuid, { + type: 'content_block_start', + index: 0, + content_block: { type: 'text', text: '' } + }) + ], + deltas: input.chunks.map((chunk, index) => + streamEvent(`${input.messageId}-delta-${index}`, { + type: 'content_block_delta', + index: 0, + delta: { type: 'text_delta', text: chunk } + }) + ), + final: { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'assistant', + uuid: input.finalUuid, + session_id: 'claude-session', + parent_tool_use_id: null, + message: { + id: input.messageId, + role: 'assistant', + content: [{ type: 'text', text }], + stop_reason: null + } + } + }, + stop: [ + streamEvent(`${input.messageId}-block-stop`, { type: 'content_block_stop', index: 0 }), + streamEvent(`${input.messageId}-message-delta`, { + type: 'message_delta', + delta: { stop_reason: 'end_turn' } + }), + streamEvent(`${input.messageId}-message-stop`, { type: 'message_stop' }), + resultFrame('success', { + is_error: false, + result: text, + stop_reason: 'end_turn', + terminal_reason: 'completed' + }) + ], + text + } +} + +function assistantMessages(items: T[]): T[] { + return items.filter((item) => item.body.kind === 'message' && item.body.role === 'assistant') +} + +function providerFrameKinds(items: { body: AgentJournalItemBody }[]): string[] { + return items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame ? [item.body.providerFrame.kind] : [] + ) +} + +const JOURNAL_IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-1', + workspaceId: 'workspace-1', + hostId: 'host-1', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: 'claude-session', leafUuid: 'leaf-1' } +} + +let journalRoot = '' + +beforeEach(async () => { + journalRoot = await mkdtemp(join(tmpdir(), 'orca-claude-journal-translation-')) +}) + +afterEach(async () => { + await rm(journalRoot, { recursive: true, force: true }) +}) + +describe('Claude structured journal translation', () => { + it('coalesces partial deltas onto the block identity and reconciles the final frame onto it', () => { + const state = sinkState() + let scheduled: (() => void) | null = null + const translator = createClaudeJournalTranslator({ + sink: state.sink, + schedule: (run, delay) => { + expect(delay).toBe(60) + scheduled = run + return () => { + scheduled = null + } + } + }) + const turn = streamedTextTurn({ + messageId: 'msg_01', + startUuid: 'block-start-1', + finalUuid: 'assistant-final-1', + chunks: ['ST', 'REAMOK_ELEC_64E632'] + }) + const streamedIdentity = { + provider: 'claude', + sessionId: 'claude-session', + uuid: 'block-start-1' + } + + for (const event of turn.start) { + translator.handle(event) + } + for (const delta of turn.deltas) { + translator.handle(delta) + } + expect(state.items).toEqual([]) + + const run = scheduled as (() => void) | null + run?.() + expect(state.items.at(-1)).toEqual({ + identity: streamedIdentity, + body: { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: turn.text }] } + }) + + translator.handle(turn.final) + for (const event of turn.stop) { + translator.handle(event) + } + const assistant = assistantMessages(state.items) + expect(assistant.at(-1)).toEqual({ + identity: streamedIdentity, + body: { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: turn.text }] } + }) + expect(new Set(assistant.map((item) => agentJournalItemKey(item.identity))).size).toBe(1) + expect(providerFrameKinds(state.items)).toEqual([]) + }) + + it('journals a count-to-200 stream as one assistant item carrying the complete reply', async () => { + const journal = await openAgentSessionJournal({ + identity: JOURNAL_IDENTITY, + journalDir: journalRoot, + now: () => 1_700_000_000_000, + mintEpoch: () => 'epoch-1' + }) + const deferred = createDeferredStructuredAgentSessionEventSink() + deferred.bind({ journal, fence: 1, publish: vi.fn() }) + let scheduled: (() => void) | null = null + const translator = createClaudeJournalTranslator({ + sink: deferred.sink, + schedule: (run) => { + scheduled = run + return () => { + scheduled = null + } + } + }) + const numbers = Array.from({ length: 200 }, (_, index) => String(index + 1)) + // The chunk boundaries the real CLI produced for this prompt. + const boundaries = [0, 1, 45, 93, 141, 189, 200] + const chunks = boundaries.slice(1).map((end, index) => { + const slice = numbers.slice(boundaries[index], end).join('\n') + return index === 0 ? slice : `\n${slice}` + }) + const turn = streamedTextTurn({ + messageId: 'msg_count', + startUuid: 'count-start', + finalUuid: 'count-final', + chunks + }) + + for (const event of turn.start) { + translator.handle(event) + } + for (const delta of turn.deltas) { + translator.handle(delta) + // Each chunk lands in its own coalescing window, as it did on the wire. + const run = scheduled as (() => void) | null + run?.() + } + translator.handle(turn.final) + for (const event of turn.stop) { + translator.handle(event) + } + await deferred.drained() + + const items: AgentJournalRenderItem[] = journal.snapshot().items + const assistant = assistantMessages(items) + expect(assistant.map((item) => item.itemId)).toEqual(['claude:claude-session:count-start']) + expect(assistant[0]?.body).toEqual({ + kind: 'message', + role: 'assistant', + blocks: [{ type: 'text', text: numbers.join('\n') }] + }) + expect(providerFrameKinds(items)).toEqual([]) + }) + + it('settles result frames, empty thinking and string user replays without painting a row', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle({ + type: 'message', + sessionId: 'orca-session', + startsTurn: true, + message: { + type: 'user', + uuid: 'user-replay-1', + session_id: 'claude-session', + parent_tool_use_id: null, + isReplay: true, + timestamp: '2026-09-01T00:00:00.000Z', + message: { role: 'user', content: 'Reply with exactly PROBE_OK_1 and nothing else.' } + } + }) + translator.handle( + message('assistant', 'assistant-thinking-empty', [ + { type: 'thinking', thinking: '', signature: 'CAQS6QcKEAgRGAI4AUIIdGhpbmtpbmc' } + ]) + ) + translator.handle( + resultFrame('success', { + is_error: false, + result: 'PROBE_OK_1', + stop_reason: 'end_turn', + terminal_reason: 'completed' + }) + ) + translator.handle( + message('user', 'user-interrupt', [{ type: 'text', text: '[Request interrupted by user]' }]) + ) + translator.handle( + resultFrame('error_during_execution', { + is_error: true, + errors: ['[ede_diagnostic] result_type=user last_content_type=n/a stop_reason=null'], + stop_reason: null, + terminal_reason: 'aborted_streaming', + permission_denials: [] + }) + ) + translator.handle(message('user', 'control-only', [])) + + expect(providerFrameKinds(state.items)).toEqual([]) + expect( + state.items.flatMap((item) => + item.body.kind === 'message' && item.body.role === 'user' ? [item.body.blocks] : [] + ) + ).toEqual([ + [{ type: 'text', text: 'Reply with exactly PROBE_OK_1 and nothing else.' }], + [{ type: 'text', text: '[Request interrupted by user]' }] + ]) + expect( + state.items.some((item) => item.body.kind === 'status' && !item.body.turnLifecycle) + ).toBe(false) + expect( + state.tombstones.flatMap((identity) => + identity.provider === 'legacy' ? [identity.recordId] : [] + ) + ).toEqual(['turn-lifecycle:user-replay-1', 'turn-lifecycle:user-interrupt']) + }) + + it('does not reopen a completed turn when the SDK replays its user row after restart', () => { + const live = sinkState() + const liveTranslator = createClaudeJournalTranslator({ sink: live.sink }) + const replay = { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'user', + uuid: 'picker-command-1', + session_id: 'claude-session', + parent_tool_use_id: null, + isReplay: true, + message: { role: 'user', content: '/model' } + } + } + + liveTranslator.handle({ ...replay, startsTurn: true }) + liveTranslator.handle(resultFrame('success', { is_error: false, result: '' })) + expect(live.tombstones).toContainEqual({ + provider: 'legacy', + agent: 'claude', + sessionId: 'claude-session', + recordId: 'turn-lifecycle:picker-command-1' + }) + liveTranslator.dispose() + + const restarted = sinkState() + const restartedTranslator = createClaudeJournalTranslator({ sink: restarted.sink }) + restartedTranslator.handle(replay) + + expect( + activeStructuredAgentSessionTurnId( + restarted.items.map((item, sequence) => ({ + itemId: agentJournalItemKey(item.identity), + revision: 1, + body: item.body, + sequence, + observedAt: sequence + })) + ) + ).toBeNull() + }) + + it('surfaces an API error carried by a success-subtype result with no assistant frame', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(message('user', 'user-1', [{ type: 'text', text: 'summarize this' }])) + // The SDK models this as a SUCCESS-subtype result whose `result` string is the + // user-facing API error. Suppressing it as ordinary turn bookkeeping ends the + // turn with nothing shown at all. + translator.handle( + resultFrame('success', { + is_error: true, + result: 'API Error: 529 upstream overloaded', + stop_reason: null, + terminal_reason: 'api_error' + }) + ) + + expect(providerFrameKinds(state.items)).toEqual(['message:result:success']) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'status', + text: 'API Error: 529 upstream overloaded' + }) + // The turn still settles: the error is an extra row, not a stuck lifecycle. + expect( + state.tombstones.flatMap((identity) => + identity.provider === 'legacy' ? [identity.recordId] : [] + ) + ).toEqual(['turn-lifecycle:user-1']) + }) + + it('drops the stream state of turns that ended without their final frame', () => { + const state = sinkState() + let scheduled: (() => void) | null = null + const translator = createClaudeJournalTranslator({ + sink: state.sink, + schedule: (run) => { + scheduled = run + return () => { + scheduled = null + } + } + }) + for (let turn = 0; turn < 3; turn += 1) { + const aborted = streamedTextTurn({ + messageId: `msg_abort_${turn}`, + startUuid: `abort-start-${turn}`, + finalUuid: `abort-final-${turn}`, + chunks: ['x'.repeat(4_000)] + }) + for (const event of [...aborted.start, ...aborted.deltas]) { + translator.handle(event) + } + const run = scheduled as (() => void) | null + run?.() + // The user interrupts: the result arrives with no final assistant frame, + // so nothing ever reconciles these blocks. + translator.handle( + resultFrame('error_during_execution', { + is_error: true, + terminal_reason: 'aborted_streaming' + }) + ) + // The partial text is already journaled; only the live state is dropped. + expect(translator.pendingStreamedBlocks).toBe(0) + } + + expect(assistantMessages(state.items)).toHaveLength(3) + }) + + it('keeps an ordinary successful result off the timeline', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(resultFrame('success', { is_error: false, result: 'done', errors: [] })) + + expect(providerFrameKinds(state.items)).toEqual([]) + }) + + it('surfaces the reason an error-subtype result stopped the turn', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + resultFrame('error_max_turns', { is_error: true, errors: ['turn limit reached'] }) + ) + + expect(providerFrameKinds(state.items)).toEqual(['message:result:error_max_turns']) + }) + + it('keeps an unmodeled result subtype on the bounded provider fallback', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + resultFrame('error_from_the_future', { is_error: true, errors: ['budget exhausted'] }) + ) + + expect(providerFrameKinds(state.items)).toEqual(['message:result:error_from_the_future']) + }) + + it('journals turn lifecycle and updates one tool row through its result', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(message('user', 'user-1', [{ type: 'text', text: 'List files' }])) + translator.handle( + message('assistant', 'assistant-tool', [ + { type: 'tool_use', id: 'tool-1', name: 'Bash', input: { command: 'ls' } } + ]) + ) + translator.handle( + message( + 'user', + 'tool-result-1', + [{ type: 'tool_result', tool_use_id: 'tool-1', content: 'a.ts\nb.ts' }], + 'tool-1' + ) + ) + + const keyed = new Map( + state.items.map((item) => [agentJournalItemKey(item.identity), item.body]) + ) + expect(keyed.get('claude:claude-session:user-1')).toMatchObject({ + kind: 'message', + role: 'user' + }) + expect(keyed.get('orca:claude-tool%3Aclaude-session%3Atool-1')).toMatchObject({ + kind: 'tool-call', + name: 'Bash', + state: 'completed', + output: { head: 'a.ts\nb.ts', truncated: false } + }) + expect( + state.items.some( + (item) => item.body.kind === 'status' && item.body.turnLifecycle?.turnId === 'user-1' + ) + ).toBe(true) + + translator.handle( + message( + 'user', + 'tool-result-2', + [{ type: 'tool_result', tool_use_id: 'tool-1', content: 'done again' }], + 'tool-1' + ) + ) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'tool-call', + name: 'tool', + input: null, + output: { head: 'done again' } + }) + + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'result', session_id: 'claude-session', uuid: 'result-1' } + }) + expect(state.tombstones.at(-1)).toMatchObject({ + provider: 'legacy', + agent: 'claude', + recordId: 'turn-lifecycle:user-1' + }) + }) + + it('bounds persisted thinking text to the shared journal payload limit', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + const thinking = 'considering '.repeat(20_000) + + translator.handle(message('assistant', 'assistant-thinking', [{ type: 'thinking', thinking }])) + + expect(state.items.at(-1)?.body).toEqual({ + kind: 'status', + text: boundInlineText(thinking, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text + }) + }) + + it('starts a cancellable lifecycle for image-only root user replays', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + message('user', 'user-image', [ + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'AA==' } } + ]) + ) + + expect(state.items.at(-1)?.body).toEqual({ + kind: 'status', + text: 'Claude is working…', + turnLifecycle: { turnId: 'user-image', state: 'running' } + }) + }) + + it('does not start a lifecycle for a top-level user tool result', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + message('user', 'tool-result-only', [ + { type: 'tool_result', tool_use_id: 'tool-1', content: 'done' } + ]) + ) + + expect(state.items.map((item) => agentJournalItemKey(item.identity))).toEqual([ + 'orca:claude-tool%3Aclaude-session%3Atool-1' + ]) + expect(state.items[0]?.body).toMatchObject({ + kind: 'tool-call', + state: 'completed', + output: { head: 'done' } + }) + expect( + state.items.some( + (item) => item.body.kind === 'status' && item.body.turnLifecycle !== undefined + ) + ).toBe(false) + }) + + it('paints nothing for a user frame that carries no content', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(message('user', 'control-only', [])) + + expect(state.items).toEqual([]) + expect(state.tombstones).toEqual([]) + }) + + it('renders unmodeled substantive Claude frames as bounded provider rows', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'system', subtype: 'local_command_output', summary: 'x'.repeat(100_000) } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'system', subtype: 'hook_response', hook_name: 'PostToolUse' } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'system', subtype: 'command_started', command: '/compact' } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'result', usage: { input_tokens: 12 }, total_cost_usd: 0.01 } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'tool_progress', tool_use_id: 'tool-1', elapsed_time_seconds: 2 } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'prompt_suggestion', suggestion: '/compact' } + }) + translator.handle( + message('user', 'attachment-1', [ + { type: 'document', source: { type: 'base64', media_type: 'application/pdf' } } + ]) + ) + translator.handle({ + type: 'provider-frame', + sessionId: 'orca-session', + kind: 'control_request:future_control', + payload: { subtype: 'future_control' } + }) + + const frames = state.items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame ? [item.body.providerFrame] : [] + ) + expect(frames.map((frame) => frame.kind)).toEqual( + expect.arrayContaining([ + 'message:system:local_command_output', + 'message:system:command_started', + 'message:result', + 'message:user:content:document', + 'control_request:future_control' + ]) + ) + expect(frames.map((frame) => frame.kind)).not.toEqual( + expect.arrayContaining([ + 'message:system:hook_response', + 'message:tool_progress', + 'message:prompt_suggestion' + ]) + ) + expect( + frames.find((frame) => frame.kind === 'message:system:local_command_output')?.payload + ).toEqual(expect.objectContaining({ truncated: true, byteLength: expect.any(Number) })) + }) + + it('preserves a question group as one addressable prompt and cancels it durably', () => { + const state = sinkState() + const bindings: unknown[][] = [] + const translator = createClaudeJournalTranslator({ + sink: state.sink, + bindPromptItemId: (...args) => bindings.push(args) + }) + const approval = prompt({ + requestId: 'permission-1', + promptKey: 'permission-1', + toolUseId: 'tool-1', + toolName: 'Bash', + kind: 'approval', + input: { command: 'git status' }, + questionIds: [] + }) + translator.handle({ type: 'prompt', sessionId: 'orca-session', prompt: approval }) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'approval', + title: 'Allow Bash?', + options: expect.arrayContaining([{ id: 'allow', label: 'Allow' }]) + }) + expect(bindings[0]).toEqual([ + 'orca:claude-prompt%3Aorca-session%3Apermission-1', + 'permission-1' + ]) + + const questions = prompt({ + requestId: 'questions-1', + promptKey: 'questions-1', + toolUseId: 'tool-q', + toolName: 'AskUserQuestion', + kind: 'question', + input: { + questions: [ + { question: 'Library?', options: [{ label: 'Luxon' }] }, + { question: 'Ship?', options: [{ label: 'Yes' }] } + ] + }, + questionIds: ['Library?', 'Ship?'] + }) + translator.handle({ type: 'prompt', sessionId: 'orca-session', prompt: questions }) + expect(state.items.filter((item) => item.body.kind === 'question')).toHaveLength(1) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'question', + questions: [ + { id: 'q1', question: 'Library?', multiSelect: false }, + { id: 'q2', question: 'Ship?', multiSelect: false } + ] + }) + expect(bindings.at(-1)).toEqual([ + 'orca:claude-prompt%3Aorca-session%3Aquestions-1', + 'questions-1' + ]) + + const multiSelect = prompt({ + requestId: 'questions-multi', + promptKey: 'questions-multi', + toolUseId: 'tool-multi', + toolName: 'AskUserQuestion', + kind: 'question', + input: { + questions: [ + { + question: 'Libraries?', + multiSelect: true, + options: [{ label: 'Luxon' }, { label: 'Temporal' }] + } + ] + }, + questionIds: ['Libraries?'] + }) + translator.handle({ type: 'prompt', sessionId: 'orca-session', prompt: multiSelect }) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'question', + question: '1 grouped question from Claude', + options: [], + questions: [ + { + id: 'q1', + question: 'Libraries?', + multiSelect: true, + options: [{ label: 'Luxon' }, { label: 'Temporal' }], + freeTextQuestionId: 'q1' + } + ] + }) + + translator.handle({ + type: 'prompt-cancelled', + sessionId: 'orca-session', + promptKey: 'questions-1' + }) + expect(state.tombstones).toHaveLength(1) + }) +}) + +function prompt( + input: Pick< + ClaudePendingPrompt, + 'requestId' | 'promptKey' | 'toolUseId' | 'toolName' | 'kind' | 'input' | 'questionIds' + > +): ClaudePendingPrompt { + return { + ...input, + suggestions: [], + answers: new Map(), + settle: () => {} + } +} diff --git a/src/main/claude/claude-structured-journal-translation.ts b/src/main/claude/claude-structured-journal-translation.ts new file mode 100644 index 00000000000..ffaad4da570 --- /dev/null +++ b/src/main/claude/claude-structured-journal-translation.ts @@ -0,0 +1,287 @@ +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import type { AgentSessionDeltaCoalescerDeps } from '../native-chat/agent-session-wire/agent-session-delta-coalescer' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' +import type { ClaudeStructuredSessionEvent } from './claude-structured-session-state' +import { + claudeMessageBody, + claudeMessageIdentity, + claudeHasReplayContent, + claudeRecord, + claudeStreamingMessageBody, + claudeText, + claudeThinkingIdentity, + claudeThinkingText, + claudeToolBody, + claudeToolIdentity, + claudeToolResults, + claudeToolUses, + readClaudeMessageEnvelope, + type ClaudeToolUse +} from './claude-structured-item-translation' +import { + claudeApprovalItem, + claudePromptIdentity, + claudeQuestionItems +} from './claude-structured-prompt-items' +import type { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import { readableProviderFrameText } from '../native-chat/agent-session-wire/unhandled-provider-frame' +import { + CLAUDE_UNRENDERABLE_CONTENT_TEXT, + claudeProviderFrameKind, + claudeResultFailure, + createClaudeProviderFrameFallback, + isModeledClaudeContent, + isSettledClaudeResultKind +} from './claude-structured-provider-fallback' +import { createClaudeStreamedBlockRegistry } from './claude-streamed-block-identity' +import { createClaudeStreamedTextCheckpoints } from './claude-streamed-text-checkpoints' + +export type ClaudeJournalTranslatorDeps = { + sink: StructuredAgentSessionEventSink + bindPromptItemId?: (journalItemId: string, promptKey: string, questionId?: string) => void + coalesceMs?: number + schedule?: AgentSessionDeltaCoalescerDeps['schedule'] + fallbackIdPrefix?: string +} + +export type ClaudeJournalTranslator = { + handle: (event: ClaudeStructuredSessionEvent) => void + flush: () => void + /** Streamed blocks still awaiting a final frame. A settled turn leaves none. */ + readonly pendingStreamedBlocks: number + dispose: () => void +} + +export function createClaudeSessionJournalTranslator( + sink: StructuredAgentSessionEventSink | undefined, + prompts: ClaudePromptRegistry, + fallbackIdPrefix: string +): ClaudeJournalTranslator | null { + return sink + ? createClaudeJournalTranslator({ + sink, + fallbackIdPrefix, + bindPromptItemId: (itemId, promptKey, questionId) => + prompts.bindJournalItemId(itemId, promptKey, questionId) + }) + : null +} + +function lifecycleIdentity(sessionId: string, turnId: string): AgentJournalItemIdentity { + return { + provider: 'legacy', + agent: 'claude', + sessionId, + recordId: `turn-lifecycle:${turnId}` + } +} + +export function createClaudeJournalTranslator( + deps: ClaudeJournalTranslatorDeps +): ClaudeJournalTranslator { + const tools = new Map() + const promptItems = new Map() + const streamedBlocks = createClaudeStreamedBlockRegistry() + let currentTurn: { sessionId: string; turnId: string } | null = null + const providerFallback = createClaudeProviderFrameFallback( + deps.sink, + deps.fallbackIdPrefix ?? 'acquisition' + ) + const streamedText = createClaudeStreamedTextCheckpoints({ + ...(deps.coalesceMs === undefined ? {} : { coalesceMs: deps.coalesceMs }), + ...(deps.schedule ? { schedule: deps.schedule } : {}), + persist: (identity, text) => { + deps.sink.appendItem(identity, claudeStreamingMessageBody(text)) + deps.sink.publish() + } + }) + + const publishLifecycle = (sessionId: string, turnId: string, running: boolean): void => { + const identity = lifecycleIdentity(sessionId, turnId) + if (running) { + deps.sink.appendItem(identity, { + kind: 'status', + text: 'Claude is working…', + turnLifecycle: { turnId, state: 'running' } + }) + } else { + deps.sink.appendTombstone(identity) + } + deps.sink.publish() + } + + const handleStream = (message: Record): boolean => { + const delta = streamedBlocks.observe(message) + if (!delta) { + return false + } + streamedText.append(delta.identity, delta.text) + return true + } + + const handleMessage = (message: Record, startsTurn: boolean): boolean => { + const envelope = readClaudeMessageEnvelope(message) + if (!envelope) { + return false + } + let changed = false + const body = claudeMessageBody(envelope) + // The final frame of a streamed block lands on the block's identity, not its own uuid. + const identity = + (body && envelope.role === 'assistant' ? streamedBlocks.reconcile(envelope) : null) ?? + claudeMessageIdentity(envelope) + streamedText.forget(agentJournalItemKey(identity)) + if (body) { + deps.sink.appendItem(identity, body) + changed = true + } + for (const tool of claudeToolUses(envelope)) { + tools.set(tool.id, tool) + deps.sink.appendItem( + claudeToolIdentity(envelope.sessionId, tool.id), + claudeToolBody({ tool }) + ) + changed = true + } + for (const result of claudeToolResults(envelope)) { + const tool = tools.get(result.toolUseId) ?? { + id: result.toolUseId, + name: 'tool', + input: null + } + deps.sink.appendItem( + claudeToolIdentity(envelope.sessionId, result.toolUseId), + claudeToolBody({ tool, result }) + ) + // Tool inputs are only needed until their matching result arrives. + tools.delete(result.toolUseId) + changed = true + } + const thinking = claudeThinkingText(envelope) + if (thinking) { + deps.sink.appendItem(claudeThinkingIdentity(envelope.sessionId, envelope.uuid), { + kind: 'status', + text: boundInlineText(thinking, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text + }) + changed = true + } + const unhandledContent = envelope.content.filter((part) => !isModeledClaudeContent(part)) + for (const part of unhandledContent) { + const partType = claudeText(claudeRecord(part)?.type) ?? 'unknown' + providerFallback.append( + `message:${envelope.role}:content:${partType}`, + part, + readableProviderFrameText(part) ?? CLAUDE_UNRENDERABLE_CONTENT_TEXT + ) + changed = true + } + // An empty user frame is a replay with nothing to show, not an unknown kind. + if (envelope.content.length === 0 && envelope.role === 'assistant') { + providerFallback.append(`message:${envelope.role}:empty`, message) + changed = true + } + if ( + envelope.role === 'user' && + startsTurn && + claudeHasReplayContent(envelope) && + message.parent_tool_use_id === null + ) { + if (currentTurn) { + publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) + } + currentTurn = { sessionId: envelope.sessionId, turnId: envelope.uuid } + publishLifecycle(envelope.sessionId, envelope.uuid, true) + } + if (changed) { + deps.sink.publish() + } + return true + } + + const handlePrompt = (event: Extract): void => { + const identities: AgentJournalItemIdentity[] = [] + if (event.prompt.kind === 'question') { + for (const question of claudeQuestionItems({ + sessionId: event.sessionId, + prompt: event.prompt + })) { + identities.push(question.identity) + deps.sink.appendItem(question.identity, question.body) + deps.bindPromptItemId?.(agentJournalItemKey(question.identity), event.prompt.promptKey) + } + } else { + const identity = claudePromptIdentity({ + sessionId: event.sessionId, + promptKey: event.prompt.promptKey + }) + identities.push(identity) + deps.sink.appendItem(identity, claudeApprovalItem(event.prompt)) + deps.bindPromptItemId?.(agentJournalItemKey(identity), event.prompt.promptKey) + } + promptItems.set(event.prompt.promptKey, identities) + deps.sink.publish() + } + + return { + handle: (event) => { + if (event.type === 'ended') { + streamedText.flush() + if (currentTurn) { + publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) + currentTurn = null + } + return + } + if (event.type === 'message' && handleStream(event.message)) { + return + } + streamedText.flush() + if (event.type === 'prompt') { + handlePrompt(event) + } else if (event.type === 'prompt-cancelled') { + for (const identity of promptItems.get(event.promptKey) ?? []) { + deps.sink.appendTombstone(identity) + } + promptItems.delete(event.promptKey) + deps.sink.publish() + } else if (event.type === 'message' && event.message.type === 'result') { + if (currentTurn) { + publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) + currentTurn = null + } + // The turn is over. A block still awaiting its final keeps the text the + // flush above journaled, but its live state goes: an interrupted turn + // would otherwise retain that text for the life of the session. + streamedBlocks.clear() + streamedText.settle() + const kind = claudeProviderFrameKind(event.message) + // Ordinary turn bookkeeping stays suppressed; a reported failure never does. + const failure = claudeResultFailure(event.message) + if (failure || !isSettledClaudeResultKind(kind)) { + providerFallback.append(kind, event.message, failure?.text) + } + } else if (event.type === 'message') { + if (!handleMessage(event.message, event.startsTurn === true)) { + providerFallback.append(claudeProviderFrameKind(event.message), event.message) + } + } else if (event.type === 'provider-frame') { + providerFallback.append(event.kind, event.payload) + } + }, + flush: streamedText.flush, + get pendingStreamedBlocks() { + return streamedText.pending + }, + dispose: () => { + streamedText.dispose() + tools.clear() + promptItems.clear() + streamedBlocks.clear() + } + } +} diff --git a/src/main/claude/claude-structured-launch-resolution.test.ts b/src/main/claude/claude-structured-launch-resolution.test.ts new file mode 100644 index 00000000000..650947cffa1 --- /dev/null +++ b/src/main/claude/claude-structured-launch-resolution.test.ts @@ -0,0 +1,392 @@ +import { chmodSync, mkdtempSync, mkdirSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { delimiter, join } from 'node:path' +import { describe, expect, it } from 'vitest' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import type { AgentSessionRecordStore } from '../runtime/agent-session-record-store' +import { AgentSessionPreSpawnError } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { claudeStructuredAuthPolicyForSettings } from '../claude-accounts/claude-structured-auth-policy' +import type { ClaudeManagedAccountGateSettings } from '../native-chat/claude-structured-managed-account-support' +import { + CLAUDE_DEFAULT_SETTING_SOURCES, + CLAUDE_STRUCTURED_BASE_OPTIONS, + claudeSdkOptionsForLaunchArgs, + claudeSessionIdForOrcaSession, + createClaudeStructuredLaunchResolver +} from './claude-structured-launch-resolution' + +const SESSION_ID = 'orca-session-1' +const IDENTITY = { sessionId: SESSION_ID } as Parameters< + ReturnType +>[0]['identity'] + +function record(overrides: Partial = {}): AgentSessionRecord { + return { + sessionId: SESSION_ID, + provider: 'claude', + location: { + executionHostId: LOCAL_EXECUTION_HOST_ID, + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'folder' + }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/home/work/.claude' }, + providerHandleChain: [], + ...overrides + } as AgentSessionRecord +} + +function identityAt(leafUuid: string | null): typeof IDENTITY { + return { + ...IDENTITY, + providerHandle: { kind: 'claude', sessionId: 'provider-current', leafUuid } + } +} + +function makeExecutable(path: string): void { + mkdirSync(join(path, '..'), { recursive: true }) + writeFileSync(path, '') + if (process.platform !== 'win32') { + chmodSync(path, 0o755) + } +} + +function resolverFor( + value: AgentSessionRecord | null, + resolveEnv?: () => Record, + stripAuthEnv = false +) { + return createClaudeStructuredLaunchResolver({ + store: { getRecord: () => value } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv }), + ...(resolveEnv ? { resolveEnv } : {}) + }) +} + +function managedAccount(id: string, managedAuthRuntime: 'host' | 'wsl') { + return { + id, + email: `${id}@example.com`, + managedAuthPath: `/managed/${id}`, + managedAuthRuntime, + authMethod: 'subscription-oauth' as const, + createdAt: 0, + updatedAt: 0, + lastAuthenticatedAt: 0 + } +} + +const HOST_SELECTED: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('host-1', 'host')], + activeClaudeManagedAccountId: 'host-1', + activeClaudeManagedAccountIdsByRuntime: { host: 'host-1', wsl: {} } +} + +/** The normalized steady state of a Windows user whose only Claude account is WSL-managed: the + * prune drops the WSL account out of the host slot and persists that. */ +const WSL_ONLY_NORMALIZED: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('wsl-1', 'wsl')], + activeClaudeManagedAccountId: null, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-1' } } +} + +const RESUMABLE = record({ + providerHandleChain: [ + { handle: { provider: 'claude', sessionId: 'provider-current', leafUuid: 'leaf-current' } } + ] as AgentSessionRecord['providerHandleChain'] +}) + +describe('claude structured launch resolution', () => { + it('pre-mints a stable provider id and pins interactive setting sources', async () => { + const first = await resolverFor(record())({ identity: IDENTITY }) + const second = await resolverFor(record())({ identity: IDENTITY }) + + expect(first.providerSessionId).toBe(claudeSessionIdForOrcaSession(SESSION_ID)) + expect(second.providerSessionId).toBe(first.providerSessionId) + expect(first).toMatchObject({ + pathToClaudeCodeExecutable: '/usr/local/bin/claude', + cwd: '/repos/workspace-1', + claudeConfigDir: '/home/work/.claude', + resumeLeafUuid: null, + resumed: false + }) + expect(first.options).toEqual({ + includePartialMessages: true, + settingSources: [...CLAUDE_DEFAULT_SETTING_SOURCES], + supportedDialogKinds: [], + extraArgs: { 'replay-user-messages': null }, + systemPrompt: { type: 'preset', preset: 'claude_code' }, + sessionId: first.providerSessionId + }) + expect(first.options.resume).toBeUndefined() + expect(CLAUDE_STRUCTURED_BASE_OPTIONS.includePartialMessages).toBe(true) + }) + + it('resumes the session and leaf at the durable chain head', async () => { + const launch = await resolverFor( + record({ + providerHandleChain: [ + { handle: { provider: 'claude', sessionId: 'provider-old', leafUuid: 'leaf-old' } }, + { + handle: { + provider: 'claude', + sessionId: 'provider-current', + leafUuid: 'leaf-current' + } + } + ] as AgentSessionRecord['providerHandleChain'] + }) + )({ identity: identityAt('leaf-current') }) + + expect(launch).toMatchObject({ + providerSessionId: 'provider-current', + resumeLeafUuid: 'leaf-current', + resumed: true + }) + expect(launch.options.resume).toBe('provider-current') + expect(launch.options.resumeSessionAt).toBe('leaf-current') + expect(launch.options.sessionId).toBeUndefined() + }) + + it('refuses a durable journal leaf that diverged before resume resolution', async () => { + const resolve = resolverFor( + record({ + providerHandleChain: [ + { + handle: { + provider: 'claude', + sessionId: 'provider-current', + leafUuid: 'leaf-current' + } + } + ] as AgentSessionRecord['providerHandleChain'] + }) + ) + + await expect(resolve({ identity: identityAt('leaf-stale') })).rejects.toThrow( + 'durable resume identity changed before spawn' + ) + }) + + it('keeps session-only resume when the durable handle has no leaf', async () => { + const launch = await resolverFor( + record({ + providerHandleChain: [ + { + handle: { + provider: 'claude', + sessionId: 'provider-current', + leafUuid: null + } + } + ] as AgentSessionRecord['providerHandleChain'] + }) + )({ identity: identityAt(null) }) + + expect(launch.options.resume).toBe('provider-current') + expect(launch.options.resumeSessionAt).toBeUndefined() + }) + + it('preserves durable Claude launch arguments as typed options and extraArgs', async () => { + const launch = await resolverFor( + record({ + launchArgs: [ + '--model', + 'claude-sonnet-4-5', + '--effort', + 'high', + '--dangerously-skip-permissions' + ] + }) + )({ identity: IDENTITY }) + + expect(launch.options.model).toBe('claude-sonnet-4-5') + expect(launch.options.effort).toBe('high') + expect(launch.options.extraArgs).toEqual({ + 'dangerously-skip-permissions': null, + 'replay-user-messages': null + }) + }) + + it('routes durable launch arguments to a typed option first and refuses what neither can carry', () => { + // The catalog's own output: each flag lands in exactly one place, so the SDK + // cannot emit it twice with two different values. + expect(claudeSdkOptionsForLaunchArgs(['--model', 'opus', '--effort', 'xhigh'])).toEqual({ + model: 'opus', + effort: 'xhigh' + }) + // An effort the SDK's union does not name still reaches the CLI, unchanged. + expect(claudeSdkOptionsForLaunchArgs(['--effort', 'ultra'])).toEqual({ + extraArgs: { effort: 'ultra' } + }) + expect(claudeSdkOptionsForLaunchArgs(['--settings=/tmp/s.json'])).toEqual({ + extraArgs: { settings: '/tmp/s.json' } + }) + expect(() => claudeSdkOptionsForLaunchArgs(['-m', 'opus'])).toThrow(/no SDK option/) + }) + + it('keeps the session launch environment pinned after account settings change', async () => { + const resolver = resolverFor(record(), () => ({ + ANTHROPIC_AUTH_TOKEN: 'rotated-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test' + })) + + expect((await resolver({ identity: IDENTITY })).env).toMatchObject({ + ANTHROPIC_AUTH_TOKEN: 'rotated-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test' + }) + expect((await resolver({ identity: IDENTITY })).env?.ANTHROPIC_AUTH_TOKEN).toBe('rotated-token') + }) + + // Stripping is the managed-account rule the terminal preflight computes at + // runtime-auth-preparation.ts:72; claude-structured-auth-parity.test.ts covers + // the system-auth half, where the user's own key has to survive. + it('strips ambient Anthropic auth under a managed account but keeps the rest of the env', async () => { + const restore = { + ANTHROPIC_API_KEY: process.env.ANTHROPIC_API_KEY, + ANTHROPIC_AUTH_TOKEN: process.env.ANTHROPIC_AUTH_TOKEN, + CLAUDE_CODE_OAUTH_TOKEN: process.env.CLAUDE_CODE_OAUTH_TOKEN, + ORCA_LAUNCH_RESOLUTION_MARKER: process.env.ORCA_LAUNCH_RESOLUTION_MARKER + } + process.env.ANTHROPIC_API_KEY = 'sk-ant-SHELL-LEAK' + process.env.ANTHROPIC_AUTH_TOKEN = 'tok-SHELL-LEAK' + process.env.CLAUDE_CODE_OAUTH_TOKEN = 'oauth-SHELL-LEAK' + process.env.ORCA_LAUNCH_RESOLUTION_MARKER = 'inherited' + try { + const launch = await resolverFor(record(), undefined, true)({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBeUndefined() + expect(launch.env?.ANTHROPIC_AUTH_TOKEN).toBeUndefined() + expect(launch.env?.CLAUDE_CODE_OAUTH_TOKEN).toBeUndefined() + // The inherited env is still the base — only auth is removed from it. + expect(launch.env?.ORCA_LAUNCH_RESOLUTION_MARKER).toBe('inherited') + expect(launch.env?.PATH ?? launch.env?.Path).toBeTruthy() + } finally { + for (const [key, value] of Object.entries(restore)) { + if (value === undefined) { + delete process.env[key] + } else { + process.env[key] = value + } + } + } + }) + + it('lets an explicit Claude env overlay override ambient auth under system auth', async () => { + const restore = process.env.ANTHROPIC_API_KEY + process.env.ANTHROPIC_API_KEY = 'sk-ant-SHELL-LEAK' + try { + const launch = await resolverFor(record(), () => ({ + ANTHROPIC_API_KEY: 'sk-ant-CONFIGURED' + }))({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBe('sk-ant-CONFIGURED') + } finally { + if (restore === undefined) { + delete process.env.ANTHROPIC_API_KEY + } else { + process.env.ANTHROPIC_API_KEY = restore + } + } + }) + + it('pairs a resolved Claude CLI with its sibling Node runtime', async () => { + const root = mkdtempSync(join(tmpdir(), 'orca-claude-launch-')) + const binDir = join(root, 'bin') + const claudeCommand = join(binDir, process.platform === 'win32' ? 'claude.cmd' : 'claude') + const nodeCommand = join(binDir, process.platform === 'win32' ? 'node.cmd' : 'node') + makeExecutable(claudeCommand) + makeExecutable(nodeCommand) + + const launch = await createClaudeStructuredLaunchResolver({ + store: { getRecord: () => record() } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => claudeCommand, + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + resolveEnv: () => ({ + PATH: '/usr/bin', + CLAUDE_CONFIG_DIR: '/accounts/selected/home' + }) + })({ identity: IDENTITY }) + + expect((launch.env?.PATH ?? launch.env?.Path)?.split(delimiter)[0]).toBe(binDir) + }) + + it('refuses other hosts, WSL, providers, and account-home variables', async () => { + await expect( + resolverFor(record({ location: { ...record().location, executionHostId: 'ssh:build' } }))({ + identity: IDENTITY + }) + ).rejects.toThrow(/local host/) + await expect( + resolverFor(record({ location: { ...record().location, wslDistro: 'Ubuntu' } }))({ + identity: IDENTITY + }) + ).rejects.toThrow(/local host/) + await expect( + resolverFor(record({ provider: 'codex' } as Partial))({ + identity: IDENTITY + }) + ).rejects.toThrow(/codex session/) + await expect( + resolverFor(record({ accountHome: { variable: 'CODEX_HOME', path: '/tmp/codex' } }))({ + identity: IDENTITY + }) + ).rejects.toThrow(/CLAUDE_CONFIG_DIR/) + }) + + /** The account state can change while a session lives, and a reacquire after an unexpected child + * exit re-resolves the launch. Without the gate here, that reacquire spawns under whatever the + * account state has become. */ + describe('managed-account gate on every acquisition', () => { + function resolverWithGate(read: () => ClaudeManagedAccountGateSettings | null) { + return createClaudeStructuredLaunchResolver({ + store: { getRecord: () => RESUMABLE } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => '/usr/local/bin/claude', + // Derived, not a literal: the gate and the policy must read the SAME account state, so a + // hardcoded value could assert a pairing production cannot produce. + resolveAuthPolicy: () => { + const settings = read() + if (!settings) { + throw new Error('the gate refuses before the auth policy is computed') + } + return claudeStructuredAuthPolicyForSettings(settings) + }, + readManagedAccountGate: read + }) + } + + it('refuses a reacquire once the account state becomes the refused shape', async () => { + let gate: ClaudeManagedAccountGateSettings | null = HOST_SELECTED + const resolve = resolverWithGate(() => gate) + + // Created while supported: the launch resolves and would spawn. + await expect(resolve({ identity: identityAt('leaf-current') })).resolves.toMatchObject({ + providerSessionId: 'provider-current' + }) + + gate = WSL_ONLY_NORMALIZED + + // Reacquire after the account state changed: refused before anything spawns. + await expect(resolve({ identity: identityAt('leaf-current') })).rejects.toBeInstanceOf( + AgentSessionPreSpawnError + ) + }) + + it('fails closed when the account state cannot be read', async () => { + await expect( + resolverWithGate(() => null)({ identity: identityAt('leaf-current') }) + ).rejects.toBeInstanceOf(AgentSessionPreSpawnError) + }) + + it('keeps resolving when no gate is wired, so other embedders are unaffected', async () => { + await expect( + resolverFor(RESUMABLE)({ identity: identityAt('leaf-current') }) + ).resolves.toMatchObject({ providerSessionId: 'provider-current' }) + }) + }) +}) diff --git a/src/main/claude/claude-structured-launch-resolution.ts b/src/main/claude/claude-structured-launch-resolution.ts new file mode 100644 index 00000000000..4f28f14ad65 --- /dev/null +++ b/src/main/claude/claude-structured-launch-resolution.ts @@ -0,0 +1,273 @@ +import { createHash } from 'node:crypto' +import type { EffortLevel, Options as ClaudeAgentSdkOptions } from '@anthropic-ai/claude-agent-sdk' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import { agentSessionProviderHandleChainHead } from '../../shared/agent-session-provider-handle' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import { withCliRuntimeOnPath } from '../../shared/node-cli-command-resolution' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE, + applyClaudeEnvPatch, + hasClaudeAuthEnvConflict +} from '../claude-accounts/environment' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { + CLAUDE_AUTH_SWITCH_SETTLE_TIMEOUT_MS, + whenClaudeAuthSwitchSettles +} from '../claude-accounts/live-pty-gate' +import { AgentSessionPreSpawnError } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { + structuredClaudeMatchesActiveManagedAccount, + type ClaudeManagedAccountGateSettings +} from '../native-chat/claude-structured-managed-account-support' +import { resolveClaudeCommand } from '../codex-cli/command' +import type { AgentSessionRecordStore } from '../runtime/agent-session-record-store' + +export const CLAUDE_DEFAULT_SETTING_SOURCES = ['user', 'project', 'local'] as const + +export type ClaudeStructuredSdkOptions = Pick< + ClaudeAgentSdkOptions, + | 'includePartialMessages' + | 'systemPrompt' + | 'settingSources' + | 'supportedDialogKinds' + | 'extraArgs' + | 'model' + | 'effort' + | 'sessionId' + | 'resume' + | 'resumeSessionAt' +> + +/** + * The options translation of the flags this transport used to build by hand. + * + * `-p`, `--input-format`, `--output-format` and `--verbose` are implied by + * `query()`; `--permission-prompt-tool stdio` is emitted because a `canUseTool` + * callback is supplied. `--replay-user-messages` has no option — the SDK never + * emits it — and Orca's send acknowledgement depends on the replay. + */ +export const CLAUDE_STRUCTURED_BASE_OPTIONS: ClaudeStructuredSdkOptions = { + includePartialMessages: true, + // Keep the SDK on Claude Code's own system-prompt contract. + systemPrompt: { type: 'preset', preset: 'claude_code' }, + settingSources: [...CLAUDE_DEFAULT_SETTING_SOURCES], + supportedDialogKinds: [], + extraArgs: { 'replay-user-messages': null } +} + +const EFFORT_LEVELS: readonly string[] = ['low', 'medium', 'high', 'xhigh', 'max'] + +function cloneDefinedEnv(env: NodeJS.ProcessEnv | Record): Record { + const next: Record = {} + for (const [key, value] of Object.entries(env)) { + if (value !== undefined) { + next[key] = value + } + } + return next +} + +/** + * Translate the record's durable launch arguments into SDK options. + * + * Typed option first so a flag is never emitted twice; `extraArgs` carries + * anything without one. A token expressible neither way is refused rather than + * dropped — a silent drop is how this lane loses launch flags. + */ +export function claudeSdkOptionsForLaunchArgs( + args: readonly string[] +): Pick { + let model: string | undefined + let effort: EffortLevel | undefined + const extraArgs: Record = {} + for (let index = 0; index < args.length; index += 1) { + const token = args[index] ?? '' + if (!token.startsWith('--') || token.length <= 2) { + throw new Error( + `claude launch argument ${token} has no SDK option; refusing rather than dropping it` + ) + } + const equals = token.indexOf('=') + const flag = equals === -1 ? token : token.slice(0, equals) + let value = equals === -1 ? null : token.slice(equals + 1) + if (value === null) { + const next = args[index + 1] + if (next !== undefined && !next.startsWith('-')) { + value = next + index += 1 + } + } + if (flag === '--model' && value !== null) { + model = value + } else if (flag === '--effort' && value !== null && EFFORT_LEVELS.includes(value)) { + effort = value as EffortLevel + } else { + extraArgs[flag.slice(2)] = value + } + } + return { + ...(model === undefined ? {} : { model }), + ...(effort === undefined ? {} : { effort }), + ...(Object.keys(extraArgs).length > 0 ? { extraArgs } : {}) + } +} + +export type ClaudeStructuredLaunch = { + /** Always Orca's resolved user CLI: the SDK's bundled binaries are excluded from the install. */ + pathToClaudeCodeExecutable: string + options: ClaudeStructuredSdkOptions + cwd: string + env?: Record + claudeConfigDir: string + providerSessionId: string + resumeLeafUuid: string | null + resumed: boolean +} + +export type ClaudeStructuredLaunchResolverDeps = { + store: AgentSessionRecordStore + resolveWorkspacePath: (workspaceId: string) => Promise + resolveCommand?: () => string + resolveEnv?: () => + | Promise | undefined> + | Record + | undefined + /** + * Required, and deliberately not defaulted. `stripAuthEnv` used to be a literal + * `true` here, so a missing dependency could not under-strip. Now it can, and the + * failure is silent — so every caller states the account's policy rather than + * inherit a guess. Build it with claudeStructuredAuthPolicyForSettings. + */ + resolveAuthPolicy: () => Promise | ClaudeStructuredAuthPolicy + /** How long an in-flight account switch may hold a launch before it is refused. */ + authSwitchSettleTimeoutMs?: number + /** Account state for the managed-account gate; null when it cannot be read, which refuses. */ + readManagedAccountGate?: () => ClaudeManagedAccountGateSettings | null +} + +/** + * Wait a running account switch out, and refuse only if it never settles. + * + * Launch resolution is reached from `acquireClaudeSession` *after* the old child has + * been closed and proved, so a plain refusal here would leave the user with a dead + * chat and no replacement — the very harm the acquire-entry guard exists to prevent. + * The entry guard still refuses outright, because nothing has been torn down yet. + */ +export async function assertClaudeAuthSwitchSettled( + timeoutMs = CLAUDE_AUTH_SWITCH_SETTLE_TIMEOUT_MS +): Promise { + if (!(await whenClaudeAuthSwitchSettles(timeoutMs))) { + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) + } +} + +export function claudeSessionIdForOrcaSession(sessionId: string): string { + const bytes = createHash('sha256').update(`orca-claude:${sessionId}`).digest().subarray(0, 16) + bytes[6] = ((bytes[6] ?? 0) & 0x0f) | 0x40 + bytes[8] = ((bytes[8] ?? 0) & 0x3f) | 0x80 + const hex = bytes.toString('hex') + return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20)}` +} + +export function createClaudeStructuredLaunchResolver( + deps: ClaudeStructuredLaunchResolverDeps +): (input: { identity: AgentSessionJournalIdentity }) => Promise { + return async ({ identity }) => { + await assertClaudeAuthSwitchSettled(deps.authSwitchSettleTimeoutMs) + const record = deps.store.getRecord(identity.sessionId) + if (!record) { + throw new Error(`no durable agent-session record for ${identity.sessionId}`) + } + if (record.provider !== 'claude') { + throw new Error(`session ${identity.sessionId} is a ${record.provider} session`) + } + if ( + record.location.executionHostId !== LOCAL_EXECUTION_HOST_ID || + record.location.wslDistro !== null + ) { + throw new Error( + `claude structured sessions run on the local host, not ${record.location.executionHostId}` + ) + } + if (record.accountHome.variable !== 'CLAUDE_CONFIG_DIR') { + throw new Error(`claude sessions pin CLAUDE_CONFIG_DIR, not ${record.accountHome.variable}`) + } + // Every acquisition, not just the first: the account state can change under a live session, and + // a reacquire after an unexpected exit would otherwise spawn under whatever it has become. + // Codex has no gate here — it resolves its account on a different path. + if ( + deps.readManagedAccountGate && + !structuredClaudeMatchesActiveManagedAccount(deps.readManagedAccountGate()) + ) { + throw new AgentSessionPreSpawnError( + 'structured Claude is not offered under the active managed Claude account' + ) + } + const head = agentSessionProviderHandleChainHead(record.providerHandleChain) + if ( + head?.handle.provider === 'claude' && + (identity.providerHandle.kind !== 'claude' || + identity.providerHandle.sessionId !== head.handle.sessionId || + identity.providerHandle.leafUuid !== head.handle.leafUuid) + ) { + throw new Error('claude durable resume identity changed before spawn') + } + const providerSessionId = + head?.handle.provider === 'claude' + ? head.handle.sessionId + : claudeSessionIdForOrcaSession(identity.sessionId) + const durable = claudeSdkOptionsForLaunchArgs(record.launchArgs ?? []) + const command = (deps.resolveCommand ?? resolveClaudeCommand)() + const auth = await deps.resolveAuthPolicy() + const overlay = await deps.resolveEnv?.() + // A switch can begin while the policy and overlay resolve, exactly as it can + // during the terminal preflight's prepareClaudeAuth — recheck after the awaits. + await assertClaudeAuthSwitchSettled(deps.authSwitchSettleTimeoutMs) + // Under a managed account the pinned credential is the only auth this launch may + // use, so an explicit override is refused rather than silently beating the pin. + if (auth.stripAuthEnv && hasClaudeAuthEnvConflict(overlay)) { + throw new Error(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + } + // Why the overlay merges onto the inherited env rather than replacing it: the child + // still needs PATH and the rest of the shell environment, and withCliRuntimeOnPath + // derives PATH from what it is handed. Ambient Anthropic auth is stripped from the + // inherited half only when a managed account owns the credential; a system-auth + // user's own key is their sign-in and must reach the child. + const env = withCliRuntimeOnPath( + command, + { + ...applyClaudeEnvPatch( + cloneDefinedEnv(process.env), + {}, + { + stripAuthEnv: auth.stripAuthEnv, + platform: process.platform + } + ), + ...(overlay ? cloneDefinedEnv(overlay) : {}) + }, + { platform: process.platform } + ) + return { + pathToClaudeCodeExecutable: command, + options: { + ...durable, + ...CLAUDE_STRUCTURED_BASE_OPTIONS, + extraArgs: { ...durable.extraArgs, ...CLAUDE_STRUCTURED_BASE_OPTIONS.extraArgs }, + ...(head?.handle.provider === 'claude' + ? { + resume: providerSessionId, + ...(head.handle.leafUuid === null ? {} : { resumeSessionAt: head.handle.leafUuid }) + } + : { sessionId: providerSessionId }) + }, + cwd: await deps.resolveWorkspacePath(record.location.workspaceId), + env, + claudeConfigDir: record.accountHome.path, + providerSessionId, + resumeLeafUuid: head?.handle.provider === 'claude' ? head.handle.leafUuid : null, + resumed: head?.handle.provider === 'claude' + } + } +} diff --git a/src/main/claude/claude-structured-location-support.test.ts b/src/main/claude/claude-structured-location-support.test.ts new file mode 100644 index 00000000000..1106667d544 --- /dev/null +++ b/src/main/claude/claude-structured-location-support.test.ts @@ -0,0 +1,91 @@ +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { + __setWindowsProcessTreeLoaderForTests, + resetWindowsProcessTableForTests +} from '../windows/windows-process-table' +import { supportsClaudeStructuredLocation } from './claude-structured-location-support' + +function setPlatform(platform: NodeJS.Platform): PropertyDescriptor | undefined { + const previous = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: platform }) + return previous +} + +describe('supportsClaudeStructuredLocation', () => { + let previousPlatform: PropertyDescriptor | undefined + + beforeEach(() => { + previousPlatform = setPlatform('darwin') + __setWindowsProcessTreeLoaderForTests() + }) + + afterEach(() => { + __setWindowsProcessTreeLoaderForTests() + resetWindowsProcessTableForTests() + if (previousPlatform) { + Object.defineProperty(process, 'platform', previousPlatform) + } + }) + + it('allows local non-WSL locations on macOS and Linux', () => { + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(true) + }) + + it('rejects Windows local locations until creation-time proof is available', () => { + previousPlatform = setPlatform('win32') + __setWindowsProcessTreeLoaderForTests(() => ({ + ProcessDataFlag: { None: 0, Memory: 1, CommandLine: 2 }, + getAllProcesses: () => undefined + })) + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(false) + }) + + it('accepts Windows local locations once creation-time proof is available', () => { + previousPlatform = setPlatform('win32') + __setWindowsProcessTreeLoaderForTests(() => ({ + ProcessDataFlag: { None: 0, Memory: 1, CommandLine: 2, CreationTime: 4 }, + getAllProcesses: () => undefined + })) + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(true) + }) + + it('rejects WSL and remote locations', () => { + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'local', + wslDistro: 'Ubuntu', + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(false) + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'runtime:env-1', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(false) + }) +}) diff --git a/src/main/claude/claude-structured-location-support.ts b/src/main/claude/claude-structured-location-support.ts new file mode 100644 index 00000000000..9c784a91325 --- /dev/null +++ b/src/main/claude/claude-structured-location-support.ts @@ -0,0 +1,11 @@ +import type { AgentSessionExecutionLocation } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import { isWindowsProcessStartTimeAvailable } from '../windows/windows-process-table' + +export function supportsClaudeStructuredLocation(location: AgentSessionExecutionLocation): boolean { + return ( + location.executionHostId === LOCAL_EXECUTION_HOST_ID && + location.wslDistro === null && + (process.platform !== 'win32' || isWindowsProcessStartTimeAvailable()) + ) +} diff --git a/src/main/claude/claude-structured-model-confirmation.test.ts b/src/main/claude/claude-structured-model-confirmation.test.ts new file mode 100644 index 00000000000..7bd8f629119 --- /dev/null +++ b/src/main/claude/claude-structured-model-confirmation.test.ts @@ -0,0 +1,209 @@ +import { describe, expect, it } from 'vitest' +import { setClaudeStructuredOption } from './claude-structured-options' +import type { ClaudeSession } from './claude-structured-session-state' +import { PROVIDER_SESSION_ID, acquired, fakeClaude } from './claude-structured-session-test-support' + +/** Verbatim rows from Claude Code 2.1.258's list_models response. */ +const CATALOG = [ + { + value: 'default', + resolvedModel: 'claude-opus-5[1m]', + displayName: 'Default (recommended)', + supportsEffort: true, + supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'] + }, + { + value: 'sonnet', + resolvedModel: 'claude-sonnet-5', + displayName: 'Sonnet', + supportsEffort: true, + supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'] + }, + { value: 'haiku', resolvedModel: 'claude-haiku-4-5-20251001', displayName: 'Haiku' } +] + +function initFrame(model: string): Record { + // Keys mirror the real per-turn system/init frame: it carries `model` as the + // resolved id, and no effort of any kind. + return { + type: 'system', + subtype: 'init', + session_id: PROVIDER_SESSION_ID, + uuid: 'turn-init-uuid', + model, + apiKeySource: 'none' + } +} + +describe('Claude model confirmation', () => { + it('adopts the model a later turn reports when nothing was set since', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'sonnet' } + }) + + // The CLI's own report of what it is running — the only channel that carries + // it, since set_model answers success for a model it never resolves. + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-haiku-4-5-20251001')) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'haiku' } + }) + }) + + it('keeps a just-set model until the next turn reports one', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + // No turn has run, so the acquisition-time report is older than the write and + // must not flip the pill back to the model the session started on. + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'haiku' } + }) + }) + + it('corrects the record when the turn runs a different model than was set', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'sonnet' } + }) + }) + + it('guards an effort against the model the turn reported, not the one that was set', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + // set_model answered success for a model it never resolved; the turn runs sonnet. + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + + // The picker offers sonnet's levels, so refusing one under haiku — a model the + // pill does not show and the child is not running — is the false positive. + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'high', fence: 7 }) + ).resolves.toMatchObject({ effort: 'high' }) + }) + + it('keeps guarding against the reported model across a second effort write', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + await adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'high', fence: 7 }) + + // The effort write bumps the option fence but does not change what the child + // runs, so the sonnet report is still current and still governs the guard. + // `max` skips the settings readback by contract, so only the catalog gates it: + // sonnet advertises it, haiku advertises no effort control at all. + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'max', fence: 7 }) + ).resolves.toMatchObject({ effort: 'max' }) + }) + + it('guards an effort against a just-set model no turn has reported yet', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + // The acquisition-time report predates the write, so haiku — which advertises + // no effort control — is still the model the guard must answer for. + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'high', fence: 7 }) + ).rejects.toThrow('claude model haiku does not accept effort high') + }) + + it('stops vouching for a confirmed effort once the model changes', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'high', fence: 7 }) + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'sonnet', effort: 'high', confirmed: ['model', 'effort'] } + }) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + // The readback was taken under sonnet; nothing has reported haiku holding it. + const options = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(options.current.effort).toBe('high') + expect(options.current.confirmed).toBeUndefined() + }) +}) + +describe('Claude effort the settings readback cannot report', () => { + function sessionWith( + reported: string, + calls: string[] = [] + ): { session: ClaudeSession; calls: string[] } { + return { + session: { + options: new Map([['model', 'sonnet']]), + reportedOptions: {}, + optionMutationSequence: 0, + confirmedOptions: new Set(), + connection: { + supportedModels: async () => { + calls.push('list_models') + return CATALOG + }, + applyFlagSettings: async (settings: { effortLevel?: string }) => { + calls.push(`apply:${settings.effortLevel}`) + }, + getSettings: async () => { + calls.push('get_settings') + return { + applied: { effort: reported }, + effective: { effortLevel: reported }, + sources: {} + } + } + } + } as unknown as ClaudeSession, + calls + } + } + + it('records a session-scoped effort the persisted settings never carry', async () => { + // `max` applies for the session and is deliberately excluded from the + // persisted effortLevel, so the readback reporting `high` is an absence of + // evidence, not a refusal — and the CLI offers `max` in its own catalog. + const { session, calls } = sessionWith('high') + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'max' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'max' }) + expect(calls).toEqual(['list_models', 'apply:max']) + }) +}) diff --git a/src/main/claude/claude-structured-option-confirmation.test.ts b/src/main/claude/claude-structured-option-confirmation.test.ts new file mode 100644 index 00000000000..ca7b8b70f1c --- /dev/null +++ b/src/main/claude/claude-structured-option-confirmation.test.ts @@ -0,0 +1,193 @@ +import { describe, expect, it } from 'vitest' +import { + applyStructuredAgentSessionOptions, + createStructuredAgentSessionOptionState, + structuredAgentSessionOptionSnapshot +} from '../../shared/structured-agent-session-options' +import { CLAUDE_SESSION_OPTION_CATALOG } from '../../shared/agent-session-option-catalog-claude-codex' +import type { AgentSessionOptionsResult } from '../../shared/agent-session-wire' +import type { SessionOptionDescriptor } from '../../shared/native-chat-session-options' +import { setClaudeStructuredOption } from './claude-structured-options' +import type { ClaudeSession } from './claude-structured-session-state' +import { PROVIDER_SESSION_ID, acquired, fakeClaude } from './claude-structured-session-test-support' + +/** Verbatim rows from Claude Code 2.1.260's list_models response: `haiku` really + * does omit both effort keys, which is what makes an effort under it refusable. */ +const CATALOG = [ + { + value: 'sonnet', + resolvedModel: 'claude-sonnet-5', + displayName: 'Sonnet', + supportsEffort: true, + supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'] + }, + { value: 'haiku', resolvedModel: 'claude-haiku-4-5-20251001', displayName: 'Haiku' } +] + +function initFrame(model: string): Record { + return { + type: 'system', + subtype: 'init', + session_id: PROVIDER_SESSION_ID, + uuid: 'turn-init-uuid', + model, + apiKeySource: 'none' + } +} + +function modelPill(result: AgentSessionOptionsResult): SessionOptionDescriptor | undefined { + const state = applyStructuredAgentSessionOptions( + createStructuredAgentSessionOptionState('claude'), + CLAUDE_SESSION_OPTION_CATALOG, + result + ) + return structuredAgentSessionOptionSnapshot(state).find((d) => d.category === 'model') +} + +/** Provenance the record keeps. Nothing renders it — the pill shows the value + * either way, and a report that disagrees is what corrects it. */ +function modelSource(result: AgentSessionOptionsResult): string | undefined { + return modelPill(result)?.valueSource +} + +function modelValue(result: AgentSessionOptionsResult): string | undefined { + const kind = modelPill(result)?.kind + return kind?.type === 'select' ? kind.currentValue : undefined +} + +describe('structured option confirmation reaches the pill', () => { + it('shows a just-set model before any turn reports it', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(result.current.confirmed ?? []).not.toContain('model') + expect(modelSource(result)).toBe('dispatched') + }) + + it('marks the model reported once the provider names it back', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-haiku-4-5-20251001')) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(result.current.confirmed).toContain('model') + expect(modelSource(result)).toBe('reported') + }) + + it('records an effort the readback could not take without confirming it', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + settings: { applied: {}, effective: {}, sources: {} }, + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + // `max` is session-scoped and absent from the persisted settings, so it records + // without a readback — recorded, never vouched for. + await adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'max', fence: 7 }) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(result.current.effort).toBe('max') + expect(result.current.confirmed ?? []).not.toContain('effort') + }) + + it('confirms an effort the readback agreed with', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + settings: { applied: { effort: 'low' }, effective: { effortLevel: 'low' }, sources: {} }, + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + await adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'low', fence: 7 }) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(result.current.confirmed).toContain('effort') + }) + + it('treats a host that reports no confirmation as unconfirmed', () => { + // Wire compatibility: an older host omits `confirmed` entirely. Absence must + // read as unconfirmed provenance, and the pill still shows the host's value. + const result = { + models: [{ id: 'haiku', label: 'Haiku', isDefault: false, efforts: [] }], + current: { model: 'haiku' } + } + expect(modelSource(result)).toBe('dispatched') + expect(modelValue(result)).toBe('haiku') + }) +}) + +describe('the provider report corrects the pill', () => { + it('moves the pill to the model the turn actually ran', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + expect(modelValue(await adapter.readOptions({ sessionId: 'session-1', fence: 7 }))).toBe( + 'haiku' + ) + + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + + const corrected = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(modelValue(corrected)).toBe('sonnet') + expect(corrected.current.confirmed).toContain('model') + }) + + it('lets a newer write outrank the report it precedes', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(modelValue(result)).toBe('haiku') + expect(result.current.confirmed ?? []).not.toContain('model') + }) +}) + +describe('confirmation never outlives the write it belongs to', () => { + it('drops an earlier effort confirmation when the value changes', async () => { + const calls: string[] = [] + let reported = 'low' + const session = { + options: new Map([['model', 'sonnet']]), + reportedOptions: {}, + optionMutationSequence: 0, + confirmedOptions: new Set(), + connection: { + supportedModels: async () => CATALOG, + applyFlagSettings: async (s: { effortLevel?: string }) => { + calls.push(`apply:${s.effortLevel}`) + }, + getSettings: async () => ({ + applied: { effort: reported }, + effective: { effortLevel: reported }, + sources: {} + }) + } + } as unknown as ClaudeSession + + await setClaudeStructuredOption(session, { key: 'effort', value: 'low' }, undefined) + expect(session.confirmedOptions.has('effort')).toBe(true) + + // The provider now reports a level it cannot represent; the stale confirmation + // must not survive into the new value. + await setClaudeStructuredOption(session, { key: 'effort', value: 'max' }, undefined) + expect(session.options.get('effort')).toBe('max') + expect(session.confirmedOptions.has('effort')).toBe(false) + expect(calls).toEqual(['apply:low', 'apply:max']) + }) +}) diff --git a/src/main/claude/claude-structured-options.test.ts b/src/main/claude/claude-structured-options.test.ts new file mode 100644 index 00000000000..0738095f45d --- /dev/null +++ b/src/main/claude/claude-structured-options.test.ts @@ -0,0 +1,53 @@ +import { describe, expect, it, vi } from 'vitest' +import { setClaudeStructuredOption } from './claude-structured-options' +import type { ClaudeSession } from './claude-structured-session-state' +import { ClaudeBackgroundTaskTracker } from './claude-background-task-tracker' + +function sessionFor(setModel: ClaudeSession['connection']['setModel']): ClaudeSession { + return { + connection: { setModel } as ClaudeSession['connection'], + providerSessionId: 'provider-session', + claudeConfigDir: '/accounts/claude', + leafUuid: null, + fence: 1, + acquisitionGeneration: 'generation-1', + prompts: {} as ClaudeSession['prompts'], + dispatchWaiters: [], + retiredDispatchWaiters: [], + replayContentFallbackBlocked: false, + backgroundTasks: new ClaudeBackgroundTaskTracker(), + dispatchSequence: 0, + optionMutationSequence: 0, + options: new Map(), + reportedOptions: {}, + reportedModelMutation: 0, + confirmedOptions: new Set(), + restoreSkippedOptions: new Set(), + capabilities: [], + events: undefined, + translator: null + } +} + +describe('Claude structured option mutation fencing', () => { + it('does not let a delayed earlier apply overwrite a later option', async () => { + let releaseFirst!: () => void + const firstApply = new Promise((resolve) => { + releaseFirst = resolve + }) + const setModel = vi + .fn() + .mockReturnValueOnce(firstApply) + .mockResolvedValue(undefined) + const session = sessionFor(setModel) + + const first = setClaudeStructuredOption(session, { key: 'model', value: 'old' }, undefined) + await vi.waitFor(() => expect(setModel).toHaveBeenCalledTimes(1)) + const second = setClaudeStructuredOption(session, { key: 'model', value: 'new' }, undefined) + await expect(second).resolves.toEqual({ model: 'new' }) + + releaseFirst() + await expect(first).resolves.toEqual({ model: 'new' }) + expect(session.options).toEqual(new Map([['model', 'new']])) + }) +}) diff --git a/src/main/claude/claude-structured-options.ts b/src/main/claude/claude-structured-options.ts new file mode 100644 index 00000000000..3d1377b12c6 --- /dev/null +++ b/src/main/claude/claude-structured-options.ts @@ -0,0 +1,147 @@ +import type { EffortLevel, PermissionMode } from '@anthropic-ai/claude-agent-sdk' +import { ClaudeControlRequestError } from './claude-stream-json-connection' +import { + AgentSessionOptionRejectedError, + isAgentSessionOptionRejectedError +} from '../native-chat/agent-session-wire/structured-agent-session-option-error' +import { + readClaudeCurrentModel, + readClaudeModelEffortLevels, + readClaudeSettingsEffort +} from './claude-structured-session-options' +import type { ClaudeSession } from './claude-structured-session-state' + +const OPTION_ORDER = ['model', 'effort', 'permissionMode'] as const + +/** + * Efforts the settings readback cannot report. `max` applies for the rest of the + * session and is excluded from the persisted `effortLevel` by contract, so + * `get_settings` answers with the level underneath it — an absence of evidence + * that must not be read as the child refusing a level its own catalog offers. + */ +const UNREPORTED_EFFORTS: ReadonlySet = new Set(['max']) + +export function restoredClaudeStructuredSessionOptions( + options: Readonly> | undefined +): Map { + return new Map( + OPTION_ORDER.flatMap((key) => { + const value = options?.[key] + return value ? [[key, value] as const] : [] + }) + ) +} + +export async function setClaudeStructuredOption( + session: ClaudeSession, + input: { key: string; value: string }, + timeoutMs: number | undefined +): Promise>> { + const apply = + input.key === 'model' + ? () => session.connection.setModel(input.value, { timeoutMs }) + : input.key === 'permissionMode' + ? () => session.connection.setPermissionMode(input.value as PermissionMode, { timeoutMs }) + : input.key === 'effort' + ? () => + session.connection.applyFlagSettings( + { effortLevel: input.value as EffortLevel }, + { timeoutMs } + ) + : null + if (!apply) { + throw new AgentSessionOptionRejectedError( + `claude stream-json has no session option named ${input.key}` + ) + } + // The child stores an effort its model has no control for and keeps it across + // every later model switch and restore, so refuse before the write rather than + // read the acceptance back as adoption. Refused here, restore drops the stale + // value instead of replaying it onto a model that cannot use it. + if (input.key === 'effort') { + const { modelId, levels } = await readClaudeModelEffortLevels(session, timeoutMs) + if (levels && !levels.has(input.value)) { + throw new AgentSessionOptionRejectedError( + `claude model ${modelId} does not accept effort ${input.value}` + ) + } + } + const modelWasConfirmed = readClaudeCurrentModel(session).confirmed + const mutationSequence = ++session.optionMutationSequence + // Only a model write can stale the model report — an effort or permission-mode + // write does not change what the child is running. Leaving the stamp behind + // would drop the session back to the written model and refuse, on the next + // effort write, a level the model actually running advertises. + if (modelWasConfirmed && input.key !== 'model') { + session.reportedModelMutation = mutationSequence + } + try { + await apply() + } catch (error) { + if (error instanceof ClaudeControlRequestError) { + throw new AgentSessionOptionRejectedError(error) + } + throw error + } + // apply_flag_settings answers `success` for an effort it then ignores, so the + // absence of a throw proves nothing. Ask what the child actually holds. + const adopted = + input.key === 'effort' && !UNREPORTED_EFFORTS.has(input.value) + ? await session.connection + .getSettings({ timeoutMs }) + .then(readClaudeSettingsEffort) + .catch(() => null) + : null + if (mutationSequence !== session.optionMutationSequence) { + return Object.fromEntries(session.options) + } + // A disagreement stops main vouching for the value, it does not veto the write: + // the pre-flight guard already refuses levels the model advertises no control + // for, and no other client refuses on a readback. Keep the child's own answer so + // the disagreement survives as the level a later read falls back to. + if (adopted !== null && adopted !== input.value) { + session.reportedOptions.effort = adopted + } + session.options.set(input.key, input.value) + // Only a readback that agreed is adoption evidence; one that disagreed or could + // not be taken records the value but must not also claim the provider vouched for it. + if (adopted !== null && adopted === input.value) { + session.confirmedOptions.add(input.key) + } else { + session.confirmedOptions.delete(input.key) + } + // The effort readback was taken under the old model, so a model switch retires + // it: the child keeps the value but nothing has reported the new model holding + // it, and vouching for it would show a confirmed effort no readback covers. + if (input.key === 'model') { + session.confirmedOptions.delete('effort') + } + return Object.fromEntries(session.options) +} + +export async function restoreClaudeStructuredSessionOptions( + session: ClaudeSession, + timeoutMs: number | undefined +): Promise { + // Any write that was already in flight belongs to the previous acquisition + // state and must not repopulate this map after restore starts. + session.optionMutationSequence += 1 + // The fence bump is not a write, so the report the session already holds is still + // current as of this instant; leaving the stamp behind would make every restored + // session read as unconfirmed until its next turn. + session.reportedModelMutation = session.optionMutationSequence + const options = [...session.options.entries()] + session.options.clear() + for (const [key, value] of options) { + try { + await setClaudeStructuredOption(session, { key, value }, timeoutMs) + } catch (error) { + if (!isAgentSessionOptionRejectedError(error)) { + throw error + } + // A stale or unavailable preference must not poison every future acquire; + // the provider's current value remains authoritative and is re-persisted. + session.restoreSkippedOptions.add(key) + } + } +} diff --git a/src/main/claude/claude-structured-owner-identity.test.ts b/src/main/claude/claude-structured-owner-identity.test.ts new file mode 100644 index 00000000000..592b61df338 --- /dev/null +++ b/src/main/claude/claude-structured-owner-identity.test.ts @@ -0,0 +1,39 @@ +import { describe, expect, it, vi } from 'vitest' +import { CLAUDE_SPAWN_TOKEN_ENV, claudeProcessIdentity } from './claude-structured-owner-identity' + +const IDENTITY = { + sessionId: 'session-identity', + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'claude' as const, + providerHandle: { kind: 'claude' as const, sessionId: 'session-1', leafUuid: 'leaf-1' } +} + +describe('claude structured owner identity', () => { + it('exports the spawn token env and records the observed process identity', async () => { + expect(CLAUDE_SPAWN_TOKEN_ENV).toBe('ORCA_AGENT_SESSION_SPAWN_TOKEN') + await expect( + claudeProcessIdentity( + { identity: IDENTITY, spawnToken: 'spawn-a', pid: 4242 }, + async () => 123 + ) + ).resolves.toEqual({ + hostId: 'local', + pid: 4242, + processStartTimeMs: 123, + spawnToken: 'spawn-a' + }) + }) + + it('retries a failed start-time read before giving up', async () => { + const readStartTime = vi + .fn<(pid: number) => Promise>() + .mockResolvedValueOnce(null) + .mockResolvedValueOnce(null) + .mockResolvedValueOnce(456) + await expect( + claudeProcessIdentity({ identity: IDENTITY, spawnToken: 'spawn-a', pid: 4242 }, readStartTime) + ).resolves.toMatchObject({ processStartTimeMs: 456 }) + expect(readStartTime).toHaveBeenCalledTimes(3) + }) +}) diff --git a/src/main/claude/claude-structured-owner-identity.ts b/src/main/claude/claude-structured-owner-identity.ts index 1d13e6ec7c2..e8f251d41f9 100644 --- a/src/main/claude/claude-structured-owner-identity.ts +++ b/src/main/claude/claude-structured-owner-identity.ts @@ -1,4 +1,7 @@ +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' import type { AgentSessionProviderHandleLink } from '../../shared/agent-session-provider-handle' +import type { AgentSessionProcessIdentity } from '../../shared/agent-session-record' +import { readProcessStartTimeMs } from '../runtime/agent-session-process-identity-probe' export function claudeProviderHandleLink(input: { sessionId: string @@ -19,3 +22,41 @@ export function claudeProviderHandleLink(input: { observedAt: input.observedAt } } + +/** The child echoes its spawn token here so the owner probe can tell a live + * child of this reservation from a same-pid stranger. */ +export const CLAUDE_SPAWN_TOKEN_ENV = 'ORCA_AGENT_SESSION_SPAWN_TOKEN' + +const START_TIME_READ_ATTEMPTS = 3 + +export async function claudeProcessIdentity( + input: { + identity: AgentSessionJournalIdentity + spawnToken: string + pid: number | undefined + }, + readStartTime: (pid: number) => Promise = readProcessStartTimeMs +): Promise { + if (input.pid === undefined) { + throw new Error('claude app-server started without a pid') + } + let processStartTimeMs: number | null = null + for ( + let attempt = 0; + attempt < START_TIME_READ_ATTEMPTS && processStartTimeMs === null; + attempt += 1 + ) { + processStartTimeMs = await readStartTime(input.pid) + } + if (processStartTimeMs === null) { + // Why: recording null makes every later owner probe indeterminate — a durable latch. + // Failing here reaps the child and leaves a retryable refusal instead. + throw new Error(`claude app-server start time for pid ${input.pid} could not be read`) + } + return { + hostId: input.identity.hostId, + pid: input.pid, + processStartTimeMs, + spawnToken: input.spawnToken + } +} diff --git a/src/main/claude/claude-structured-prompt-items.test.ts b/src/main/claude/claude-structured-prompt-items.test.ts new file mode 100644 index 00000000000..79916d6a507 --- /dev/null +++ b/src/main/claude/claude-structured-prompt-items.test.ts @@ -0,0 +1,110 @@ +import { describe, expect, it } from 'vitest' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { encodeAgentSessionQuestionAnswers } from '../../shared/agent-session-question-answer' +import { claudeQuestionItems } from './claude-structured-prompt-items' +import { + applyClaudePromptAnswer, + encodeClaudeQuestionOptionId, + type ClaudePendingPrompt +} from './claude-structured-prompt-replies' + +describe('Claude structured question addressing', () => { + it('keeps wire IDs bounded while returning the original question and choice', () => { + const questionId = 'Which option? '.repeat(100) + const label = 'A detailed choice '.repeat(100) + const prompt: ClaudePendingPrompt = { + requestId: 'question-1', + promptKey: 'question-1', + toolUseId: 'tool-1', + toolName: 'AskUserQuestion', + kind: 'question', + input: { questions: [{ question: questionId, options: [{ label }] }] }, + suggestions: [], + questionIds: [questionId], + answers: new Map(), + settle: () => {} + } + + const item = claudeQuestionItems({ sessionId: 'session-1', prompt })[0]! + expect(agentJournalItemKey(item.identity).length).toBeLessThan(512) + expect(item.body.options[0]!.id.length).toBeLessThan(512) + expect(item.body.freeTextQuestionId).toBe('q1') + expect(applyClaudePromptAnswer({ prompt }, item.body.options[0]!.id)).toMatchObject({ + updatedInput: { answers: { [questionId]: label } } + }) + }) + + it('preserves colon-containing free-text answers', () => { + const questionId = 'Where should this run?' + const prompt: ClaudePendingPrompt = { + requestId: 'question-1', + promptKey: 'question-1', + toolUseId: 'tool-1', + toolName: 'AskUserQuestion', + kind: 'question', + input: { questions: [{ question: questionId }] }, + suggestions: [], + questionIds: [questionId], + answers: new Map(), + settle: () => {} + } + const answer = 'https://example.test:8443/path' + + expect( + applyClaudePromptAnswer({ prompt }, encodeClaudeQuestionOptionId('q1', answer)) + ).toMatchObject({ + updatedInput: { answers: { [questionId]: answer } } + }) + }) + + it('returns arrays for multi-select and preserves mixed single and Other answers', () => { + const multiQuestion = 'Which targets?' + const singleQuestion = 'Which mode?' + const otherQuestion = 'Where should it run?' + const prompt: ClaudePendingPrompt = { + requestId: 'question-1', + promptKey: 'question-1', + toolUseId: 'tool-1', + toolName: 'AskUserQuestion', + kind: 'question', + input: { + questions: [ + { + question: multiQuestion, + multiSelect: true, + options: [{ label: 'frontend' }, { label: 'backend' }] + }, + { + question: singleQuestion, + options: [{ label: 'fast' }, { label: 'safe' }] + }, + { question: otherQuestion, options: [] } + ] + }, + suggestions: [], + questionIds: [multiQuestion, singleQuestion, otherQuestion], + answers: new Map(), + settle: () => {} + } + const item = claudeQuestionItems({ sessionId: 'session-1', prompt })[0]! + const questions = item.body.questions! + const encoded = encodeAgentSessionQuestionAnswers([ + { + questionId: 'q1', + optionIds: [questions[0]!.options[0]!.id, questions[0]!.options[1]!.id] + }, + { questionId: 'q2', optionIds: [questions[1]!.options[1]!.id] }, + { questionId: 'q3', optionIds: [], other: 'remote host' } + ]) + + expect(applyClaudePromptAnswer({ prompt }, encoded)).toMatchObject({ + updatedInput: { + answers: { + [multiQuestion]: ['frontend', 'backend'], + [singleQuestion]: 'safe', + [otherQuestion]: 'remote host' + } + } + }) + }) +}) diff --git a/src/main/claude/claude-structured-prompt-items.ts b/src/main/claude/claude-structured-prompt-items.ts new file mode 100644 index 00000000000..3bdf8ab6091 --- /dev/null +++ b/src/main/claude/claude-structured-prompt-items.ts @@ -0,0 +1,134 @@ +import type { + AgentJournalApprovalItem, + AgentJournalItemIdentity, + AgentJournalPromptOption, + AgentJournalQuestion, + AgentJournalQuestionItem +} from '../../shared/agent-session-journal-types' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' +import { claudeRecord, claudeText } from './claude-structured-item-translation' +import { + CLAUDE_APPROVAL_DECISIONS, + encodeClaudeQuestionOptionId, + type ClaudeApprovalDecision, + type ClaudePendingPrompt +} from './claude-structured-prompt-replies' + +const APPROVAL_LABELS: Record = { + allow: 'Allow', + allowForSession: 'Allow for this session', + deny: 'Deny', + cancel: 'Stop' +} + +const PENDING = { + state: 'pending', + selectedOptionId: null, + resolvedBy: null, + resolvedAt: null +} as const + +export function claudePromptIdentity(input: { + sessionId: string + promptKey: string + questionId?: string +}): AgentJournalItemIdentity { + const suffix = input.questionId ? `:${input.questionId}` : '' + return { + provider: 'orca', + clientMessageId: `claude-prompt:${input.sessionId}:${input.promptKey}${suffix}` + } +} + +export function claudeApprovalItem(prompt: ClaudePendingPrompt): AgentJournalApprovalItem { + const serialized = JSON.stringify(prompt.input) + return { + kind: 'approval', + title: `Allow ${prompt.toolName}?`, + detail: serialized ? boundInlineText(serialized, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text : null, + options: CLAUDE_APPROVAL_DECISIONS.map((decision) => ({ + id: decision, + label: APPROVAL_LABELS[decision] + })), + resolution: { ...PENDING } + } +} + +export type ClaudeQuestionItem = { + identity: AgentJournalItemIdentity + body: AgentJournalQuestionItem +} + +function questionOptions( + question: Record, + questionAddress: string +): AgentJournalPromptOption[] { + if (!Array.isArray(question.options)) { + return [] + } + return question.options.flatMap((value, index) => { + const option = claudeRecord(value) + const label = claudeText(option?.label) + const description = claudeText(option?.description) + return label + ? [ + { + id: encodeClaudeQuestionOptionId(questionAddress, `choice-${index + 1}`), + label, + ...(description ? { description } : {}) + } + ] + : [] + }) +} + +export function claudeQuestionItems(input: { + sessionId: string + prompt: ClaudePendingPrompt +}): ClaudeQuestionItem[] { + const values = Array.isArray(input.prompt.input.questions) ? input.prompt.input.questions : [] + const questions = values.flatMap((value, index): AgentJournalQuestion[] => { + const question = claudeRecord(value) + const questionAddress = `q${index + 1}` + const text = claudeText(question?.question) ?? claudeText(question?.header) + const header = claudeText(question?.header) + return question && input.prompt.questionIds[index] && text + ? [ + { + id: questionAddress, + question: text, + ...(header ? { header } : {}), + options: questionOptions(question, questionAddress), + multiSelect: question.multiSelect === true, + freeTextQuestionId: questionAddress + } + ] + : [] + }) + if (questions.length === 0) { + return [] + } + const legacyCompatible = questions.length === 1 && questions[0]?.multiSelect === false + const first = questions[0]! + return [ + { + identity: claudePromptIdentity({ + sessionId: input.sessionId, + promptKey: input.prompt.promptKey + }), + body: { + kind: 'question', + question: legacyCompatible + ? first.question + : `${questions.length} grouped question${questions.length === 1 ? '' : 's'} from Claude`, + options: legacyCompatible ? first.options : [], + ...(legacyCompatible ? { freeTextQuestionId: first.freeTextQuestionId } : {}), + questions, + resolution: { ...PENDING } + } + } + ] +} diff --git a/src/main/claude/claude-structured-prompt-replies.ts b/src/main/claude/claude-structured-prompt-replies.ts new file mode 100644 index 00000000000..deec74b7308 --- /dev/null +++ b/src/main/claude/claude-structured-prompt-replies.ts @@ -0,0 +1,297 @@ +import { decodeAgentSessionQuestionAnswers } from '../../shared/agent-session-question-answer' + +export const CLAUDE_APPROVAL_DECISIONS = ['allow', 'allowForSession', 'deny', 'cancel'] as const +export type ClaudeApprovalDecision = (typeof CLAUDE_APPROVAL_DECISIONS)[number] + +/** Settles the SDK's `canUseTool` promise; `null` is the SDK's "no response written" sentinel. */ +export type ClaudePromptSettle = (response: Record | null) => void + +export type ClaudePendingPrompt = { + requestId: string + promptKey: string + toolUseId: string + toolName: string + kind: 'approval' | 'question' + input: Record + suggestions: unknown[] + questionIds: readonly string[] + answers: Map + settle: ClaudePromptSettle +} + +export type ClaudePromptRegistration = { + requestId: string + toolName: string + toolUseId: string + input: Record + suggestions: unknown[] + settle: ClaudePromptSettle +} + +type PromptBinding = { + address: string + questionId?: string +} + +function isRecord(value: unknown): value is Record { + return typeof value === 'object' && value !== null && !Array.isArray(value) +} + +function readString(value: unknown): string | null { + return typeof value === 'string' && value.trim().length > 0 ? value : null +} + +function questionsFrom(input: Record): Record[] { + return Array.isArray(input.questions) ? input.questions.filter(isRecord) : [] +} + +function questionIdFromAddress(prompt: ClaudePendingPrompt, address: string): string | null { + const match = /^q([1-9]\d*)$/.exec(address) + const index = match ? Number(match[1]) - 1 : -1 + return index >= 0 ? (prompt.questionIds[index] ?? null) : null +} + +function questionAnswer(prompt: ClaudePendingPrompt, questionId: string, optionId: string): string { + const decoded = decodeClaudeQuestionOptionId(optionId) + if (!decoded) { + return optionId + } + const questionIndex = prompt.questionIds.indexOf(questionId) + if (questionIndex === -1) { + return optionId + } + const choice = /^choice-([1-9]\d*)$/.exec(decoded.answer) + const optionIndex = choice ? Number(choice[1]) - 1 : -1 + const question = questionsFrom(prompt.input)[questionIndex] + const options = Array.isArray(question?.options) ? question.options : [] + const option = options[optionIndex] + const label = isRecord(option) ? readString(option.label) : null + if (decoded.questionId === `q${questionIndex + 1}` && label) { + return label + } + if (decoded.questionId === `q${questionIndex + 1}`) { + return decoded.answer + } + const legacyChoice = options.some( + (candidate) => isRecord(candidate) && readString(candidate.label) === decoded.answer + ) + return decoded.questionId === questionId && (legacyChoice || decoded.answer.trim().length > 0) + ? decoded.answer + : optionId +} + +function questionId(question: Record, index: number): string { + return readString(question.question) ?? readString(question.header) ?? `question-${index + 1}` +} + +export function encodeClaudeQuestionOptionId(questionId: string, answer: string): string { + return `${encodeURIComponent(questionId)}:${encodeURIComponent(answer)}` +} + +export function decodeClaudeQuestionOptionId( + optionId: string +): { questionId: string; answer: string } | null { + const separator = optionId.indexOf(':') + if (separator <= 0) { + return null + } + try { + return { + questionId: decodeURIComponent(optionId.slice(0, separator)), + answer: decodeURIComponent(optionId.slice(separator + 1)) + } + } catch { + return null + } +} + +export class ClaudePromptRegistry { + private readonly prompts = new Map() + private readonly journalBindings = new Map() + + register(registration: ClaudePromptRegistration): ClaudePendingPrompt | null { + const toolUseId = readString(registration.toolUseId) + const toolName = readString(registration.toolName) + const input = isRecord(registration.input) ? registration.input : null + if (!toolUseId || !toolName || !input) { + return null + } + const questions = toolName === 'AskUserQuestion' ? questionsFrom(input) : [] + const prompt: ClaudePendingPrompt = { + requestId: registration.requestId, + promptKey: registration.requestId, + toolUseId, + toolName, + kind: questions.length > 0 ? 'question' : 'approval', + input, + suggestions: Array.isArray(registration.suggestions) ? registration.suggestions : [], + questionIds: questions.map(questionId), + answers: new Map(), + settle: registration.settle + } + this.prompts.set(prompt.promptKey, prompt) + return prompt + } + + /** True only if the prompt was still pending; lets an abort and an answer race settle once. */ + forgetIfPending(prompt: ClaudePendingPrompt): boolean { + if (!this.prompts.has(prompt.promptKey)) { + return false + } + this.forget(prompt) + return true + } + + bindJournalItemId(journalItemId: string, promptKey: string, questionIdForItem?: string): void { + this.journalBindings.set(journalItemId, { + address: promptKey, + ...(questionIdForItem ? { questionId: questionIdForItem } : {}) + }) + } + + find(itemId: string): { prompt: ClaudePendingPrompt; questionId?: string } | null { + const binding = this.journalBindings.get(itemId) + const prompt = this.prompts.get(binding?.address ?? itemId) + return prompt + ? { prompt, ...(binding?.questionId ? { questionId: binding.questionId } : {}) } + : null + } + + cancel(requestId: string): ClaudePendingPrompt | null { + const prompt = this.prompts.get(requestId) ?? null + if (prompt) { + this.forget(prompt) + } + return prompt + } + + forget(prompt: ClaudePendingPrompt): void { + this.prompts.delete(prompt.promptKey) + for (const [itemId, binding] of this.journalBindings) { + if (binding.address === prompt.promptKey) { + this.journalBindings.delete(itemId) + } + } + } + + clear(): ClaudePendingPrompt[] { + const pending = [...this.prompts.values()] + this.prompts.clear() + this.journalBindings.clear() + return pending + } +} + +function approvalResponse(prompt: ClaudePendingPrompt, optionId: string): Record { + if (!(CLAUDE_APPROVAL_DECISIONS as readonly string[]).includes(optionId)) { + throw new Error(`${optionId} is not a Claude approval decision`) + } + const decision = optionId as ClaudeApprovalDecision + if (decision === 'allow' || decision === 'allowForSession') { + return { + behavior: 'allow', + updatedInput: prompt.input, + ...(decision === 'allowForSession' && prompt.suggestions.length > 0 + ? { updatedPermissions: prompt.suggestions } + : {}), + toolUseID: prompt.toolUseId + } + } + return { + behavior: 'deny', + message: decision === 'cancel' ? 'User stopped this turn.' : 'User denied this action.', + ...(decision === 'cancel' ? { interrupt: true } : {}), + toolUseID: prompt.toolUseId + } +} + +function questionResponse( + prompt: ClaudePendingPrompt, + optionId: string, + boundQuestionId?: string +): Record | null { + const decoded = decodeClaudeQuestionOptionId(optionId) + const decodedQuestionId = decoded + ? (questionIdFromAddress(prompt, decoded.questionId) ?? + (prompt.questionIds.includes(decoded.questionId) ? decoded.questionId : null)) + : null + const selectedQuestionId = + boundQuestionId ?? + decodedQuestionId ?? + (prompt.questionIds.length === 1 ? prompt.questionIds[0] : null) + if (!selectedQuestionId || !prompt.questionIds.includes(selectedQuestionId)) { + throw new Error(`${optionId} does not name a question on Claude prompt ${prompt.promptKey}`) + } + const answer = questionAnswer(prompt, selectedQuestionId, optionId) + prompt.answers.set(selectedQuestionId, answer) + if (prompt.questionIds.some((id) => !prompt.answers.has(id))) { + return null + } + const answers: Record = {} + for (const id of prompt.questionIds) { + answers[id] = prompt.answers.get(id) as string + } + return { + behavior: 'allow', + updatedInput: { ...prompt.input, answers }, + toolUseID: prompt.toolUseId + } +} + +function groupedQuestionResponse( + prompt: ClaudePendingPrompt, + optionId: string +): Record | null { + const grouped = decodeAgentSessionQuestionAnswers(optionId) + if (!grouped) { + return null + } + const questions = questionsFrom(prompt.input) + if (grouped.length !== prompt.questionIds.length) { + throw new Error(`Grouped answer does not match Claude prompt ${prompt.promptKey}`) + } + const answers: Record = {} + for (let index = 0; index < questions.length; index += 1) { + const question = questions[index]! + const providerQuestionId = prompt.questionIds[index] + const answer = grouped.find((entry) => entry.questionId === `q${index + 1}`) + if (!providerQuestionId || !answer) { + throw new Error(`Grouped answer does not name question ${index + 1}`) + } + const selected = answer.optionIds.map((selectedId) => + questionAnswer(prompt, providerQuestionId, selectedId) + ) + const other = answer.other?.trim() + if (question.multiSelect === true) { + const values = [...selected, ...(other ? [other] : [])] + if (values.length === 0) { + throw new Error(`Grouped answer leaves question ${index + 1} empty`) + } + answers[providerQuestionId] = values + } else { + const value = other || selected[0] + if (!value || selected.length > 1) { + throw new Error(`Grouped answer is invalid for question ${index + 1}`) + } + answers[providerQuestionId] = value + } + } + return { + behavior: 'allow', + updatedInput: { ...prompt.input, answers }, + toolUseID: prompt.toolUseId + } +} + +export function applyClaudePromptAnswer( + found: { prompt: ClaudePendingPrompt; questionId?: string }, + optionId: string +): Record | null { + if (found.prompt.kind === 'approval') { + return approvalResponse(found.prompt, optionId) + } + return ( + groupedQuestionResponse(found.prompt, optionId) ?? + questionResponse(found.prompt, optionId, found.questionId) + ) +} diff --git a/src/main/claude/claude-structured-provider-fallback.test.ts b/src/main/claude/claude-structured-provider-fallback.test.ts new file mode 100644 index 00000000000..e8b27114da4 --- /dev/null +++ b/src/main/claude/claude-structured-provider-fallback.test.ts @@ -0,0 +1,117 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentSessionJournalIdentity +} from '../../shared/agent-session-journal-types' +import { openAgentSessionJournal } from '../native-chat/agent-session-journal/journal-store-factory' +import { createDeferredStructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' +import type { ClaudeStructuredSessionEvent } from './claude-structured-session-state' + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-1', + workspaceId: 'workspace-1', + hostId: 'host-1', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: 'provider-1', leafUuid: 'leaf-1' } +} + +let root = '' + +function message( + role: 'assistant' | 'user', + uuid: string, + content: unknown[] +): ClaudeStructuredSessionEvent { + return { + type: 'message', + sessionId: 'orca-session', + message: { + type: role, + uuid, + session_id: 'provider-1', + message: { role, content } + } + } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-claude-provider-fallback-')) +}) + +afterEach(async () => { + await rm(root, { recursive: true, force: true }) +}) + +describe('Claude provider fallback', () => { + it('drops suppressed init frames instead of dereferencing a null translation', () => { + const items: { identity: unknown; body: AgentJournalItemBody }[] = [] + const sink = { + appendItem: (identity: unknown, body: AgentJournalItemBody) => { + items.push({ identity, body }) + }, + appendTombstone: vi.fn(), + publish: vi.fn() + } + const translator = createClaudeJournalTranslator({ sink }) + const initEvent: ClaudeStructuredSessionEvent = { + type: 'message', + sessionId: 'orca-session', + message: { + type: 'system', + subtype: 'init', + session_id: 'provider-1', + uuid: 'init-1' + } + } + + expect(() => translator.handle(initEvent)).not.toThrow() + expect(items).toEqual([]) + }) + + it('keeps provider-fallback rows distinct across acquisitions', async () => { + const journal = await openAgentSessionJournal({ + identity: IDENTITY, + journalDir: root, + now: () => 1_700_000_000_000, + mintEpoch: () => 'epoch-1' + }) + const deferred = createDeferredStructuredAgentSessionEventSink() + deferred.bind({ + journal, + fence: 1, + publish: vi.fn() + }) + + const first = createClaudeJournalTranslator({ sink: deferred.sink, fallbackIdPrefix: '1' }) + const second = createClaudeJournalTranslator({ sink: deferred.sink, fallbackIdPrefix: '2' }) + + first.handle(message('assistant', 'assistant-1', [{ type: 'future_event', message: 'first' }])) + await deferred.drained() + second.handle( + message('assistant', 'assistant-2', [{ type: 'future_event', message: 'second' }]) + ) + await deferred.drained() + + const fallbackRows = journal + .snapshot() + .items.filter( + (item) => + item.body.kind === 'status' && + item.body.providerFrame?.kind === 'message:assistant:content:future_event' + ) + + expect(fallbackRows).toHaveLength(2) + expect(fallbackRows.map(statusText)).toEqual(['first', 'second']) + }) +}) + +function statusText(row: { body: AgentJournalItemBody }): string { + if (row.body.kind !== 'status') { + throw new Error('expected status row') + } + return row.body.text +} diff --git a/src/main/claude/claude-structured-provider-fallback.ts b/src/main/claude/claude-structured-provider-fallback.ts new file mode 100644 index 00000000000..2528ac027df --- /dev/null +++ b/src/main/claude/claude-structured-provider-fallback.ts @@ -0,0 +1,125 @@ +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' +import { CLAUDE_STREAM_JSON_FRAME_KINDS } from '../native-chat/agent-session-wire/claude-stream-json-frame-schema' +import { unhandledProviderFrameJournalItem } from '../native-chat/agent-session-wire/unhandled-provider-frame' +import { claudeRecord, claudeText } from './claude-structured-item-translation' + +export function claudeProviderFrameKind(message: Record): string { + const type = claudeText(message.type) ?? 'unknown' + const subtype = claudeText(message.subtype) + const eventType = claudeText(claudeRecord(message.event)?.type) + return ['message', type, subtype ?? eventType].filter(Boolean).join(':') +} + +const SETTLED_RESULT_KINDS: ReadonlySet = new Set( + CLAUDE_STREAM_JSON_FRAME_KINDS.filter((kind) => kind.startsWith('message:result:')) +) + +/** A catalogued result subtype is the turn-complete signal the translator settles + * itself; only an unmodeled subtype still needs the provider-fallback row. */ +export function isSettledClaudeResultKind(kind: string): boolean { + return SETTLED_RESULT_KINDS.has(kind) +} + +/** + * The failure a result frame carries that the turn's own frames never showed. + * + * Suppression is by meaning, not by kind. The SDK models an API failure as a + * SUCCESS-subtype result whose `result` string IS the error text and which has + * no assistant frame behind it, so keying on the subtype tombstones the turn and + * shows the user a completed, empty reply. A turn the user aborted is the + * opposite: its interrupt frame already says so, and the diagnostic in `errors` + * would only be noise. + */ +export function claudeResultFailure( + message: Record +): { text: string | null } | null { + if (message.is_error !== true) { + return null + } + const terminalReason = claudeText(message.terminal_reason) + if (terminalReason === 'aborted_streaming' || terminalReason === 'aborted_tools') { + return null + } + const result = claudeText(message.result)?.trim() + if (result) { + return { text: result } + } + const errors = Array.isArray(message.errors) + ? message.errors.flatMap((entry) => { + const text = claudeText(entry)?.trim() + return text ? [text] : [] + }) + : [] + // Nothing readable to lead with, but a reported failure still gets its row. + return { text: errors.length > 0 ? errors.join('\n') : null } +} + +/** + * What a message part that Orca cannot render says for itself. The kinds under + * `message::content:*` are synthesised from whatever `part.type` the CLI + * sends, so they can never be catalogued ahead of time; printing one is leaking + * wire vocabulary at a user who cannot act on it. The frame stays on the row's + * disclosure, so nothing is dropped and the next reader can still name it. + */ +export const CLAUDE_UNRENDERABLE_CONTENT_TEXT = 'Claude sent content Orca cannot display yet' + +export function isModeledClaudeContent(value: unknown): boolean { + const part = claudeRecord(value) + if (!part) { + return false + } + if (part.type === 'text') { + return claudeText(part.text) !== null + } + if (part.type === 'image') { + const source = claudeRecord(part.source) + if (source?.type === 'url') { + return claudeText(source.url) !== null + } + // A local attachment is replayed as the base64 (or file) source Orca itself + // sent, so it is content we recognise -- not an unknown part to surface. + return source?.type === 'base64' || source?.type === 'file' + } + if (part.type === 'tool_use') { + return claudeText(part.id) !== null && claudeText(part.name) !== null + } + if (part.type === 'tool_result') { + return claudeText(part.tool_use_id) !== null + } + // Redacted thinking arrives as an empty string plus a signature. + return part.type === 'thinking' || part.type === 'redacted_thinking' +} + +export function createClaudeProviderFrameFallback( + sink: StructuredAgentSessionEventSink, + acquisitionId: string +): { + /** `displayText` leads the row when Claude knows the sentence the frame itself does not name. */ + append: (kind: string, payload: unknown, displayText?: string | null) => void +} { + let sequence = 0 + return { + append: (kind, payload, displayText) => { + sequence += 1 + const translated = unhandledProviderFrameJournalItem('claude', kind, payload) + if (!translated) { + return + } + const bounded = displayText + ? boundInlineText(displayText, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text + : null + sink.appendItem( + { + provider: 'orca', + clientMessageId: `provider-frame:claude:${acquisitionId}:${sequence}` + }, + bounded ? { ...translated.body, text: bounded } : translated.body + ) + sink.publish() + } + } +} diff --git a/src/main/claude/claude-structured-real-cli.test.ts b/src/main/claude/claude-structured-real-cli.test.ts new file mode 100644 index 00000000000..0f22c175cc6 --- /dev/null +++ b/src/main/claude/claude-structured-real-cli.test.ts @@ -0,0 +1,299 @@ +import { spawnSync } from 'node:child_process' +import { randomUUID } from 'node:crypto' +import { mkdtemp, rm } from 'node:fs/promises' +import { homedir, tmpdir } from 'node:os' +import { basename, join, relative } from 'node:path' +import { describe, expect, it } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import { resolveClaudeCommand } from '../codex-cli/command' +import { resolveSessionFilePath } from '../native-chat/session-file-resolver' +import { getSpawnArgsForWindows } from '../win32-utils' +import { CLAUDE_STRUCTURED_BASE_OPTIONS } from './claude-structured-launch-resolution' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' + +const command = resolveClaudeCommand() +const versionLaunch = getSpawnArgsForWindows(command, ['--version']) +const realClaudeAvailable = + spawnSync(versionLaunch.spawnCmd, versionLaunch.spawnArgs, { + stdio: 'ignore', + windowsHide: true, + timeout: 5_000 + }).status === 0 +const authStatusLaunch = getSpawnArgsForWindows(command, ['auth', 'status', '--json']) +/** The CLI's own account report — the only source of truth for where it writes that + * is not derived from Orca's own path expressions. */ +const realClaudeAuthStatus = (() => { + if (!realClaudeAvailable) { + return null + } + const result = spawnSync(authStatusLaunch.spawnCmd, authStatusLaunch.spawnArgs, { + encoding: 'utf8', + windowsHide: true, + timeout: 5_000 + }) + if (result.status !== 0) { + return null + } + try { + return JSON.parse(result.stdout) as { loggedIn?: boolean; projectsDirectory?: string } + } catch { + return null + } +})() +const realClaudeAuthenticated = realClaudeAuthStatus?.loggedIn === true + +function realAdapter( + providerSessionId: string, + claudeConfigDir: string, + events: ClaudeStructuredSessionEvent[] = [] +): ClaudeStructuredSessionAdapter { + return new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: command, + options: { ...CLAUDE_STRUCTURED_BASE_OPTIONS, sessionId: providerSessionId }, + cwd: process.cwd(), + claudeConfigDir, + providerSessionId, + resumeLeafUuid: null, + resumed: false + }), + onEvent: (event) => events.push(event), + readProcessStartTime: async () => 1, + now: () => 2, + initTimeoutMs: 5_000 + }) +} + +function identity(providerSessionId: string): AgentSessionJournalIdentity { + return { + sessionId: 'real-cli-handshake', + workspaceId: 'real-cli-workspace', + hostId: 'local', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: providerSessionId, leafUuid: null } + } +} + +/** The CLI flushes its transcript on its own schedule; poll rather than race it. */ +async function waitForResolvedTranscript( + providerSessionId: string, + timeoutMs = 15_000 +): Promise { + const deadline = Date.now() + timeoutMs + for (;;) { + // No options: the exact call transcript-read-cache.ts makes for mobile. + const resolved = await resolveSessionFilePath('claude', providerSessionId) + if (resolved || Date.now() >= deadline) { + return resolved + } + await new Promise((resolve) => setTimeout(resolve, 250)) + } +} + +describe.skipIf(!realClaudeAvailable)('Claude structured real CLI handshake', () => { + it.skipIf(!realClaudeAuthenticated)( + 'proves a pre-minted session before the first user message', + async () => { + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = realAdapter(providerSessionId, claudeConfigDir, events) + + try { + const acquisition = await adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli' + }) + const observedSubtypes = events.flatMap((event) => + event.type === 'message' ? [event.message.subtype] : [] + ) + + expect(acquisition.link.handle).toMatchObject({ + provider: 'claude', + sessionId: providerSessionId, + // Init/SessionStart UUIDs are protocol frames, not resumable + // main-transcript leaves; no cursor exists before the first user turn. + leafUuid: null + }) + expect(observedSubtypes).toContain('hook_started') + } finally { + await adapter.closeAll() + } + }, + 10_000 + ) + + // Unit tests can only pin the shape we read, which is exactly how the blank + // Effort pill survived every gate: the fixture invented an `effortLevel` on a + // frame the CLI does not send. This asserts both halves against the live + // binary — that get_settings reports the effort, and that init does not. + it.skipIf(!realClaudeAuthenticated)( + 'reports the current effort through get_settings and never on the init frame', + async () => { + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = realAdapter(providerSessionId, claudeConfigDir, events) + + try { + await adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli-effort' + }) + const published = events.flatMap((event) => + event.type === 'message' ? [event.message] : [] + ) + const options = await adapter.readOptions({ sessionId: 'real-cli-handshake', fence: 1 }) + + expect(published.length).toBeGreaterThan(0) + // Not just the init frame: no frame the CLI publishes carries an effort + // at all. Goes red the day one does, which is when the simpler fix + // becomes available. Which frame proves the session varies by host, so + // this asserts over all of them rather than picking one. + expect(published.filter((frame) => 'effortLevel' in frame)).toEqual([]) + // Goes red if `effective.effortLevel` is renamed or dropped, which no + // fixture-backed test can see. + expect(options.current.effort).toEqual(expect.any(String)) + } finally { + await adapter.closeAll() + } + }, + 15_000 + ) + + // Mobile native chat never reads the structured journal — it reads the CLI's own + // transcript through native-chat/session-file-resolver.ts. So this resolves the way + // transcript-read-cache.ts:104 does, with NO root override, and checks the answer + // against the root the CLI itself reports. Deriving the expected root from Orca's own + // `CLAUDE_CONFIG_DIR || ~/.claude` expression — the same one the code under test uses — + // would move both sides together and stay green in exactly the environment that + // blacks mobile out. + // The turn is what creates the file: an init-only handshake writes nothing. + it.skipIf(!realClaudeAuthenticated || !realClaudeAuthStatus?.projectsDirectory)( + 'writes its transcript where the mobile session-file resolver looks for it', + async () => { + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const adapter = realAdapter(providerSessionId, claudeConfigDir) + const cliProjectsDir = realClaudeAuthStatus?.projectsDirectory as string + + let transcriptPath: string | null = null + try { + await adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli-transcript' + }) + await adapter.dispatch({ + sessionId: 'real-cli-handshake', + clientMessageId: 'real-cli-transcript-1', + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hi' }] }, + fence: 1 + }) + transcriptPath = await waitForResolvedTranscript(providerSessionId) + } finally { + await adapter.closeAll() + } + + expect(transcriptPath).not.toBeNull() + expect(basename(transcriptPath ?? '')).toBe(`${providerSessionId}.jsonl`) + // `//.jsonl` + expect(relative(cliProjectsDir, transcriptPath ?? '').split(/[\\/]/)).toHaveLength(2) + // And the pinned account home is that same root, so the host-side leaf recovery + // (structured-claude-runtime-adapter.ts:64) and mobile agree. + expect(join(claudeConfigDir, 'projects')).toBe(cliProjectsDir) + }, + 45_000 + ) + + // The model half of the same lesson: a fixture can only pin the shape we read. + // set_model answers success for a model it never resolves — a nonexistent id is + // accepted and only fails once a turn runs — so the CLI's own report is the only + // adoption evidence, and it arrives on the init frame that opens each turn. This + // asserts that frame carries the resolved model against the live binary; it goes + // red the day the CLI stops reporting it, which is the day the confirmation + // silently degrades to echoing back whatever Orca sent. + it.skipIf(!realClaudeAuthenticated)( + 'reports the model it adopted on the init frame that opens each turn', + async () => { + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = realAdapter(providerSessionId, claudeConfigDir, events) + + try { + await adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli-model' + }) + await adapter.setOption({ + sessionId: 'real-cli-handshake', + key: 'model', + value: 'haiku', + fence: 1 + }) + const before = events.length + await adapter.dispatch({ + sessionId: 'real-cli-handshake', + clientMessageId: 'real-cli-model-1', + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'Say ok' }] }, + fence: 1 + }) + const deadline = Date.now() + 60_000 + let frames: Record[] = [] + for (;;) { + frames = events + .slice(before) + .flatMap((event) => + event.type === 'message' && + event.message.type === 'system' && + event.message.subtype === 'init' + ? [event.message] + : [] + ) + if (frames.length > 0 || Date.now() >= deadline) { + break + } + await new Promise((resolve) => setTimeout(resolve, 250)) + } + + expect(frames).not.toHaveLength(0) + // Both halves: the field exists, and it names the model the picker asked + // for in the catalog's resolved shape rather than the id Orca sent. + expect(frames[0]?.model).toEqual(expect.any(String)) + expect(frames[0]?.model).toBe('claude-haiku-4-5-20251001') + await expect( + adapter.readOptions({ sessionId: 'real-cli-handshake', fence: 1 }) + ).resolves.toMatchObject({ current: { model: 'haiku' } }) + } finally { + await adapter.closeAll() + } + }, + 90_000 + ) + + it('turns a real silent unauthenticated startup into sign-in guidance', async () => { + const claudeConfigDir = await mkdtemp(join(tmpdir(), 'orca-claude-no-auth-')) + const providerSessionId = randomUUID() + const adapter = realAdapter(providerSessionId, claudeConfigDir) + + try { + await expect( + adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli-no-auth' + }) + ).rejects.toThrow(/not signed in.*Claude CLI.*CLAUDE_CONFIG_DIR/s) + } finally { + await adapter.closeAll() + await rm(claudeConfigDir, { recursive: true, force: true }) + } + }, 10_000) +}) diff --git a/src/main/claude/claude-structured-session-acquisition-processless.test.ts b/src/main/claude/claude-structured-session-acquisition-processless.test.ts new file mode 100644 index 00000000000..2c83dce1875 --- /dev/null +++ b/src/main/claude/claude-structured-session-acquisition-processless.test.ts @@ -0,0 +1,71 @@ +import { describe, expect, it, vi } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import { AgentSessionPreSpawnError } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { + ClaudeStreamJsonConnection, + openClaudeStreamJsonConnection +} from './claude-stream-json-connection' +import { ClaudeStructuredSessionAdapter } from './claude-structured-session-adapter' + +const PROVIDER_SESSION_ID = '819cf9f8-e43c-4ad7-b50f-54aa158a726a' +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-processless', + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'claude', + providerHandle: { kind: 'opaque', agent: 'claude', value: 'pending' } +} + +describe('Claude structured processless acquisition', () => { + it('classifies pre-pid error and close as processless with idempotent cleanup', async () => { + const fault = new Error('spawn claude ENOENT') + const close = vi.fn(async () => true) + const openConnection: typeof openClaudeStreamJsonConnection = async ( + _launch, + handlers = {} + ) => { + const connection: ClaudeStreamJsonConnection = { + pid: undefined, + closed: true, + exitVerdict: { root: 'processless', tree: 'exited' }, + initializationResult: async () => { + handlers.onFault?.(fault) + throw fault + }, + getSettings: async () => ({}), + supportedModels: async () => [], + interrupt: async () => undefined, + cancelAsyncMessage: async () => {}, + setModel: async () => {}, + setPermissionMode: async () => {}, + applyFlagSettings: async () => {}, + send: async () => {}, + stopTask: async () => {}, + close + } + return connection + } + const adapter = new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: null, + resumed: false + }), + openConnection + }) + + const error = await adapter + .acquire({ identity: IDENTITY, fence: 7, spawnToken: 'spawn-9' }) + .catch((cause: unknown) => cause) + + expect(error).toBeInstanceOf(AgentSessionPreSpawnError) + expect(error).toMatchObject({ message: fault.message }) + expect(close).toHaveBeenCalledOnce() + await expect(adapter.releaseAcquisition({ sessionId: IDENTITY.sessionId })).resolves.toBe(true) + expect(close).toHaveBeenCalledOnce() + }) +}) diff --git a/src/main/claude/claude-structured-session-acquisition.ts b/src/main/claude/claude-structured-session-acquisition.ts new file mode 100644 index 00000000000..e8d09bd78d9 --- /dev/null +++ b/src/main/claude/claude-structured-session-acquisition.ts @@ -0,0 +1,298 @@ +import { + AgentSessionAcquisitionExitUnprovenError, + AgentSessionPreSpawnError +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { + AgentSessionAcquisition, + StructuredAgentSessionAcquireInput +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE } from '../claude-accounts/environment' +import { isClaudeAuthSwitchInProgress } from '../claude-accounts/live-pty-gate' +import { openClaudeStreamJsonConnection } from './claude-stream-json-connection' +import { buildClaudePermissionCallbacks } from './claude-structured-inbound-control' +import { resolveClaudeReplayWaiter } from './claude-structured-dispatch' +import { + claudeAuthDiagnostic, + readClaudeCapabilities, + readClaudeFrameString, + readClaudeInit, + readClaudeModels +} from './claude-structured-init-proof' +import { + createClaudeInitDeadline, + requestClaudeInitialization +} from './claude-structured-init-deadline' +import { claudeConfigDirEnvPatch } from './claude-config-dir-pin' +import { CLAUDE_SPAWN_TOKEN_ENV, claudeProcessIdentity } from './claude-structured-owner-identity' +import { + restoreClaudeStructuredSessionOptions, + restoredClaudeStructuredSessionOptions +} from './claude-structured-options' +import { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import { createClaudeSessionJournalTranslator } from './claude-structured-journal-translation' +import { readClaudeSettingsEffort } from './claude-structured-session-options' +import { createClaudeSessionPublication } from './claude-structured-session-publication' +import { + cancelClaudeAcquisitionAttempt, + mintClaudeAcquisitionGeneration, + type ClaudeAcquisitionRegistry, + type ClaudeSession, + type ClaudeSessionExit, + type ClaudeStructuredSessionAdapterDeps, + type ClaudeAcquireCallbacks +} from './claude-structured-session-state' +import { + closeClaudePublishedSessionForDeps, + claudeAcquisitionCleanupError +} from './claude-structured-session-close' +import { readClaudeTranscriptEntryUuid } from './claude-tui-exit' + +export const CLAUDE_STRUCTURED_INIT_TIMEOUT_MS = 10_000 + +export async function acquireClaudeSession({ + input, + deps, + sessions, + acquisitions, + exits, + callbacks +}: { + input: StructuredAgentSessionAcquireInput + deps: ClaudeStructuredSessionAdapterDeps + sessions: Map + acquisitions: ClaudeAcquisitionRegistry + exits: Map + callbacks: ClaudeAcquireCallbacks +}): Promise { + // A managed-account switch is mid-swap of the pinned credential home; refuse here, + // before this acquisition cancels the previous attempt and closes the live session. + if (isClaudeAuthSwitchInProgress()) { + throw new AgentSessionPreSpawnError(new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE)) + } + const sessionId = input.identity.sessionId + const prompts = new ClaudePromptRegistry() + const translator = createClaudeSessionJournalTranslator( + input.events, + prompts, + String(input.fence) + ) + const { previous, attempt } = acquisitions.start(sessionId, prompts) + let liveSession: ClaudeSession | null = null + let observedLeafUuid: string | null = null, + expectedProviderSessionId: string | null = null + // Frames are admitted only after launch resolution proves the provider session + // this acquisition owns. Keep the check ahead of every stateful consumer. + const initTimeoutMs = deps.initTimeoutMs ?? CLAUDE_STRUCTURED_INIT_TIMEOUT_MS + const initDeadline = createClaudeInitDeadline(sessionId, initTimeoutMs) + + const onMessage = (message: Record): void => { + const init = readClaudeInit(message) + if (readClaudeFrameString(message, 'session_id') !== expectedProviderSessionId) { + // An init proof for another (or unnamed) provider must fail acquisition + // promptly, while ordinary foreign frames stay quarantined silently. + if (init || (message.type === 'system' && message.subtype === 'init')) { + initDeadline.reject(new Error('claude provider session expected')) + } + return + } + if (init) { + initDeadline.resolve(init) + // Every turn opens with an init frame naming the model the CLI is actually + // running; set_model answers success for a model it never resolves, so this + // report is the session's only adoption evidence. + if (liveSession && init.model) { + liveSession.reportedOptions.model = init.model + liveSession.reportedModelMutation = liveSession.optionMutationSequence + } + } + observedLeafUuid = readClaudeTranscriptEntryUuid(message) ?? observedLeafUuid + if (liveSession) { + liveSession.leafUuid = observedLeafUuid + } + const startsTurn = liveSession ? resolveClaudeReplayWaiter(liveSession, message) : false + callbacks.deliver(attempt, sessionId, () => + callbacks.emit(liveSession, input.events, { + type: 'message', + sessionId, + message, + ...(startsTurn ? { startsTurn: true } : {}) + }) + ) + } + const { canUseTool, onUserDialog } = buildClaudePermissionCallbacks({ + sessionId, + prompts, + emit: (event) => + callbacks.deliver(attempt, sessionId, () => callbacks.emit(liveSession, input.events, event)) + }) + + try { + if (previous && !(await cancelClaudeAcquisitionAttempt(previous))) { + acquisitions.restoreIfCurrent(sessionId, attempt, previous) + throw new AgentSessionAcquisitionExitUnprovenError( + new Error(`claude acquisition for session ${sessionId} could not be stopped`) + ) + } + acquisitions.assertCurrent(sessionId, attempt) + let resumeSession = sessions.get(sessionId) + if (!(await closeClaudePublishedSessionForDeps(sessions, sessionId, deps))) { + throw new AgentSessionAcquisitionExitUnprovenError( + new Error(`claude session ${sessionId} could not be stopped`) + ) + } + // A first-hand exit that has not yet proved its full tree still owns a cleanup + // obligation; never let a new acquisition hide that evidence by omission. + const retainedExit = exits.get(sessionId) + if (retainedExit) { + const firstProof = retainedExit.closePromise ? await retainedExit.closePromise : false + const proven = firstProof || (await retainedExit.connection.close().catch(() => false)) + if (!proven) { + throw claudeAcquisitionCleanupError(retainedExit.connection, retainedExit.error) + } + // The old child is superseded by this acquisition. Settle its lifecycle + // before discarding the retained proof so its cursor and callbacks are + // cleaned up exactly once. + await callbacks.settleExit(sessionId, retainedExit) + resumeSession ??= retainedExit.session + } + acquisitions.assertCurrent(sessionId, attempt) + // Both close paths persist their final leaf, so launch validates that durable head. + const launchIdentity = resumeSession + ? { + ...input.identity, + providerHandle: { + kind: 'claude' as const, + sessionId: resumeSession.providerSessionId, + leafUuid: resumeSession.leafUuid + } + } + : input.identity + const launch = await deps + .resolveLaunch({ identity: launchIdentity }) + .catch((error: unknown) => { + throw error instanceof AgentSessionPreSpawnError + ? error + : new AgentSessionPreSpawnError(error) + }) + expectedProviderSessionId = launch.providerSessionId + observedLeafUuid = launch.resumeLeafUuid + acquisitions.assertCurrent(sessionId, attempt) + const open = deps.openConnection ?? openClaudeStreamJsonConnection + const connection = await open( + { + pathToClaudeCodeExecutable: launch.pathToClaudeCodeExecutable, + options: launch.options, + cwd: launch.cwd, + env: { + ...launch.env, + [CLAUDE_SPAWN_TOKEN_ENV]: input.spawnToken, + // Compared against what the child would otherwise inherit, so the record's + // account home still wins over a diverging overlay without a needless pin. + // (`process` is shadowed by a local later in this function, so it is not named here.) + ...claudeConfigDirEnvPatch(launch.claudeConfigDir, launch.env ? { env: launch.env } : {}) + } + }, + { + onMessage, + canUseTool, + onUserDialog, + onFault: (error) => { + if (!attempt.published) { + initDeadline.reject(error) + } + }, + onExit: (error) => { + if (!attempt.published) { + initDeadline.reject(error) + } + callbacks.handleExit(sessionId, attempt, error) + } + } + ) + attempt.connection = connection + acquisitions.assertCurrent(sessionId, attempt) + initDeadline.start() + const [initialization, init] = await Promise.all([ + requestClaudeInitialization(connection, sessionId, initTimeoutMs), + initDeadline.promise + ]) + const models = readClaudeModels(initialization) + callbacks.deliver(attempt, sessionId, () => + callbacks.emit(liveSession, input.events, { type: 'options', sessionId, models }) + ) + initDeadline.clear() + acquisitions.assertCurrent(sessionId, attempt) + if (init.providerSessionId !== launch.providerSessionId) { + throw new Error( + `claude proved session ${init.providerSessionId}, expected ${launch.providerSessionId}` + ) + } + const settings = await connection + .getSettings({ timeoutMs: deps.requestTimeoutMs }) + .catch(() => null) + callbacks.deliver(attempt, sessionId, () => + callbacks.emit(liveSession, input.events, { + type: 'auth-diagnostic', + sessionId, + diagnostic: claudeAuthDiagnostic(init, settings) + }) + ) + const process = await claudeProcessIdentity( + { ...input, pid: connection.pid }, + deps.readProcessStartTime + ) + acquisitions.assertCurrent(sessionId, attempt) + if (connection.closed) { + throw new Error(`claude stream-json for session ${sessionId} exited while being acquired`) + } + const publication = createClaudeSessionPublication({ + connection, + init, + claudeConfigDir: launch.claudeConfigDir, + leafUuid: observedLeafUuid, + fence: input.fence, + effort: readClaudeSettingsEffort(settings), + resumed: launch.resumed, + prompts, + translator, + events: input.events, + process, + acquisitionGeneration: mintClaudeAcquisitionGeneration(deps), + options: restoredClaudeStructuredSessionOptions(input.options), + capabilities: readClaudeCapabilities(init, initialization), + ...(deps.mintLinkId ? { linkId: deps.mintLinkId() } : {}), + observedAt: deps.now?.() ?? Date.now() + }) + const acquired: AgentSessionAcquisition = publication.acquisition + liveSession = publication.session + await restoreClaudeStructuredSessionOptions(liveSession, deps.requestTimeoutMs) + acquisitions.assertCurrent(sessionId, attempt) + acquisitions.deleteIfCurrent(sessionId, attempt) + sessions.set(sessionId, liveSession) + attempt.published = true + for (const event of attempt.buffered.splice(0)) { + event() + } + return acquired + } catch (error) { + initDeadline.clear() + let acquisitionError = error + if (sessions.get(sessionId)?.connection !== attempt.connection) { + translator?.dispose() + // Settle any callback that fired before the failure so no SDK promise dangles. + for (const prompt of prompts.clear()) { + prompt.settle(null) + } + const closed = (await attempt.connection?.close()) ?? true + if (attempt.connection?.exitVerdict.root === 'processless') { + acquisitionError = new AgentSessionPreSpawnError(error) + } else if (!closed) { + acquisitionError = claudeAcquisitionCleanupError(attempt.connection, error) + } + } + acquisitions.deleteIfCurrent(sessionId, attempt) + throw acquisitionError + } finally { + attempt.finish() + } +} diff --git a/src/main/claude/claude-structured-session-adapter.test.ts b/src/main/claude/claude-structured-session-adapter.test.ts new file mode 100644 index 00000000000..859ba9a8ed8 --- /dev/null +++ b/src/main/claude/claude-structured-session-adapter.test.ts @@ -0,0 +1,891 @@ +import { homedir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it, vi } from 'vitest' +import { + AgentSessionAcquisitionExitUnprovenError, + AgentSessionAcquisitionRefusal, + AgentSessionAcquisitionRootExitObservedError +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { ClaudeStreamJsonConnection } from './claude-stream-json-connection' +import { ClaudeControlRequestError } from './claude-stream-json-connection' +import { CLAUDE_SPAWN_TOKEN_ENV } from './claude-structured-owner-identity' +import { encodeClaudeQuestionOptionId } from './claude-structured-prompt-replies' +import { + CLAUDE_STRUCTURED_INIT_TIMEOUT_MS, + type ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' +import { + acquired, + adapterFor, + fakeClaude, + identityFor, + invokeCanUseTool, + PROVIDER_SESSION_ID, + tick, + USER_MESSAGE, + type FakeConnection +} from './claude-structured-session-test-support' + +describe('ClaudeStructuredSessionAdapter.acquire', () => { + it('finishes its startup deadline before the paired mobile request deadline', () => { + expect(CLAUDE_STRUCTURED_INIT_TIMEOUT_MS).toBeLessThan(30_000) + }) + + it('pins the account and proves init without treating the system-frame uuid as a chain leaf', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + + const acquisition = await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9' + }) + + expect(claude.connections[0].launch).toMatchObject({ + cwd: '/work/repo', + env: { + [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-9', + CLAUDE_CONFIG_DIR: '/accounts/claude' + } + }) + // supportedDialogKinds is now a query() launch option, not an initialize request param. + expect(claude.connections[0].calls.slice(0, 2)).toEqual([ + { subtype: 'initialize' }, + { subtype: 'get_settings' } + ]) + expect(acquisition.process).toEqual({ + hostId: 'host-1', + pid: 4321, + processStartTimeMs: 1_700_000_000_000, + spawnToken: 'spawn-9' + }) + expect(acquisition.link).toEqual({ + linkId: `claude-7-${PROVIDER_SESSION_ID}-empty`, + handle: { provider: 'claude', sessionId: PROVIDER_SESSION_ID, leafUuid: null }, + origin: 'created', + mintedAtFence: 7, + observedAt: 1_700_000_000_500 + }) + expect(events[0]).toMatchObject({ type: 'message', message: { subtype: 'init' } }) + }) + + it('restores persisted model and effort before publishing a reacquired session', async () => { + const claude = fakeClaude() + const adapter = adapterFor(claude, { resumed: true }) + + await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + options: { model: 'opus', effort: 'high' } + }) + + expect(claude.connections[0].calls.slice(-4)).toEqual([ + { subtype: 'set_model', params: { model: 'opus' } }, + // The restored model's advertised levels gate the replay, so a stale effort + // is dropped rather than re-applied to a model with no effort control. + { subtype: 'list_models' }, + { subtype: 'apply_flag_settings', params: { settings: { effortLevel: 'high' } } }, + // The effort is only recorded once the child reports having adopted it. + { subtype: 'get_settings' } + ]) + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'opus', effort: 'high' } + }) + }) + + it.each([ + ['model', 'set_model', { model: 'retired-model' }], + ['effort', 'apply_flag_settings', { effort: 'retired-effort' }], + ['permissionMode', 'set_permission_mode', { permissionMode: 'retired-mode' }] + ] as const)( + 'self-heals a persisted %s rejected during restore', + async (key, subtype, options) => { + const claude = fakeClaude({ + routes: { + [subtype]: () => { + throw new ClaudeControlRequestError(subtype, 'value is no longer available') + } + } + }) + const adapter = adapterFor(claude) + + await expect( + adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + options + }) + ).resolves.toBeDefined() + expect(adapter.readOptionRestoreFailures('session-1')).toEqual([key]) + } + ) + + it('does not treat a transport timeout while restoring an option as recoverable', async () => { + const claude = fakeClaude({ + routes: { + set_model: () => { + throw new Error('claude set_model request timed out') + } + } + }) + const adapter = adapterFor(claude) + const input = { + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + options: { model: 'temporarily-unavailable' } + } + + await expect(adapter.acquire(input)).rejects.toThrow('claude set_model request timed out') + expect(claude.connections[0]?.closeCount).toBe(1) + }) + + it('recovers a cancellable lifecycle when a timed-out replay arrives late', async () => { + const claude = fakeClaude({ replayUuid: null }) + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + const sent = claude.connections[0]!.sent[0]! + claude.connections[0]!.handlers.onMessage?.({ + ...sent, + uuid: 'late-turn-1' + }) + + expect(events).toContainEqual( + expect.objectContaining({ + type: 'message', + startsTurn: true, + message: expect.objectContaining({ uuid: 'late-turn-1' }) + }) + ) + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'late-turn-1', fence: 7 }) + ).resolves.toEqual({ cancelled: true }) + }) + + it('quarantines SDK frames without the acquired session identity', async () => { + const claude = fakeClaude({ replayUuid: null }) + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + const connection = claude.connections[0]! + + connection.handlers.onMessage?.({ + type: 'assistant', + uuid: 'foreign-leaf', + session_id: 'foreign-provider-session', + message: { role: 'assistant', content: [{ type: 'text', text: 'do not admit' }] } + }) + connection.handlers.onMessage?.({ + type: 'assistant', + uuid: 'missing-session-leaf', + message: { role: 'assistant', content: [{ type: 'text', text: 'do not admit' }] } + }) + + const dispatch = adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + await Promise.resolve() + expect(connection.sent).toHaveLength(1) + connection.handlers.onMessage?.({ + ...connection.sent[0], + uuid: 'foreign-replay', + session_id: 'foreign-provider-session' + }) + await Promise.resolve() + expect(events.filter((event) => event.type === 'message')).toHaveLength(1) + + connection.handlers.onMessage?.({ + ...connection.sent[0], + session_id: PROVIDER_SESSION_ID + }) + await expect(dispatch).resolves.toMatchObject({ + state: 'accepted', + providerIdentity: { uuid: connection.sent[0]!.uuid } + }) + }) + + it('forwards configured launch environment while keeping ownership pins authoritative', async () => { + const claude = fakeClaude() + const adapter = adapterFor(claude, { + env: { + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + CLAUDE_CONFIG_DIR: '/wrong/account', + [CLAUDE_SPAWN_TOKEN_ENV]: 'wrong-token' + } + }) + + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + expect(claude.connections[0].launch.env).toEqual({ + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + CLAUDE_CONFIG_DIR: '/accounts/claude', + [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-9' + }) + }) + + it('leaves CLAUDE_CONFIG_DIR unset when the account home is the CLI default', async () => { + const claude = fakeClaude() + const adapter = adapterFor(claude, { claudeConfigDir: join(homedir(), '.claude'), env: {} }) + + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + // Pinning the CLI's own default suppresses the macOS Keychain and breaks claude.ai login. + expect(claude.connections[0].launch.env).toEqual({ [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-9' }) + }) + + it('re-pins the account home when the launch env would send the child elsewhere', async () => { + const claude = fakeClaude() + const accountHome = join(homedir(), '.claude') + const adapter = adapterFor(claude, { + claudeConfigDir: accountHome, + env: { CLAUDE_CONFIG_DIR: '/other/account' } + }) + + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + expect(claude.connections[0].launch.env).toEqual({ + CLAUDE_CONFIG_DIR: accountHome, + [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-9' + }) + }) + + it('accepts SessionStart as pre-turn proof without treating its system uuid as a leaf', async () => { + const claude = fakeClaude({ initProof: 'session-start', initUuid: 'session-start-uuid' }) + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + + const acquisition = await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9' + }) + + expect(acquisition.link.handle).toEqual({ + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: null + }) + expect(events[0]).toMatchObject({ + type: 'message', + message: { subtype: 'hook_started', hook_name: 'SessionStart:startup' } + }) + }) + + it('records only non-secret effective auth-lane diagnostics', async () => { + const claude = fakeClaude({ + settings: { + env: { + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + ANTHROPIC_AUTH_TOKEN: 'secret' + } + } + }) + const events: ClaudeStructuredSessionEvent[] = [] + await acquired(claude, {}, events) + + const diagnostic = events.find((event) => event.type === 'auth-diagnostic') + expect(diagnostic).toEqual({ + type: 'auth-diagnostic', + sessionId: 'session-1', + diagnostic: { + apiKeySourceConfigured: false, + baseUrlConfigured: true, + authTokenConfigured: true, + apiKeyConfigured: false, + settingSources: ['user', 'project', 'local'] + } + }) + expect(JSON.stringify(diagnostic)).not.toContain('secret') + expect(JSON.stringify(diagnostic)).not.toContain('gateway.example.test') + }) + + it('resumes the same provider id and refuses an init proof for another session', async () => { + const resumedClaude = fakeClaude() + const resumed = adapterFor(resumedClaude, { + resumed: true, + resumeLeafUuid: 'leaf-before' + }) + const acquisition = await resumed.acquire({ + identity: identityFor(), + fence: 9, + spawnToken: 'spawn-9' + }) + expect(acquisition.link.origin).toBe('resumed') + expect(acquisition.link.handle).toEqual({ + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: 'leaf-before' + }) + + const wrongClaude = fakeClaude({ initSessionId: 'different-session' }) + const wrong = adapterFor(wrongClaude) + await expect( + wrong.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + ).rejects.toThrow(/expected/) + expect(wrongClaude.connections[0].closeCount).toBe(1) + }) + + it('surfaces a CLI startup failure instead of waiting for the init deadline', async () => { + const claude = fakeClaude({ exitBeforeInit: 'Claude login required' }) + const adapter = adapterFor(claude) + + await expect( + adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + ).rejects.toThrow('Claude login required') + expect(claude.connections[0].closeCount).toBe(1) + }) + + it('closes a silent unauthenticated startup with actionable account guidance', async () => { + const claude = fakeClaude({ initProof: 'none' }) + const adapter = adapterFor(claude, {}, [], [], 20) + + const error = await adapter + .acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + .catch((cause: unknown) => cause) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionRefusal) + expect(error).toMatchObject({ + message: expect.stringMatching(/selected Claude account is signed in.*CLAUDE_CONFIG_DIR/s) + }) + expect(claude.connections[0].calls[0]).toEqual({ subtype: 'initialize' }) + expect(claude.connections[0].closeCount).toBe(1) + }) + + it('refuses an unauthenticated initialize response even when SessionStart runs', async () => { + const claude = fakeClaude({ + initProof: 'session-start', + initAccount: { apiProvider: 'firstParty', tokenSource: 'none' } + }) + const adapter = adapterFor(claude) + + await expect( + adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + ).rejects.toThrow(/not signed in.*Claude CLI.*CLAUDE_CONFIG_DIR/s) + expect(claude.connections[0].closeCount).toBe(1) + }) +}) + +describe('ClaudeStructuredSessionAdapter turns and controls', () => { + it('accepts a dispatch only after Claude replays its provider uuid', async () => { + const claude = fakeClaude({ replayUuid: 'user-provider-uuid' }) + const adapter = await acquired(claude) + + const result = await adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + + expect(result).toEqual({ + state: 'accepted', + providerIdentity: { + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + uuid: 'user-provider-uuid' + } + }) + expect(claude.connections[0].sent[0]).toMatchObject({ + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'ship it' }] }, + session_id: PROVIDER_SESSION_ID + }) + }) + + it('leaves delivery unconfirmed when no replay uuid arrives', async () => { + const adapter = await acquired(fakeClaude({ replayUuid: null })) + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + }) + + it('requires an acknowledged interrupt and supports controlled options', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-1', fence: 7 }) + ).resolves.toEqual({ cancelled: true }) + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'sonnet', fence: 7 }) + ).resolves.toEqual({ model: 'sonnet' }) + expect(claude.connections[0].calls.slice(-2)).toEqual([ + { subtype: 'interrupt', params: {} }, + { subtype: 'set_model', params: { model: 'sonnet' } } + ]) + + claude.routes.interrupt = () => { + throw new ClaudeControlRequestError('interrupt', 'not running') + } + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-2', fence: 7 }) + ).resolves.toEqual({ cancelled: false }) + + claude.routes.interrupt = () => { + throw new Error('claude interrupt request timed out') + } + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-3', fence: 7 }) + ).rejects.toThrow('timed out') + }) + + it('does not let a delayed cancellation for an earlier turn interrupt the later turn', async () => { + const claude = fakeClaude({ replayUuids: ['turn-T', 'turn-U'] }) + const adapter = await acquired(claude) + + await adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-T', + body: USER_MESSAGE, + fence: 7 + }) + await adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-U', + body: USER_MESSAGE, + fence: 7 + }) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-T', fence: 7 }) + ).resolves.toEqual({ cancelled: false }) + expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( + 0 + ) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-U', fence: 6 }) + ).resolves.toEqual({ cancelled: false }) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-U', fence: 7 }) + ).resolves.toEqual({ cancelled: true }) + expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( + 1 + ) + }) + + it('does not cancel an acknowledged turn after a later dispatch returns unknown', async () => { + const claude = fakeClaude({ replayUuids: ['turn-T', null] }) + const adapter = await acquired(claude) + + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-T', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toMatchObject({ + state: 'accepted', + providerIdentity: { uuid: 'turn-T' } + }) + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-U', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + expect(claude.connections[0].sent).toHaveLength(2) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-T', fence: 7 }) + ).resolves.toEqual({ cancelled: false }) + expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( + 0 + ) + }) + + it('classifies provider-declined options without treating timeouts as settled', async () => { + const claude = fakeClaude({ + routes: { + set_model: () => { + throw new ClaudeControlRequestError('set_model', 'model unavailable') + } + } + }) + const adapter = await acquired(claude) + + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'fable', fence: 7 }) + ).rejects.toMatchObject({ name: 'AgentSessionOptionRejectedError' }) + claude.routes.set_model = () => { + throw new Error('claude set_model request timed out') + } + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'opus', fence: 7 }) + ).rejects.toThrow('timed out') + }) + + it('hydrates live model choices and maps the resolved current model to its CLI id', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { + list_models: () => [ + { value: 'default', resolvedModel: 'claude-opus-5', displayName: 'Default' }, + { + value: 'opus', + resolvedModel: 'claude-opus-5', + displayName: 'Opus', + supportsEffort: true, + supportedEffortLevels: ['low', 'high'] + }, + { + value: 'sonnet', + resolvedModel: 'claude-sonnet-5', + displayName: 'Sonnet' + } + ] + } + }) + const adapter = await acquired(claude) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toEqual({ + models: [ + { + id: 'opus', + label: 'Opus', + isDefault: true, + efforts: [ + { value: 'low', label: 'Low' }, + { value: 'high', label: 'High' } + ] + }, + { id: 'sonnet', label: 'Sonnet', isDefault: false, efforts: [] } + ], + current: { model: 'sonnet', effort: 'high', confirmed: ['model', 'effort'] } + }) + }) + + it('keeps the shared Claude seed when live model discovery is unavailable', async () => { + const claude = fakeClaude({ + initModel: 'custom-model', + routes: { + list_models: () => { + throw new Error('unsupported') + } + } + }) + const adapter = await acquired(claude) + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + + expect(result.models.map((model) => model.id)).toEqual([ + 'fable', + 'opus', + 'sonnet', + 'haiku', + 'custom-model' + ]) + expect(result.current).toEqual({ + model: 'custom-model', + effort: 'high', + confirmed: ['model', 'effort'] + }) + }) +}) + +describe('ClaudeStructuredSessionAdapter acquisition cleanup', () => { + /** A start that fails after the child self-exited, with its close verdict scripted. */ + function failedStart( + unprovenCloseVerdict: ClaudeStreamJsonConnection['exitVerdict'] + ): Promise { + const claude = fakeClaude({ + exitBeforeInit: 'claude stream-json exited (code 1): not logged in', + unprovenCloseVerdict + }) + return adapterFor(claude) + .acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + .catch((error: unknown) => error) + } + + it('releases on a first-hand root exit while still carrying the CLI diagnostic', async () => { + // The root's pid and start time are the lease's identity, and they are + // provably dead: latching the session would strand a signed-out user. + const error = await failedStart({ root: 'exited', tree: 'unverifiable' }) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + expect((error as Error).message).toBe('claude stream-json exited (code 1): not logged in') + }) + + it('never releases while a descendant was observed alive', async () => { + const error = await failedStart({ root: 'exited', tree: 'live' }) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) + expect(error).not.toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + }) + + it('never releases for a root Orca never saw leave', async () => { + const error = await failedStart({ root: 'live', tree: 'unverifiable' }) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) + expect(error).not.toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + }) + + /** A published session whose CLI then exits first-hand, with the verdict its ladder holds. */ + async function exitedAfterPublish( + exitVerdict: ClaudeStreamJsonConnection['exitVerdict'] + ): Promise<{ adapter: ClaudeStructuredSessionAdapter; connection: FakeConnection }> { + const claude = fakeClaude({ unprovenCloseVerdict: exitVerdict }) + const adapter = await acquired(claude) + const connection = claude.connections[0] + connection.handlers.onExit?.(new Error('claude stream-json exited (code 1): crashed')) + return { adapter, connection } + } + + it('classifies cleanup after a first-hand exit removed the session as a root exit, never as proven', async () => { + // The host may still be committing or proving the lease when the child dies; + // its cleanup must find the exit the ladder observed, not an absence. + const { adapter, connection } = await exitedAfterPublish({ + root: 'exited', + tree: 'unverifiable' + }) + const error = await adapter.releaseAcquisition({ sessionId: 'session-1' }).catch((e) => e) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + expect((error as Error).message).toBe('claude stream-json exited (code 1): crashed') + expect(connection.closeCount).toBe(2) + }) + + it('never releases after an exit that left a descendant observed alive', async () => { + const { adapter } = await exitedAfterPublish({ root: 'exited', tree: 'live' }) + const error = await adapter.releaseAcquisition({ sessionId: 'session-1' }).catch((e) => e) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) + expect(error).not.toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + }) + + it('forgets a retained exit once the session is acquired again', async () => { + const options: Parameters[0] = {} + const claude = fakeClaude(options) + const adapter = await acquired(claude) + const first = claude.connections[0] + first.handlers.onExit?.(new Error('claude stream-json exited (code 1): crashed')) + first.exitVerdict = { root: 'exited', tree: 'unverifiable' } + first.close = async () => false + options.exitBeforeInit = 'claude stream-json exited (code 1): not logged in' + + await expect( + adapter.acquire({ identity: identityFor(), fence: 8, spawnToken: 'spawn-10' }) + ).rejects.toThrow('not logged in') + // The second start's own proven close is the answer; the first exit is stale. + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(true) + expect(first.closeCount).toBe(1) + }) + + it('reports unproven published-session cleanup so callers can retry safely', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + const connection = claude.connections[0] + connection.close = vi + .fn<() => Promise>() + .mockResolvedValueOnce(false) + .mockResolvedValueOnce(true) as unknown as FakeConnection['close'] + + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(false) + expect(await adapter.readOptions({ sessionId: 'session-1', fence: 7 })).toMatchObject({ + current: { model: 'claude-sonnet-5' } + }) + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(true) + expect(() => adapter.readOptions({ sessionId: 'session-1', fence: 7 })).toThrow( + 'no live claude stream-json session' + ) + }) + + it('does not report a second release as successful while retained exit evidence is unproven', async () => { + const claude = fakeClaude({ unprovenCloseVerdict: { root: 'exited', tree: 'unverifiable' } }) + const adapter = await acquired(claude) + const connection = claude.connections[0] + connection.handlers.onExit?.(new Error('claude stream-json exited (code 1): crashed')) + connection.close = vi.fn().mockResolvedValue(false) as unknown as FakeConnection['close'] + + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).rejects.toBeInstanceOf( + AgentSessionAcquisitionRootExitObservedError + ) + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).rejects.toBeInstanceOf( + AgentSessionAcquisitionRootExitObservedError + ) + expect(connection.close).toHaveBeenCalledTimes(2) + }) + + it('keeps shutdown pending until a retained unexpected-exit proof settles', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + const connection = claude.connections[0] + const proof = Promise.withResolvers() + connection.close = vi + .fn<() => Promise>() + .mockImplementationOnce(() => proof.promise) + .mockResolvedValueOnce(true) as unknown as FakeConnection['close'] + + connection.handlers.onExit?.(new Error('crashed')) + await tick() + let settled = false + const closing = adapter.closeAll().then(() => { + settled = true + }) + await tick() + expect(settled).toBe(false) + + proof.resolve(false) + await expect(closing).resolves.toBeUndefined() + expect(connection.close).toHaveBeenCalledTimes(2) + }) + + it('does not claim shutdown success for a retained false exit proof', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + const connection = claude.connections[0] + connection.close = vi + .fn<() => Promise>() + .mockResolvedValue(false) as unknown as FakeConnection['close'] + + connection.handlers.onExit?.(new Error('crashed')) + await tick() + + await expect(adapter.closeAll()).rejects.toThrow( + 'claude structured session shutdown could not prove every child stopped' + ) + expect(events.filter((event) => event.type === 'ended')).toEqual([]) + expect(connection.close).toHaveBeenCalledTimes(4) + }) +}) + +describe('ClaudeStructuredSessionAdapter prompts', () => { + it('turns can_use_tool into an addressable durable approval that settles the SDK callback', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + const answered = invokeCanUseTool(claude.connections[0], 'Bash', 'permission-1', 'tool-1', { + input: { command: 'git status' }, + suggestions: [{ type: 'addRules' }] + }) + expect(events.at(-1)).toMatchObject({ + type: 'prompt', + prompt: { kind: 'approval', toolName: 'Bash', promptKey: 'permission-1' } + }) + + adapter.bindPromptItemId('session-1', 'journal-approval', 'permission-1') + await adapter.answerPrompt({ + sessionId: 'session-1', + itemId: 'journal-approval', + kind: 'approval', + optionId: 'allowForSession', + fence: 7 + }) + // The answer resolves the SDK's own callback promise; the SDK writes the wire response. + await expect(answered.promise).resolves.toEqual({ + behavior: 'allow', + updatedInput: { command: 'git status' }, + updatedPermissions: [{ type: 'addRules' }], + toolUseID: 'tool-1' + }) + }) + + it('collects every AskUserQuestion card before settling the one callback', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + const answered = invokeCanUseTool( + claude.connections[0], + 'AskUserQuestion', + 'question-1', + 'tool-question', + { + input: { + questions: [ + { question: 'Library?', options: [{ label: 'Luxon' }] }, + { question: 'Ship now?', options: [{ label: 'Yes' }] } + ] + } + } + ) + adapter.bindPromptItemId('session-1', 'journal-q1', 'question-1', 'Library?') + adapter.bindPromptItemId('session-1', 'journal-q2', 'question-1', 'Ship now?') + + await adapter.answerPrompt({ + sessionId: 'session-1', + itemId: 'journal-q1', + kind: 'question', + optionId: encodeClaudeQuestionOptionId('Library?', 'Luxon'), + fence: 7 + }) + await tick() + expect(answered.settled()).toBe(false) + await adapter.answerPrompt({ + sessionId: 'session-1', + itemId: 'journal-q2', + kind: 'question', + optionId: encodeClaudeQuestionOptionId('Ship now?', 'Yes'), + fence: 7 + }) + await expect(answered.promise).resolves.toMatchObject({ + behavior: 'allow', + updatedInput: { answers: { 'Library?': 'Luxon', 'Ship now?': 'Yes' } }, + toolUseID: 'tool-question' + }) + }) + + it('leaves a prompt cancelled and unanswerable once the SDK abort signal fires', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + const controller = new AbortController() + const answered = invokeCanUseTool(claude.connections[0], 'Bash', 'permission-9', 'tool-9', { + input: { command: 'rm -rf /' }, + signal: controller.signal + }) + adapter.bindPromptItemId('session-1', 'journal-9', 'permission-9') + + controller.abort() + // A cancelled request is forgotten and settled with null — never an authorization. + await expect(answered.promise).resolves.toBeNull() + expect(events.at(-1)).toMatchObject({ type: 'prompt-cancelled', promptKey: 'permission-9' }) + // A late answer after the abort must not authorize the wrong tool. + await expect( + adapter.answerPrompt({ + sessionId: 'session-1', + itemId: 'journal-9', + kind: 'approval', + optionId: 'allow', + fence: 7 + }) + ).rejects.toThrow(/no longer waiting/) + }) + + it('settles an in-flight permission callback when the session closes, leaving no dangling promise', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + const answered = invokeCanUseTool(claude.connections[0], 'Bash', 'permission-close', 'tool-c', { + input: { command: 'ls' } + }) + await tick() + expect(answered.settled()).toBe(false) + + await adapter.closeSession('session-1') + + await expect(answered.promise).resolves.toBeNull() + }) +}) diff --git a/src/main/claude/claude-structured-session-adapter.ts b/src/main/claude/claude-structured-session-adapter.ts new file mode 100644 index 00000000000..9bd92e1e839 --- /dev/null +++ b/src/main/claude/claude-structured-session-adapter.ts @@ -0,0 +1,306 @@ +import type { + AgentSessionAcquisition, + StructuredAgentSessionAcquireInput, + StructuredAgentSessionAdapter +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { + answerClaudePrompt, + cancelClaudeTurn, + stopClaudeBackgroundTasks +} from './claude-structured-control-actions' +import { dispatchClaudeTurn } from './claude-structured-dispatch' +import { releaseClaudeAcquisition } from './claude-structured-acquisition-release' +import { acquireClaudeSession } from './claude-structured-session-acquisition' +export { CLAUDE_STRUCTURED_INIT_TIMEOUT_MS } from './claude-structured-session-acquisition' +import { supportsClaudeStructuredLocation } from './claude-structured-location-support' +import { setClaudeStructuredOption } from './claude-structured-options' +import { readClaudeStructuredSessionOptions } from './claude-structured-session-options' +import { + ClaudeAcquisitionRegistry, + type ClaudeAcquisitionAttempt, + type ClaudeSession, + type ClaudeSessionExit, + type ClaudeStructuredSessionAdapterDeps, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-state' +import { + closeAllClaudeSessions, + closeClaudeSession, + settleClaudeExitedSession +} from './claude-structured-session-close' +import { readClaudeTranscriptLeafWithReproof } from './claude-transcript-branch-proof' + +export type { ClaudeStructuredLaunch } from './claude-structured-launch-resolution' +export type { + ClaudeAuthDiagnostic, + ClaudeStructuredSessionAdapterDeps, + ClaudeStructuredSessionEvent +} from './claude-structured-session-state' + +const DISPATCH_ACK_TIMEOUT_MS = 10_000 + +export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAdapter { + private readonly sessions = new Map() + private readonly acquisitions = new ClaudeAcquisitionRegistry() + private readonly exits = new Map() + + constructor(private readonly deps: ClaudeStructuredSessionAdapterDeps) {} + + supportsLocation = supportsClaudeStructuredLocation + + acquire = (input: StructuredAgentSessionAcquireInput): Promise => + acquireClaudeSession({ + input, + deps: this.deps, + sessions: this.sessions, + acquisitions: this.acquisitions, + exits: this.exits, + callbacks: { + deliver: (attempt, sessionId, event) => this.deliver(attempt, sessionId, event), + emit: (session, events, event) => this.emit(session, events, event), + handleExit: (sessionId, attempt, error) => this.handleExit(sessionId, attempt, error), + settleExit: (sessionId, exit) => this.settleUnexpectedExit(sessionId, exit) + } + }) + + private deliver(attempt: ClaudeAcquisitionAttempt, sessionId: string, event: () => void): void { + if (!attempt.published) { + attempt.buffered.push(event) + return + } + if (this.sessions.get(sessionId)?.connection === attempt.connection) { + event() + } + } + + private handleExit(sessionId: string, attempt: ClaudeAcquisitionAttempt, error: Error): void { + const session = this.sessions.get(sessionId) + if (!session || session.connection !== attempt.connection) { + return + } + this.sessions.delete(sessionId) + // Re-enter the provider's close ladder before publishing lifecycle recovery. + // An exit callback is root evidence only; the retained tree proof must run + // before the host releases and reacquires this exact child. + const closePromise = session.connection.close().catch(() => false) + const exit: ClaudeSessionExit = { + connection: session.connection, + session, + error, + closePromise + } + this.exits.set(sessionId, exit) + exit.publication = closePromise + .then((proven) => (proven ? this.settleUnexpectedExit(sessionId, exit) : undefined)) + .catch(() => undefined) + } + + /** Resolves once every first-hand exit observed so far has published its + * lifecycle event — or has failed its tree proof and stayed indexed for a + * retry. Publication trails observation by the close ladder and the + * transcript cursor write, so nothing outside can otherwise tell the two + * apart without guessing at wall-clock. */ + drainObservedExits = async (): Promise => { + const awaited = new Set>() + for (;;) { + const pending = [...this.exits.values()] + .map((exit) => exit.publication) + .filter( + (publication): publication is Promise => + publication !== undefined && !awaited.has(publication) + ) + if (pending.length === 0) { + return + } + for (const publication of pending) { + awaited.add(publication) + } + // A publication can settle an exit that itself observes another; only the + // ones this pass has not already awaited keep the loop going. + await Promise.all(pending) + } + } + + /** Lifecycle recovery is published only after the child tree proof is true. */ + private settleUnexpectedExit(sessionId: string, exit: ClaudeSessionExit): Promise { + exit.settlementPromise ??= (async () => { + if (this.exits.get(sessionId) !== exit) { + settleClaudeExitedSession(exit.session) + return + } + // Persist the transcript-derived cursor before publishing the lifecycle + // event that lets the host release and reacquire this exact child. + await this.persistSessionHandle(sessionId, exit.session).catch(() => undefined) + if (this.exits.get(sessionId) !== exit) { + settleClaudeExitedSession(exit.session) + return + } + this.exits.delete(sessionId) + const ended: ClaudeStructuredSessionEvent = { + type: 'ended', + sessionId, + reason: exit.error.message, + cause: 'unexpected-exit', + fence: exit.session.fence, + acquisitionGeneration: exit.session.acquisitionGeneration + } + try { + this.emit(exit.session, exit.session.events, ended) + } finally { + settleClaudeExitedSession(exit.session) + } + })() + return exit.settlementPromise + } + + private async persistSessionHandle(sessionId: string, session: ClaudeSession): Promise { + try { + const transcriptLeaf = this.deps.readTranscriptLeaf + ? await readClaudeTranscriptLeafWithReproof({ + readTranscriptLeaf: this.deps.readTranscriptLeaf, + providerSessionId: session.providerSessionId, + previousLeafUuid: session.leafUuid, + claudeConfigDir: session.claudeConfigDir + }) + : null + if (transcriptLeaf) { + session.leafUuid = transcriptLeaf + } + } catch { + // A stale or unavailable tail must not overwrite the last observed leaf. + } + await this.deps.persistHandle?.({ + sessionId, + providerSessionId: session.providerSessionId, + leafUuid: session.leafUuid, + fence: session.fence + }) + } + + private emit( + session: ClaudeSession | null, + _events: StructuredAgentSessionEventSink | undefined, + event: ClaudeStructuredSessionEvent + ): void { + const backgroundTasksChanged = + event.type === 'ended' + ? (session?.backgroundTasks.clear() ?? false) + : event.type === 'message' + ? (session?.backgroundTasks.observe(event.message, event.startsTurn === true) ?? false) + : false + session?.translator?.handle(event) + this.deps.onEvent?.(event) + if (backgroundTasksChanged) { + this.deps.onBackgroundTasksChanged?.(event.sessionId, session?.backgroundTasks.state ?? null) + } + } + + bindPromptItemId( + sessionId: string, + journalItemId: string, + promptKey: string, + questionId?: string + ): void { + this.sessions.get(sessionId)?.prompts.bindJournalItemId(journalItemId, promptKey, questionId) + } + + dispatch: StructuredAgentSessionAdapter['dispatch'] = (input) => + dispatchClaudeTurn( + this.session(input.sessionId), + input, + this.deps.dispatchAckTimeoutMs ?? DISPATCH_ACK_TIMEOUT_MS + ) + + cancelTurn: StructuredAgentSessionAdapter['cancelTurn'] = (input) => { + const session = this.session(input.sessionId) + const acquisitionGeneration = session.acquisitionGeneration + return cancelClaudeTurn(session, this.deps.requestTimeoutMs, () => { + // Keep every ownership check adjacent to the provider interrupt. The + // session map check fences a replaced child; the turn check fences a + // delayed cancel after a newer turn was admitted on the same child. + return ( + this.sessions.get(input.sessionId) === session && + session.fence === input.fence && + session.acquisitionGeneration === acquisitionGeneration && + (session.activeTurnId === undefined + ? session.dispatchSequence === 0 + : session.activeTurnId === input.turnId && + session.activeTurnSequence === session.dispatchSequence) + ) + }) + } + stopBackgroundTasks: StructuredAgentSessionAdapter['stopBackgroundTasks'] = (input) => { + const session = this.session(input.sessionId) + const acquisitionGeneration = session.acquisitionGeneration + return stopClaudeBackgroundTasks(session, this.deps.requestTimeoutMs, () => + Boolean( + this.sessions.get(input.sessionId) === session && + session.fence === input.fence && + session.acquisitionGeneration === acquisitionGeneration && + session.backgroundTasks.state + ) + ) + } + backgroundTaskState: NonNullable = ( + sessionId + ) => this.sessions.get(sessionId)?.backgroundTasks.state + answerPrompt: StructuredAgentSessionAdapter['answerPrompt'] = (input) => + answerClaudePrompt(this.session(input.sessionId), input) + setOption: StructuredAgentSessionAdapter['setOption'] = (input) => + setClaudeStructuredOption(this.session(input.sessionId), input, this.deps.requestTimeoutMs) + readOptions = (input: { sessionId: string; fence: number }) => + readClaudeStructuredSessionOptions(this.session(input.sessionId), this.deps.requestTimeoutMs) + + readOptionRestoreFailures = (sessionId: string): readonly string[] => [ + ...(this.sessions.get(sessionId)?.restoreSkippedOptions ?? []) + ] + + releaseAcquisition = (input: { sessionId: string }): Promise => + releaseClaudeAcquisition({ + sessionId: input.sessionId, + sessions: this.sessions, + acquisitions: this.acquisitions, + exits: this.exits, + onExitProven: (sessionId, exit) => this.settleUnexpectedExit(sessionId, exit), + ...(this.deps.persistHandle ? { persistHandle: this.deps.persistHandle } : {}), + ...(this.deps.onBackgroundTasksChanged + ? { onBackgroundTasksChanged: this.deps.onBackgroundTasksChanged } + : {}), + ...(this.deps.onEvent ? { onEvent: this.deps.onEvent } : {}) + }) + + closeSession = (sessionId: string): Promise => { + if (this.exits.has(sessionId)) { + return this.releaseAcquisition({ sessionId }) + } + return closeClaudeSession({ + sessionId, + sessions: this.sessions, + acquisitions: this.acquisitions, + ...(this.deps.persistHandle ? { persistHandle: this.deps.persistHandle } : {}), + ...(this.deps.readTranscriptLeaf ? { readTranscriptLeaf: this.deps.readTranscriptLeaf } : {}), + ...(this.deps.onBackgroundTasksChanged + ? { onBackgroundTasksChanged: this.deps.onBackgroundTasksChanged } + : {}), + ...(this.deps.onEvent ? { onEvent: this.deps.onEvent } : {}) + }) + } + + closeAll = (): Promise => + closeAllClaudeSessions({ + sessions: this.sessions, + acquisitions: this.acquisitions, + exits: this.exits, + closeSession: this.closeSession, + closeExit: (sessionId) => this.releaseAcquisition({ sessionId }) + }) + + private session(sessionId: string): ClaudeSession { + const session = this.sessions.get(sessionId) + if (!session) { + throw new Error(`no live claude stream-json session for ${sessionId}`) + } + return session + } +} diff --git a/src/main/claude/claude-structured-session-close.test.ts b/src/main/claude/claude-structured-session-close.test.ts new file mode 100644 index 00000000000..0df0e913e50 --- /dev/null +++ b/src/main/claude/claude-structured-session-close.test.ts @@ -0,0 +1,82 @@ +import { describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import type { + ClaudeStructuredSessionAdapterDeps, + ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' +import { + PROVIDER_SESSION_ID, + adapterFor, + fakeClaude, + identityFor +} from './claude-structured-session-test-support' +import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire' + +describe('Claude published session close lifecycle', () => { + it('ends the session even when the durable handle write rejects', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistenceError = new Error('store unavailable') + const persistHandle = vi + .fn>() + .mockRejectedValueOnce(persistenceError) + .mockResolvedValueOnce(undefined) + const backgroundStates: (AgentSessionBackgroundTaskState | null)[] = [] + const adapter = adapterFor( + claude, + {}, + events, + [], + undefined, + undefined, + persistHandle, + (_sessionId, state) => backgroundStates.push(state) + ) + const journalSink: StructuredAgentSessionEventSink = { + appendItem: () => {}, + appendTombstone: () => {}, + publish: () => {} + } + await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + events: journalSink + }) + claude.connections[0]!.handlers.onMessage?.({ + type: 'system', + subtype: 'task_started', + session_id: PROVIDER_SESSION_ID, + uuid: 'task-start', + task_id: 'background-1', + task_type: 'local_agent', + is_backgrounded: true + }) + expect(backgroundStates).toEqual([ + { state: 'monitoring', tasks: [{ id: 'background-1', kind: 'agent' }] } + ]) + const session = ( + adapter as unknown as { + sessions: Map void } | null }> + } + ).sessions.get('session-1') + const disposeTranslator = vi.spyOn(session!.translator!, 'dispose') + + await expect(adapter.closeSession('session-1')).rejects.toBe(persistenceError) + // The child is provably dead; a failed cursor write may not suppress the end. + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + expect(events.filter((event) => event.type === 'handle')).toHaveLength(0) + expect(disposeTranslator).toHaveBeenCalledOnce() + expect(backgroundStates).toEqual([ + { state: 'monitoring', tasks: [{ id: 'background-1', kind: 'agent' }] }, + null + ]) + + await expect(adapter.closeSession('session-1')).resolves.toBe(true) + expect(persistHandle).toHaveBeenCalledTimes(2) + // The retry persists the same cursor without a second lifecycle end. + expect(events.filter((event) => event.type === 'handle')).toHaveLength(1) + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + expect(disposeTranslator).toHaveBeenCalledOnce() + }) +}) diff --git a/src/main/claude/claude-structured-session-close.ts b/src/main/claude/claude-structured-session-close.ts new file mode 100644 index 00000000000..de07439919d --- /dev/null +++ b/src/main/claude/claude-structured-session-close.ts @@ -0,0 +1,272 @@ +import type { + ClaudeAcquisitionRegistry, + ClaudeSession, + ClaudeSessionExit, + ClaudeStructuredSessionEvent +} from './claude-structured-session-state' +import { cancelClaudeAcquisitionAttempt } from './claude-structured-session-state' +import { + AgentSessionAcquisitionExitUnprovenError, + AgentSessionAcquisitionRootExitObservedError, + AgentSessionPreSpawnError +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { ClaudeStreamJsonConnection } from './claude-stream-json-connection' +import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire' +import { closeProcessRegistry } from '../../shared/child-process/close-process-registry' +import { readClaudeTranscriptLeafWithReproof } from './claude-transcript-branch-proof' + +export function claudeAcquisitionCleanupError( + connection: ClaudeStreamJsonConnection | null | undefined, + cause: unknown +): Error { + const verdict = connection?.exitVerdict + if (verdict?.root === 'processless') { + return new AgentSessionPreSpawnError(cause) + } + return verdict?.root === 'exited' && verdict.tree === 'unverifiable' + ? new AgentSessionAcquisitionRootExitObservedError(cause) + : new AgentSessionAcquisitionExitUnprovenError(cause) +} + +export function settleClaudeDispatchWaiters(session: ClaudeSession): void { + for (const waiter of session.dispatchWaiters.splice(0)) { + clearTimeout(waiter.timer) + waiter.resolve(null) + } +} + +export function settleClaudeExitedSession(session: ClaudeSession): void { + settleClaudeDispatchWaiters(session) + for (const prompt of session.prompts.clear()) { + prompt.settle(null) + } + session.translator?.dispose() +} + +type CloseClaudePublishedSessionInput = { + sessions: Map + sessionId: string + persistHandle?: (handle: { + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + }) => Promise + onEvent?: (event: ClaudeStructuredSessionEvent) => void + onBackgroundTasksChanged?: ( + sessionId: string, + state: AgentSessionBackgroundTaskState | null + ) => void + readTranscriptLeaf?: (input: { + providerSessionId: string + previousLeafUuid: string | null + claudeConfigDir: string + }) => Promise +} + +async function finalizeClaudePublishedSession( + input: CloseClaudePublishedSessionInput, + session: ClaudeSession +): Promise { + settleClaudeDispatchWaiters(session) + // Settle every in-flight permission callback so closing leaves no dangling promise; `null` + // writes no response, and the SDK ignores any post-cleanup answer regardless. + for (const prompt of session.prompts.clear()) { + prompt.settle(null) + } + if ((await session.connection.close()) !== true) { + return false + } + if (session.backgroundTasks.clear()) { + input.onBackgroundTasksChanged?.(input.sessionId, null) + } + try { + const transcriptLeaf = input.readTranscriptLeaf + ? await readClaudeTranscriptLeafWithReproof({ + readTranscriptLeaf: input.readTranscriptLeaf, + providerSessionId: session.providerSessionId, + previousLeafUuid: session.leafUuid, + claudeConfigDir: session.claudeConfigDir + }) + : null + if (transcriptLeaf) { + session.leafUuid = transcriptLeaf + } + } catch { + // Keep the last observed main-transcript frame when the durable tail is + // unavailable or proves a stale/divergent branch. + } + const persistence = + session.closePersistence ?? + (session.closePersistence = (async () => { + await input.persistHandle?.({ + sessionId: input.sessionId, + providerSessionId: session.providerSessionId, + leafUuid: session.leafUuid, + fence: session.fence + }) + })()) + const ended = { + type: 'ended', + sessionId: input.sessionId, + reason: 'claude session closed' + } as const + let callbackError: unknown + let callbackThrew = false + const deliver = (event: ClaudeStructuredSessionEvent): void => { + try { + input.onEvent?.(event) + } catch (error) { + callbackThrew = true + callbackError ??= error + } + } + let persistenceError: unknown + try { + await persistence + session.closeFinalized = true + input.sessions.delete(input.sessionId) + deliver({ + type: 'handle', + sessionId: input.sessionId, + providerSessionId: session.providerSessionId, + leafUuid: session.leafUuid, + fence: session.fence + }) + } catch (error) { + // Keep the closed session indexed so a retry can persist the same cursor. + // Removing it first would turn a durable-write failure into a no-op retry. + if (session.closePersistence === persistence) { + session.closePersistence = undefined + } + persistenceError = error + } + // The connection already proved the child dead, so the session has ended + // whatever the durable write did: withholding it would strand the renderer on + // a session nothing re-drives. Emitted once, so a retry only re-persists. + if (!session.closeEnded) { + session.closeEnded = true + try { + try { + session.translator?.handle(ended) + } catch (error) { + callbackThrew = true + callbackError ??= error + } + deliver(ended) + } finally { + session.translator?.dispose() + } + } + if (persistenceError) { + throw persistenceError + } + if (callbackThrew) { + throw callbackError + } + return true +} + +export async function closeClaudePublishedSession( + input: CloseClaudePublishedSessionInput +): Promise { + const session = input.sessions.get(input.sessionId) + if (!session) { + return true + } + if (session.closeFinalized) { + return true + } + if (session.closeFinalization) { + return session.closeFinalization + } + const finalization = finalizeClaudePublishedSession(input, session) + session.closeFinalization = finalization + try { + return await finalization + } finally { + if (session.closeFinalization === finalization && !session.closeFinalized) { + session.closeFinalization = undefined + } + } +} + +export function closeClaudePublishedSessionForDeps( + sessions: Map, + sessionId: string, + deps: { + persistHandle?: (handle: { + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + }) => Promise + onEvent?: (event: ClaudeStructuredSessionEvent) => void + onBackgroundTasksChanged?: ( + sessionId: string, + state: AgentSessionBackgroundTaskState | null + ) => void + readTranscriptLeaf?: (input: { + providerSessionId: string + previousLeafUuid: string | null + claudeConfigDir: string + }) => Promise + } +): Promise { + return closeClaudePublishedSession({ sessions, sessionId, ...deps }) +} + +export async function closeClaudeSession(input: { + sessionId: string + sessions: Map + acquisitions: ClaudeAcquisitionRegistry + persistHandle?: (handle: { + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + }) => Promise + onEvent?: (event: ClaudeStructuredSessionEvent) => void + onBackgroundTasksChanged?: ( + sessionId: string, + state: AgentSessionBackgroundTaskState | null + ) => void + readTranscriptLeaf?: (input: { + providerSessionId: string + previousLeafUuid: string | null + claudeConfigDir: string + }) => Promise +}): Promise { + const attempt = input.acquisitions.get(input.sessionId) + if (!(await cancelClaudeAcquisitionAttempt(attempt))) { + return false + } + if (attempt) { + input.acquisitions.deleteIfCurrent(input.sessionId, attempt) + } + return closeClaudePublishedSession(input) +} + +export async function closeAllClaudeSessions(input: { + sessions: Map + acquisitions: ClaudeAcquisitionRegistry + exits: Map + closeSession: (sessionId: string) => Promise + closeExit: (sessionId: string) => Promise +}): Promise { + input.acquisitions.close() + await closeProcessRegistry({ + attempts: 3, + hasEntries: () => + input.sessions.size > 0 || input.acquisitions.size > 0 || input.exits.size > 0, + entryIds: () => + new Set([ + ...input.sessions.keys(), + ...input.acquisitions.sessionIds(), + ...input.exits.keys() + ]), + closeEntry: async (sessionId) => + input.exits.has(sessionId) ? input.closeExit(sessionId) : input.closeSession(sessionId), + failureMessage: 'claude structured session shutdown could not prove every child stopped' + }) +} diff --git a/src/main/claude/claude-structured-session-options.ts b/src/main/claude/claude-structured-session-options.ts new file mode 100644 index 00000000000..afb4fd65076 --- /dev/null +++ b/src/main/claude/claude-structured-session-options.ts @@ -0,0 +1,183 @@ +import type { + AgentSessionModelOption, + AgentSessionOptionChoice, + AgentSessionOptionsResult +} from '../../shared/agent-session-wire' +import { CLAUDE_SESSION_OPTION_CATALOG } from '../../shared/agent-session-option-catalog-claude-codex' +import type { CatalogModel } from '../../shared/agent-session-option-catalog-types' +import type { ClaudeSession } from './claude-structured-session-state' + +type ListedModel = AgentSessionModelOption & { resolvedModel: string | null } + +function record(value: unknown): Record | null { + return typeof value === 'object' && value !== null && !Array.isArray(value) + ? (value as Record) + : null +} + +function text(value: unknown): string | null { + return typeof value === 'string' && value.trim() ? value : null +} + +/** + * The session's current effort, which only `get_settings` reports: the + * `system/init` frame carries `model` but has never carried an effort of any + * kind. Null when the provider stops reporting it, so the pill goes empty + * rather than showing an effort nothing measured. + */ +export function readClaudeSettingsEffort(settings: unknown): string | null { + return text(record(record(settings)?.effective)?.effortLevel) +} + +function effortLabel(value: string): string { + return value === 'xhigh' ? 'Extra high' : `${value.charAt(0).toUpperCase()}${value.slice(1)}` +} + +function listedEfforts(row: Record): AgentSessionOptionChoice[] { + return row.supportsEffort === true && Array.isArray(row.supportedEffortLevels) + ? row.supportedEffortLevels.flatMap((value) => { + const effort = text(value) + return effort ? [{ value: effort, label: effortLabel(effort) }] : [] + }) + : [] +} + +function listedModels(value: unknown): ListedModel[] { + const response = record(value) + const rows = Array.isArray(response?.models) + ? response.models.map(record).filter((row): row is Record => row !== null) + : [] + const defaultRow = rows.find((row) => text(row.value) === 'default') + const defaultResolvedModel = text(defaultRow?.resolvedModel) + const seen = new Set() + return rows.flatMap((row) => { + const id = text(row.value) + if (!id || id === 'default' || seen.has(id)) { + return [] + } + seen.add(id) + const resolvedModel = text(row.resolvedModel) + const description = text(row.description) + return [ + { + id, + label: text(row.displayName) ?? id, + ...(description ? { description } : {}), + isDefault: resolvedModel !== null && resolvedModel === defaultResolvedModel, + efforts: listedEfforts(row), + resolvedModel + } + ] + }) +} + +function seedEfforts(model: CatalogModel): AgentSessionOptionChoice[] { + const effort = model.options.find((option) => option.id === 'effort') + return effort?.kind.type === 'select' ? effort.kind.choices : [] +} + +function seedModels(): ListedModel[] { + return CLAUDE_SESSION_OPTION_CATALOG.models.map((model) => ({ + id: model.id, + label: model.label, + ...(model.description ? { description: model.description } : {}), + isDefault: model.isDefault === true, + efforts: seedEfforts(model), + resolvedModel: null + })) +} + +function currentModelId(models: ListedModel[], reportedModel: string | undefined): string { + const matched = reportedModel + ? models.find((model) => model.id === reportedModel || model.resolvedModel === reportedModel) + : undefined + return ( + matched?.id ?? reportedModel ?? models.find((model) => model.isDefault)?.id ?? models[0]!.id + ) +} + +/** + * The model the session is running. A report the CLI made after the last write + * outranks the write: it names the model the session ran. An older one does not + * — a model set between turns has no report yet, and deferring to the previous + * turn's would flip the pill back. + * + * Sole resolver of that question: every surface that acts on "the current model" + * — the pill, the effort guard, the rejection it names — reads it here, so two + * of them cannot answer it differently and offer an effort a third then refuses. + */ +export function readClaudeCurrentModel(session: ClaudeSession): { + id: string | undefined + confirmed: boolean +} { + const confirmed = + session.reportedModelMutation === session.optionMutationSequence && + session.reportedOptions.model !== undefined + return { + id: confirmed + ? session.reportedOptions.model + : (session.options.get('model') ?? session.reportedOptions.model), + confirmed + } +} + +/** + * The effort levels the session's current model advertises, with the catalog id + * that matched so a refusal names the model the pill shows. Levels are null when + * nothing identified the model: `apply_flag_settings` accepts and stores any + * level for a model with no effort control, so the catalog is the only evidence + * of a refusal — and an absent or unlisted one is not evidence, or a live CLI + * that predates `list_models` would have every effort refused under it. + */ +export async function readClaudeModelEffortLevels( + session: ClaudeSession, + timeoutMs: number | undefined +): Promise<{ modelId: string | undefined; levels: ReadonlySet | null }> { + const modelId = readClaudeCurrentModel(session).id + if (!modelId) { + return { modelId, levels: null } + } + const catalog = await session.connection.supportedModels({ timeoutMs }).catch(() => null) + const matched = catalog + ? listedModels({ models: catalog }).find( + (model) => model.id === modelId || model.resolvedModel === modelId + ) + : undefined + return { + modelId: matched?.id ?? modelId, + levels: matched ? new Set(matched.efforts.map((choice) => choice.value)) : null + } +} + +export async function readClaudeStructuredSessionOptions( + session: ClaudeSession, + timeoutMs: number | undefined +): Promise { + const catalog = await session.connection.supportedModels({ timeoutMs }).catch(() => null) + const discovered = listedModels(catalog ? { models: catalog } : null) + const models = discovered.length > 0 ? discovered : seedModels() + const current = readClaudeCurrentModel(session) + const model = currentModelId(models, current.id) + if (!models.some((entry) => entry.id === model)) { + models.push({ id: model, label: model, isDefault: false, efforts: [], resolvedModel: null }) + } + const effort = session.options.get('effort') ?? session.reportedOptions.effort + const confirmed = [ + ...(current.confirmed ? ['model'] : []), + ...(effort && session.confirmedOptions.has('effort') ? ['effort'] : []) + ] + return { + models: models.map((entry) => ({ + id: entry.id, + label: entry.label, + ...(entry.description ? { description: entry.description } : {}), + isDefault: entry.isDefault, + efforts: entry.efforts + })), + current: { + model, + ...(effort ? { effort } : {}), + ...(confirmed.length > 0 ? { confirmed } : {}) + } + } +} diff --git a/src/main/claude/claude-structured-session-publication.ts b/src/main/claude/claude-structured-session-publication.ts new file mode 100644 index 00000000000..7f8cc4b5692 --- /dev/null +++ b/src/main/claude/claude-structured-session-publication.ts @@ -0,0 +1,70 @@ +import type { AgentSessionAcquisition } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { ClaudeInitObservation } from './claude-structured-init-proof' +import { claudeProviderHandleLink } from './claude-structured-owner-identity' +import type { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import type { ClaudeJournalTranslator } from './claude-structured-journal-translation' +import type { ClaudeSession } from './claude-structured-session-state' +import { ClaudeBackgroundTaskTracker } from './claude-background-task-tracker' + +export function createClaudeSessionPublication(input: { + connection: ClaudeSession['connection'] + init: ClaudeInitObservation + claudeConfigDir: string + leafUuid: string | null + fence: number + acquisitionGeneration: string + resumed: boolean + prompts: ClaudePromptRegistry + translator: ClaudeJournalTranslator | null + events: ClaudeSession['events'] + process: AgentSessionAcquisition['process'] + linkId?: string + observedAt: number + options?: ReadonlyMap + capabilities: readonly string[] + /** Read from `get_settings`; `system/init` never reports an effort. */ + effort: string | null +}): { acquisition: AgentSessionAcquisition; session: ClaudeSession } { + const model = input.init.model + const effort = input.effort + return { + acquisition: { + process: input.process, + link: claudeProviderHandleLink({ + sessionId: input.init.providerSessionId, + leafUuid: input.leafUuid, + resumed: input.resumed, + fence: input.fence, + ...(input.linkId ? { linkId: input.linkId } : {}), + observedAt: input.observedAt + }), + acquisitionGeneration: input.acquisitionGeneration + }, + session: { + connection: input.connection, + providerSessionId: input.init.providerSessionId, + claudeConfigDir: input.claudeConfigDir, + leafUuid: input.leafUuid, + fence: input.fence, + acquisitionGeneration: input.acquisitionGeneration, + prompts: input.prompts, + dispatchWaiters: [], + retiredDispatchWaiters: [], + replayContentFallbackBlocked: false, + backgroundTasks: new ClaudeBackgroundTaskTracker(), + dispatchSequence: 0, + optionMutationSequence: 0, + options: new Map(input.options), + capabilities: input.capabilities, + reportedOptions: { + ...(model ? { model } : {}), + ...(effort ? { effort } : {}) + }, + reportedModelMutation: 0, + confirmedOptions: new Set(effort ? ['effort'] : []), + restoreSkippedOptions: new Set(), + translator: input.translator, + events: input.events + } + } +} diff --git a/src/main/claude/claude-structured-session-recovery.test.ts b/src/main/claude/claude-structured-session-recovery.test.ts new file mode 100644 index 00000000000..5bfe56cf156 --- /dev/null +++ b/src/main/claude/claude-structured-session-recovery.test.ts @@ -0,0 +1,619 @@ +import { describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionAdapterDeps, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' +import { ClaudeTranscriptPreviousCursorMissingError } from './claude-transcript-branch-proof' +import { + adapterFor, + fakeClaude, + identityFor, + invokeCanUseTool, + PROVIDER_SESSION_ID, + tick +} from './claude-structured-session-test-support' + +describe('ClaudeStructuredSessionAdapter transcript-derived recovery', () => { + it('shares concurrent close finalization and emits lifecycle once', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistence = Promise.withResolvers() + const persistHandle = vi.fn(() => persistence.promise) + const adapter = adapterFor(claude, {}, events, [], undefined, undefined, persistHandle) + const journalSink: StructuredAgentSessionEventSink = { + appendItem: () => {}, + appendTombstone: () => {}, + publish: () => {} + } + await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + events: journalSink + }) + const session = ( + adapter as unknown as { + sessions: Map void } | null }> + } + ).sessions.get('session-1') + const disposeTranslator = vi.spyOn(session!.translator!, 'dispose') + + const first = adapter.closeSession('session-1') + const second = adapter.closeSession('session-1') + await tick() + expect(persistHandle).toHaveBeenCalledOnce() + expect(claude.connections[0].closeCount).toBe(1) + + persistence.resolve() + await expect(Promise.all([first, second])).resolves.toEqual([true, true]) + expect(events.filter((event) => event.type === 'handle')).toHaveLength(1) + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + expect(disposeTranslator).toHaveBeenCalledOnce() + }) + + it('still emits ended and disposes state when handle delivery throws', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const callbackError = new Error('handle delivery failed') + const adapter = new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: null, + resumed: false + }), + onEvent: (event) => { + events.push(event) + if (event.type === 'handle') { + throw callbackError + } + }, + openConnection: claude.openConnection, + readProcessStartTime: async () => 1_700_000_000_000, + persistHandle: vi.fn(async () => undefined) + }) + const journalSink: StructuredAgentSessionEventSink = { + appendItem: () => {}, + appendTombstone: () => {}, + publish: () => {} + } + await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + events: journalSink + }) + const session = ( + adapter as unknown as { + sessions: Map void } | null }> + } + ).sessions.get('session-1') + const disposeTranslator = vi.spyOn(session!.translator!, 'dispose') + + await expect(adapter.closeSession('session-1')).rejects.toBe(callbackError) + expect(events.filter((event) => event.type === 'handle')).toHaveLength(1) + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + expect(disposeTranslator).toHaveBeenCalledOnce() + }) + + it('retains a closed session until its durable cursor persistence succeeds', async () => { + const claude = fakeClaude() + const persistenceError = new Error('store unavailable') + const persistHandle = vi + .fn>() + .mockRejectedValueOnce(persistenceError) + .mockResolvedValueOnce(undefined) + const adapter = adapterFor(claude, {}, [], [], undefined, undefined, persistHandle) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + await expect(adapter.closeSession('session-1')).rejects.toBe(persistenceError) + expect(persistHandle).toHaveBeenCalledTimes(1) + await expect(adapter.closeSession('session-1')).resolves.toBe(true) + expect(persistHandle).toHaveBeenCalledTimes(2) + }) + + it('persists only the last transcript-entry uuid before graceful close', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistedHandles: unknown[] = [] + const adapter = adapterFor(claude, {}, events, persistedHandles) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'assistant-leaf' + }) + claude.connections[0].handlers.onMessage?.({ + type: 'result', + session_id: PROVIDER_SESSION_ID, + uuid: 'result-frame-uuid' + }) + claude.connections[0].handlers.onMessage?.({ + type: 'stream_event', + session_id: PROVIDER_SESSION_ID, + uuid: 'stream-event-frame-uuid' + }) + + await adapter.closeSession('session-1') + + expect(persistedHandles).toEqual([ + { + sessionId: 'session-1', + providerSessionId: PROVIDER_SESSION_ID, + leafUuid: 'assistant-leaf', + fence: 7 + } + ]) + expect(events.at(-2)).toEqual({ + type: 'handle', + sessionId: 'session-1', + providerSessionId: PROVIDER_SESSION_ID, + leafUuid: 'assistant-leaf', + fence: 7 + }) + expect(claude.connections[0].closeCount).toBe(1) + }) + + it('prefers a validated durable transcript leaf at graceful close', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi.fn().mockResolvedValue('durable-tail') + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-tail' + }) + + await adapter.closeSession('session-1') + + expect(readTranscriptLeaf).toHaveBeenCalledWith({ + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'observed-tail', + claudeConfigDir: '/accounts/claude' + }) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'durable-tail' }) + }) + + it('passes the pinned Claude account home to transcript validation', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi.fn().mockResolvedValue('durable-tail') + const adapter = adapterFor( + claude, + { claudeConfigDir: '/accounts/selected' }, + [], + persistedHandles, + undefined, + readTranscriptLeaf + ) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-tail' + }) + + await adapter.closeSession('session-1') + + expect(readTranscriptLeaf).toHaveBeenCalledWith({ + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'observed-tail', + claudeConfigDir: '/accounts/selected' + }) + }) + + it('re-proves from the transcript root when the observed cursor is missing', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi + .fn() + .mockRejectedValueOnce(new ClaudeTranscriptPreviousCursorMissingError()) + .mockResolvedValueOnce('reproved-main-leaf') + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-tail' + }) + + await adapter.closeSession('session-1') + + expect(readTranscriptLeaf).toHaveBeenNthCalledWith(1, { + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'observed-tail', + claudeConfigDir: '/accounts/claude' + }) + expect(readTranscriptLeaf).toHaveBeenNthCalledWith(2, { + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: null, + claudeConfigDir: '/accounts/claude' + }) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'reproved-main-leaf' }) + }) + + it('keeps the observed leaf when transcript validation proves a sibling branch', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi + .fn() + .mockRejectedValue(new Error('latest marker is on a sibling branch')) + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-tail' + }) + + await adapter.closeSession('session-1') + + expect(readTranscriptLeaf).toHaveBeenCalledTimes(1) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'observed-tail' }) + }) + + it('persists the last transcript leaf before an unexpected first-hand exit', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events, persistedHandles) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'crash-leaf' + }) + + claude.connections[0].handlers.onExit?.( + new Error('claude stream-json exited (code 1): crashed unexpectedly') + ) + await tick() + + expect(persistedHandles).toContainEqual({ + sessionId: 'session-1', + providerSessionId: PROVIDER_SESSION_ID, + leafUuid: 'crash-leaf', + fence: 7 + }) + expect(events.at(-1)).toMatchObject({ + type: 'ended', + cause: 'unexpected-exit', + fence: 7, + acquisitionGeneration: expect.any(String) + }) + }) + + it('derives the crash cursor from the validated transcript tail', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const adapter = adapterFor( + claude, + {}, + [], + persistedHandles, + undefined, + vi.fn().mockResolvedValue('durable-crash-leaf') + ) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'stale-observed-tail' + }) + claude.connections[0].handlers.onExit?.( + new Error('claude stream-json exited (signal SIGKILL): crashed') + ) + await tick() + + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'durable-crash-leaf' }) + }) + + it('re-proves a first-hand crash cursor from the transcript root after stale validation', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi + .fn() + .mockRejectedValueOnce(new ClaudeTranscriptPreviousCursorMissingError()) + .mockResolvedValueOnce('reproved-crash-leaf') + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'stale-observed-tail' + }) + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(readTranscriptLeaf).toHaveBeenNthCalledWith(1, { + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'stale-observed-tail', + claudeConfigDir: '/accounts/claude' + }) + expect(readTranscriptLeaf).toHaveBeenNthCalledWith(2, { + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: null, + claudeConfigDir: '/accounts/claude' + }) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'reproved-crash-leaf' }) + }) + + it('keeps the observed crash leaf when transcript validation proves a sibling branch', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi + .fn() + .mockRejectedValue(new Error('latest marker is on a sibling branch')) + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-crash-tail' + }) + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(readTranscriptLeaf).toHaveBeenCalledTimes(1) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'observed-crash-tail' }) + }) + + it('publishes lifecycle recovery even when crash-cursor persistence fails', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor( + claude, + {}, + events, + [], + undefined, + undefined, + vi.fn().mockRejectedValue(new Error('store unavailable')) + ) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(events.at(-1)).toMatchObject({ type: 'ended', cause: 'unexpected-exit' }) + }) + + it('runs the child close proof before publishing unexpected-exit recovery', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + const close = vi.spyOn(claude.connections[0], 'close').mockResolvedValue(true) + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(close).toHaveBeenCalledOnce() + expect(events.at(-1)).toMatchObject({ type: 'ended', cause: 'unexpected-exit' }) + }) + + it('does not publish recovery while an unexpected-exit close proof is false', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistedHandles: unknown[] = [] + const adapter = adapterFor(claude, {}, events, persistedHandles) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].close = vi + .fn<() => Promise>() + .mockResolvedValueOnce(false) + .mockResolvedValueOnce(true) as unknown as (typeof claude.connections)[0]['close'] + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(events.filter((event) => event.type === 'ended')).toEqual([]) + expect(persistedHandles).toEqual([]) + }) + + it('retains pending prompts while an unexpected-exit proof is unproven', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + const answered = invokeCanUseTool(claude.connections[0], 'Bash', 'permission-1', 'tool-1') + claude.connections[0].close = vi + .fn<() => Promise>() + .mockResolvedValue(false) as unknown as (typeof claude.connections)[0]['close'] + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(answered.settled()).toBe(false) + expect(events.filter((event) => event.type === 'ended')).toEqual([]) + }) + + it('publishes unexpected recovery exactly once after a retained proof retries successfully', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].close = vi + .fn<() => Promise>() + .mockResolvedValueOnce(false) + .mockResolvedValueOnce(true) as unknown as (typeof claude.connections)[0]['close'] + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(true) + await tick() + + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + }) + + it('launches the first replacement from the settled retained transcript cursor', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistedHandles: unknown[] = [] + const journalSink: StructuredAgentSessionEventSink = { + appendItem: () => {}, + appendTombstone: () => {}, + publish: () => {} + } + const readTranscriptLeaf = vi.fn().mockResolvedValue('durable-retained-leaf') + let durableLeafUuid: string | null = null + const resolveLaunch = vi.fn(async ({ identity }) => { + if ( + identity.providerHandle.kind !== 'claude' || + identity.providerHandle.sessionId !== PROVIDER_SESSION_ID || + identity.providerHandle.leafUuid !== durableLeafUuid + ) { + throw new Error('claude durable resume identity changed before spawn') + } + if (durableLeafUuid === null) { + return { + pathToClaudeCodeExecutable: 'claude', + options: { sessionId: PROVIDER_SESSION_ID }, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: null, + resumed: false + } + } + return { + pathToClaudeCodeExecutable: 'claude', + options: { resume: PROVIDER_SESSION_ID, resumeSessionAt: durableLeafUuid }, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: durableLeafUuid, + resumed: true + } + }) + const persistHandle = vi.fn>( + async (handle) => { + durableLeafUuid = handle.leafUuid + persistedHandles.push(handle) + } + ) + const adapter = new ClaudeStructuredSessionAdapter({ + resolveLaunch, + openConnection: claude.openConnection, + onEvent: (event) => events.push(event), + readProcessStartTime: async () => 1_700_000_000_000, + now: () => 1_700_000_000_500, + readTranscriptLeaf, + persistHandle + }) + const firstAcquisition = await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + events: journalSink + }) + const first = claude.connections[0] + const oldPrompt = invokeCanUseTool(first, 'Bash', 'permission-retained', 'tool-retained') + const oldSession = ( + adapter as unknown as { + sessions: Map< + string, + { + translator: { dispose: () => void } | null + prompts: { + find: (itemId: string) => { prompt: { settle: (value: unknown) => void } } | null + } + } + > + } + ).sessions.get('session-1') + expect(oldSession?.translator).not.toBeNull() + const disposeTranslator = vi.spyOn(oldSession!.translator!, 'dispose') + const pendingPrompt = oldSession?.prompts.find('permission-retained') + expect(pendingPrompt).not.toBeNull() + const settlePrompt = vi.spyOn(pendingPrompt!.prompt, 'settle') + first.handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-retained-leaf' + }) + first.close = vi + .fn<() => Promise>() + .mockResolvedValueOnce(false) + .mockResolvedValueOnce(true) as unknown as (typeof first)['close'] + first.handlers.onExit?.(new Error('crashed before replacement')) + await tick() + + expect(oldPrompt.settled()).toBe(false) + expect(events.filter((event) => event.type === 'ended')).toEqual([]) + + const replacement = await adapter.acquire({ + identity: { + ...identityFor(), + providerHandle: { + kind: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: 'observed-retained-leaf' + } + }, + fence: 8, + spawnToken: 'spawn-10', + events: journalSink + }) + + expect(disposeTranslator).toHaveBeenCalledOnce() + expect(settlePrompt).toHaveBeenCalledOnce() + expect(settlePrompt).toHaveBeenCalledWith(null) + expect(persistHandle).toHaveBeenCalledOnce() + expect(persistedHandles).toEqual([ + { + sessionId: 'session-1', + providerSessionId: PROVIDER_SESSION_ID, + leafUuid: 'durable-retained-leaf', + fence: 7 + } + ]) + expect(readTranscriptLeaf).toHaveBeenCalledOnce() + expect(readTranscriptLeaf).toHaveBeenCalledWith({ + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'observed-retained-leaf', + claudeConfigDir: '/accounts/claude' + }) + expect(resolveLaunch).toHaveBeenNthCalledWith(2, { + identity: { + ...identityFor(), + providerHandle: { + kind: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: 'durable-retained-leaf' + } + } + }) + expect(oldPrompt.settled()).toBe(true) + expect(events.filter((event) => event.type === 'ended')).toEqual([ + { + type: 'ended', + sessionId: 'session-1', + reason: 'crashed before replacement', + cause: 'unexpected-exit', + fence: 7, + acquisitionGeneration: firstAcquisition.acquisitionGeneration + } + ]) + expect(replacement.link).toMatchObject({ + handle: { + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: 'durable-retained-leaf' + }, + origin: 'resumed', + mintedAtFence: 8 + }) + expect(claude.connections[1]?.launch.options).toMatchObject({ + resume: PROVIDER_SESSION_ID, + resumeSessionAt: 'durable-retained-leaf' + }) + expect(claude.connections).toHaveLength(2) + }) +}) diff --git a/src/main/claude/claude-structured-session-state.ts b/src/main/claude/claude-structured-session-state.ts new file mode 100644 index 00000000000..1c0b1862913 --- /dev/null +++ b/src/main/claude/claude-structured-session-state.ts @@ -0,0 +1,286 @@ +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import type { + ClaudeStreamJsonConnection, + openClaudeStreamJsonConnection +} from './claude-stream-json-connection' +import type { ClaudeStructuredLaunch } from './claude-structured-launch-resolution' +import type { ClaudeJournalTranslator } from './claude-structured-journal-translation' +import type { ClaudePendingPrompt, ClaudePromptRegistry } from './claude-structured-prompt-replies' +import { cancelProcessAcquisition } from '../../shared/child-process/cancel-process-acquisition' +import { randomUUID } from 'node:crypto' +import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire' +import type { ClaudeBackgroundTaskTracker } from './claude-background-task-tracker' + +export type ClaudeAuthDiagnostic = { + apiKeySourceConfigured: boolean + baseUrlConfigured: boolean + authTokenConfigured: boolean + apiKeyConfigured: boolean + settingSources: readonly string[] +} + +export type ClaudeStructuredSessionEvent = + | { + type: 'message' + sessionId: string + message: Record + /** Present only when this replay acknowledged Orca's in-flight dispatch. */ + startsTurn?: true + } + | { type: 'provider-frame'; sessionId: string; kind: string; payload: unknown } + | { type: 'prompt'; sessionId: string; prompt: ClaudePendingPrompt } + | { type: 'prompt-cancelled'; sessionId: string; promptKey: string } + | { type: 'options'; sessionId: string; models: unknown[] } + | { + type: 'handle' + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + } + | { type: 'auth-diagnostic'; sessionId: string; diagnostic: ClaudeAuthDiagnostic } + | { + type: 'ended' + sessionId: string + reason: string + /** Present for first-hand child exits so the host can fence recovery. */ + cause?: 'unexpected-exit' | 'requested-close' + fence?: number + acquisitionGeneration?: string + settlementRetryRequired?: boolean + } + +export type ClaudeStructuredSessionAdapterDeps = { + resolveLaunch: (input: { + identity: AgentSessionJournalIdentity + }) => Promise + onEvent?: (event: ClaudeStructuredSessionEvent) => void + onBackgroundTasksChanged?: ( + sessionId: string, + state: AgentSessionBackgroundTaskState | null + ) => void + openConnection?: typeof openClaudeStreamJsonConnection + readProcessStartTime?: (pid: number) => Promise + mintLinkId?: () => string + mintAcquisitionGeneration?: () => string + now?: () => number + requestTimeoutMs?: number + initTimeoutMs?: number + dispatchAckTimeoutMs?: number + persistHandle?: (input: { + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + }) => Promise + /** Read the durable transcript branch after a child has flushed its final rows. */ + readTranscriptLeaf?: (input: { + providerSessionId: string + previousLeafUuid: string | null + /** Account-scoped Claude config root that owns this provider session. */ + claudeConfigDir: string + }) => Promise +} + +export type ClaudeDispatchWaiter = { + resolve: (uuid: string | null) => void + timer: ReturnType + acceptsResult: boolean + /** Client uuid echoed by Claude so a replay is tied to its own dispatch. */ + sentUuid: string + /** Sequence used to fence a late identity from a newer dispatch. */ + dispatchSequence: number + /** Set when the provider replay settled this waiter before send returned. */ + settledUuid?: string + /** The waiter timed out or its write failed, but its replay may still arrive. */ + retired?: boolean + /** Bounded digest/summary for compatibility CLIs that mint UUIDs. */ + replayContentKey: string +} + +export type ClaudeSession = { + connection: ClaudeStreamJsonConnection + providerSessionId: string + /** Durable transcript files live under this account's `projects` directory. */ + claudeConfigDir: string + leafUuid: string | null + fence: number + acquisitionGeneration: string + prompts: ClaudePromptRegistry + dispatchWaiters: ClaudeDispatchWaiter[] + /** Bounded identities for dispatches whose ack was unknown when they returned. */ + retiredDispatchWaiters: ClaudeDispatchWaiter[] + /** Once a retired waiter is evicted, legacy content-only replay matching is unsafe. */ + replayContentFallbackBlocked: boolean + options: Map + reportedOptions: { model?: string; effort?: string } + /** `optionMutationSequence` when `reportedOptions.model` was last observed, so a + * write still awaiting its first turn outranks the report it will replace. */ + reportedModelMutation: number + /** Options whose recorded value the provider reported, not merely accepted. */ + confirmedOptions: Set + restoreSkippedOptions: Set + /** CLI-advertised protocol capabilities from init; gates interrupt-receipt handling. */ + capabilities: readonly string[] + /** Provider uuid of the most recently admitted turn, if one is active. */ + activeTurnId?: string + backgroundTasks: ClaudeBackgroundTaskTracker + /** Monotonic fence advanced when a dispatch starts, including unresolved dispatches. */ + dispatchSequence: number + /** Dispatch sequence that admitted activeTurnId. */ + activeTurnSequence?: number + /** Fences overlapping option writes so a late completion cannot restore stale state. */ + optionMutationSequence: number + /** Shared durable-close write; a failed write clears this for a retry. */ + closePersistence?: Promise + /** Shared full close/finalization operation; a failed operation clears this for a retry. */ + closeFinalization?: Promise + /** Set only after the durable close write succeeds, before lifecycle emission. */ + closeFinalized?: boolean + /** Set once `ended` has been emitted, so a persistence retry cannot repeat it. */ + closeEnded?: boolean + translator: ClaudeJournalTranslator | null + events: StructuredAgentSessionEventSink | undefined +} + +export function mintClaudeAcquisitionGeneration(deps: ClaudeStructuredSessionAdapterDeps): string { + return deps.mintAcquisitionGeneration?.() ?? randomUUID() +} + +/** + * The first-hand exit that removed a published session. Kept until the session + * is acquired again so acquisition cleanup that arrives after the exit finds + * what the ladder observed, not an absence it would otherwise report as proven. + */ +export type ClaudeSessionExit = { + connection: ClaudeStreamJsonConnection + /** Full session identity retained until its child tree is proven gone. */ + session: ClaudeSession + error: Error + /** The exit path's first proof attempt; retries must observe this result. */ + closePromise?: Promise + /** Shared lifecycle settlement for concurrent proof retries. */ + settlementPromise?: Promise + /** The whole ladder-then-settle tail, retained so a barrier can await an exit + * that is observed but not yet published. Never rejects. */ + publication?: Promise +} + +export type ClaudeAcquisitionAttempt = { + connection: ClaudeStreamJsonConnection | null + prompts: ClaudePromptRegistry + buffered: (() => void)[] + published: boolean + cancelled: boolean + exitProven: boolean + finished: Promise + finish: () => void +} + +export function createClaudeAcquisitionAttempt( + prompts: ClaudePromptRegistry +): ClaudeAcquisitionAttempt { + let finish = (): void => {} + const finished = new Promise((resolve) => { + finish = resolve + }) + return { + connection: null, + prompts, + buffered: [], + published: false, + cancelled: false, + exitProven: false, + finished, + finish + } +} + +export class ClaudeAcquisitionRegistry { + private readonly attempts = new Map() + private closing = false + + get size(): number { + return this.attempts.size + } + + start( + sessionId: string, + prompts: ClaudePromptRegistry + ): { + previous: ClaudeAcquisitionAttempt | undefined + attempt: ClaudeAcquisitionAttempt + } { + if (this.closing) { + throw new Error('claude structured session adapter is closing') + } + const previous = this.attempts.get(sessionId) + const attempt = createClaudeAcquisitionAttempt(prompts) + this.attempts.set(sessionId, attempt) + return { previous, attempt } + } + + assertCurrent(sessionId: string, attempt: ClaudeAcquisitionAttempt): void { + if (this.closing || attempt.cancelled || this.attempts.get(sessionId) !== attempt) { + throw new Error(`claude session ${sessionId} was superseded while being acquired`) + } + } + + get(sessionId: string): ClaudeAcquisitionAttempt | undefined { + return this.attempts.get(sessionId) + } + + deleteIfCurrent(sessionId: string, attempt: ClaudeAcquisitionAttempt): void { + if (this.attempts.get(sessionId) === attempt) { + this.attempts.delete(sessionId) + } + } + + restoreIfCurrent( + sessionId: string, + replacement: ClaudeAcquisitionAttempt, + previous: ClaudeAcquisitionAttempt + ): void { + if (this.attempts.get(sessionId) === replacement) { + this.attempts.set(sessionId, previous) + } + } + + sessionIds(): IterableIterator { + return this.attempts.keys() + } + + close(): void { + this.closing = true + } +} + +export async function cancelClaudeAcquisitionAttempt( + attempt: ClaudeAcquisitionAttempt | undefined +): Promise { + if (!attempt) { + return true + } + return cancelProcessAcquisition({ + cancel: () => { + attempt.cancelled = true + }, + connection: () => attempt.connection, + exitProven: () => attempt.exitProven, + finished: attempt.finished + }) +} + +/** What an acquisition hands back to the adapter that owns the session map: + * event delivery ordered against publication, and the two exit settlements. */ +export type ClaudeAcquireCallbacks = { + deliver: (attempt: ClaudeAcquisitionAttempt, sessionId: string, event: () => void) => void + emit: ( + session: ClaudeSession | null, + events: StructuredAgentSessionEventSink | undefined, + event: ClaudeStructuredSessionEvent + ) => void + handleExit: (sessionId: string, attempt: ClaudeAcquisitionAttempt, error: Error) => void + settleExit: (sessionId: string, exit: ClaudeSessionExit) => Promise +} diff --git a/src/main/claude/claude-structured-session-test-support.ts b/src/main/claude/claude-structured-session-test-support.ts new file mode 100644 index 00000000000..903cafae416 --- /dev/null +++ b/src/main/claude/claude-structured-session-test-support.ts @@ -0,0 +1,266 @@ +import type { + AgentJournalMessageItem, + AgentSessionJournalIdentity +} from '../../shared/agent-session-journal-types' +import type { + ClaudeStreamJsonConnection, + ClaudeStreamJsonConnectionHandlers, + ClaudeStreamJsonLaunch, + openClaudeStreamJsonConnection +} from './claude-stream-json-connection' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionAdapterDeps, + type ClaudeStructuredLaunch, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' + +export const PROVIDER_SESSION_ID = '819cf9f8-e43c-4ad7-b50f-54aa158a726a' + +export const USER_MESSAGE: AgentJournalMessageItem = { + kind: 'message', + role: 'user', + blocks: [{ type: 'text', text: 'ship it' }] +} + +export function identityFor(sessionId = 'session-1'): AgentSessionJournalIdentity { + return { + sessionId, + workspaceId: 'workspace-1', + hostId: 'host-1', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: PROVIDER_SESSION_ID, leafUuid: null } + } +} + +type Route = (params: Record | undefined) => unknown + +export type FakeConnection = Omit & { + closed: boolean + exitVerdict: ClaudeStreamJsonConnection['exitVerdict'] + launch: ClaudeStreamJsonLaunch + handlers: ClaudeStreamJsonConnectionHandlers + calls: { subtype: string; params?: Record }[] + sent: Record[] + closeCount: number +} + +export function fakeClaude( + options: { + initSessionId?: string + initUuid?: string + initModel?: string + initProof?: 'init' | 'session-start' | 'none' + initAccount?: unknown + exitBeforeInit?: string + settings?: unknown + replayUuid?: string | null + replayUuids?: (string | null)[] + capabilities?: string[] + unprovenCloseVerdict?: ClaudeStreamJsonConnection['exitVerdict'] + routes?: Record + } = {} +): { + connections: FakeConnection[] + openConnection: typeof openClaudeStreamJsonConnection + routes: Record +} { + const connections: FakeConnection[] = [] + const routes = options.routes ?? {} + let replayIndex = 0 + const routed = (subtype: string, params?: Record): unknown => { + const route = routes[subtype] + return route ? route(params) : undefined + } + const openConnection = (async (launch, handlers = {}) => { + const connection: FakeConnection = { + launch, + handlers, + calls: [], + sent: [], + closeCount: 0, + pid: 4321, + closed: false, + initializationResult: async () => { + connection.calls.push({ subtype: 'initialize' }) + if (options.exitBeforeInit) { + handlers.onExit?.(new Error(options.exitBeforeInit)) + return { models: [] } + } + if (options.initProof === 'session-start') { + handlers.onMessage?.({ + type: 'system', + subtype: 'hook_started', + hook_name: 'SessionStart:startup', + session_id: options.initSessionId ?? PROVIDER_SESSION_ID, + uuid: options.initUuid ?? 'init-uuid' + }) + } else if (options.initProof !== 'none') { + // Keys mirror the real system/init frame, which carries `model` but no + // effort of any kind: the current effort only comes back from + // get_settings. Never add a field the CLI does not send. + handlers.onMessage?.({ + type: 'system', + subtype: 'init', + session_id: options.initSessionId ?? PROVIDER_SESSION_ID, + uuid: options.initUuid ?? 'init-uuid', + model: options.initModel ?? 'claude-sonnet-5', + apiKeySource: 'none', + ...(options.capabilities ? { capabilities: options.capabilities } : {}) + }) + } + return { + models: [{ value: 'claude-sonnet', displayName: 'Sonnet' }], + ...(options.initAccount === undefined ? {} : { account: options.initAccount }) + } + }, + getSettings: async () => { + connection.calls.push({ subtype: 'get_settings' }) + // Shape measured from Claude Code 2.1.258: {applied, effective, sources}, + // and the only place the session's current effort is reported. + return ( + options.settings ?? { + applied: { model: 'claude-sonnet-5', effort: 'high', advisor: null, ultracode: false }, + effective: { model: 'claude-sonnet-5', effortLevel: 'high', env: {} }, + sources: {} + } + ) + }, + supportedModels: async () => { + connection.calls.push({ subtype: 'list_models' }) + return (routed('list_models') as unknown[] | undefined) ?? [] + }, + setModel: async (model) => { + connection.calls.push({ subtype: 'set_model', params: { model } }) + routed('set_model', { model }) + }, + setPermissionMode: async (mode) => { + connection.calls.push({ subtype: 'set_permission_mode', params: { mode } }) + routed('set_permission_mode', { mode }) + }, + applyFlagSettings: async (settings) => { + connection.calls.push({ subtype: 'apply_flag_settings', params: { settings } }) + routed('apply_flag_settings', { settings }) + }, + interrupt: async (interruptOptions) => { + connection.calls.push({ + subtype: 'interrupt', + params: interruptOptions?.cancelQueued ? { cancelQueued: true } : {} + }) + return routed('interrupt', interruptOptions) as + | Awaited> + | undefined + }, + cancelAsyncMessage: async (uuid) => { + connection.calls.push({ subtype: 'cancel_async_message', params: { uuid } }) + routed('cancel_async_message', { uuid }) + }, + stopTask: async (taskId) => { + connection.calls.push({ subtype: 'stop_task', params: { taskId } }) + routed('stop_task', { taskId }) + }, + send: async (message) => { + connection.sent.push(message) + if (message.type === 'user' && options.replayUuid !== null) { + const configuredReplayUuid = options.replayUuids + ? options.replayUuids[replayIndex++] + : options.replayUuid + const replayUuid = + configuredReplayUuid === undefined ? `user-uuid-${replayIndex}` : configuredReplayUuid + if (replayUuid !== null) { + handlers.onMessage?.({ + ...message, + uuid: replayUuid + }) + } + } + }, + exitVerdict: options.unprovenCloseVerdict ?? { root: 'live', tree: 'unverifiable' }, + close: async () => { + connection.closeCount += 1 + connection.closed = true + return options.unprovenCloseVerdict === undefined + } + } + connections.push(connection) + return connection + }) as typeof openClaudeStreamJsonConnection + return { connections, openConnection, routes } +} + +export function adapterFor( + claude: ReturnType, + launch: Partial = {}, + events: ClaudeStructuredSessionEvent[] = [], + persistedHandles: unknown[] = [], + initTimeoutMs?: number, + readTranscriptLeaf?: ClaudeStructuredSessionAdapterDeps['readTranscriptLeaf'], + persistHandle?: ClaudeStructuredSessionAdapterDeps['persistHandle'], + onBackgroundTasksChanged?: ClaudeStructuredSessionAdapterDeps['onBackgroundTasksChanged'] +): ClaudeStructuredSessionAdapter { + return new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: null, + resumed: false, + ...launch + }), + onEvent: (event) => events.push(event), + openConnection: claude.openConnection, + readProcessStartTime: async () => 1_700_000_000_000, + now: () => 1_700_000_000_500, + ...(initTimeoutMs === undefined ? {} : { initTimeoutMs }), + dispatchAckTimeoutMs: 10, + persistHandle: + persistHandle ?? + (async (handle) => { + persistedHandles.push(handle) + }), + ...(onBackgroundTasksChanged ? { onBackgroundTasksChanged } : {}), + ...(readTranscriptLeaf ? { readTranscriptLeaf } : {}) + }) +} + +export async function acquired( + claude: ReturnType, + launch: Partial = {}, + events: ClaudeStructuredSessionEvent[] = [] +): Promise { + const adapter = adapterFor(claude, launch, events) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + return adapter +} + +export function tick(): Promise { + return new Promise((resolve) => setImmediate(resolve)) +} + +export function invokeCanUseTool( + connection: FakeConnection, + toolName: string, + requestId: string, + toolUseID: string, + extra: { + input?: Record + suggestions?: unknown[] + signal?: AbortSignal + } = {} +): { promise: Promise; settled: () => boolean } { + const options = { + requestId, + toolUseID, + signal: extra.signal ?? new AbortController().signal, + ...(extra.suggestions ? { suggestions: extra.suggestions } : {}) + } as unknown as Parameters>[2] + let done = false + const promise = Promise.resolve( + connection.handlers.canUseTool?.(toolName, extra.input ?? {}, options) + ).finally(() => { + done = true + }) + return { promise, settled: () => done } +} diff --git a/src/main/claude/claude-transcript-branch-proof.ts b/src/main/claude/claude-transcript-branch-proof.ts index d7065caa275..605f619eb92 100644 --- a/src/main/claude/claude-transcript-branch-proof.ts +++ b/src/main/claude/claude-transcript-branch-proof.ts @@ -5,6 +5,10 @@ const MAX_CLAUDE_TRANSCRIPT_ANCESTRY = 10_000 type TranscriptNode = { parentUuid: string | null sessionId: string | null + /** First line where this UUID was observed in the append-only transcript. */ + lineIndex: number + /** UUIDs from result/init/stream frames and sidechains are never leaves. */ + disallowedLeaf: boolean } export type ClaudeTranscriptBranchProof = { @@ -27,6 +31,54 @@ export class ClaudeTranscriptTailIncompleteError extends Error { } } +/** The sampled cursor is no longer present, so a root proof may still recover safely. */ +export class ClaudeTranscriptPreviousCursorMissingError extends Error { + constructor() { + super( + 'Claude transcript branch proof failed: previous cursor is missing from the session graph' + ) + this.name = 'ClaudeTranscriptPreviousCursorMissingError' + } +} + +function proveMainLineAncestry( + nodes: Map, + startUuid: string, + providerSessionId: string +): void { + const visited = new Set() + let cursor: string | null = startUuid + for (let depth = 0; cursor !== null && depth < MAX_CLAUDE_TRANSCRIPT_ANCESTRY; depth += 1) { + if (visited.has(cursor)) { + throw transcriptError('cycle in parentUuid ancestry') + } + visited.add(cursor) + const node = nodes.get(cursor) + if (!node || node.sessionId !== providerSessionId) { + throw transcriptError(`missing ancestor ${cursor}`) + } + if (node.disallowedLeaf) { + throw transcriptError(`ancestor ${cursor} is not on the main transcript`) + } + cursor = node.parentUuid + } + if (cursor !== null) { + throw transcriptError('ancestry exceeds the bounded proof limit') + } +} + +function proveAppendOrder(nodes: Map): void { + for (const node of nodes.values()) { + if (!node.parentUuid) { + continue + } + const parent = nodes.get(node.parentUuid) + if (parent && parent.lineIndex >= node.lineIndex) { + throw transcriptError('parent row follows descendant') + } + } +} + export function proveClaudeTranscriptBranchFromJsonl(input: { contents: string providerSessionId: string @@ -34,6 +86,7 @@ export function proveClaudeTranscriptBranchFromJsonl(input: { }): ClaudeTranscriptBranchProof { const nodes = new Map() let leafUuid: string | null = null + let leafMarkerLineIndex = -1 const lines = input.contents.split('\n') for (const [index, line] of lines.entries()) { if (!line.trim()) { @@ -59,6 +112,7 @@ export function proveClaudeTranscriptBranchFromJsonl(input: { throw transcriptError('invalid last-prompt marker') } leafUuid = markerLeaf + leafMarkerLineIndex = index } const uuid = nonEmptyString(row.uuid) if (!uuid) { @@ -70,27 +124,60 @@ export function proveClaudeTranscriptBranchFromJsonl(input: { } const sessionId = nonEmptyString(row.sessionId) const existing = nodes.get(uuid) - if (existing && (existing.parentUuid !== parentUuid || existing.sessionId !== sessionId)) { + const disallowedLeaf = + row.isSidechain === true || + row.parent_tool_use_id != null || + row.type === 'result' || + row.type === 'stream_event' || + (row.type === 'system' && row.subtype === 'init') + if ( + existing && + (existing.parentUuid !== parentUuid || + existing.sessionId !== sessionId || + existing.disallowedLeaf !== disallowedLeaf) + ) { throw transcriptError(`record ${uuid} has conflicting ancestry`) } - nodes.set(uuid, { parentUuid, sessionId }) + nodes.set(uuid, { + parentUuid, + sessionId, + lineIndex: existing?.lineIndex ?? index, + disallowedLeaf + }) } if (!leafUuid) { throw transcriptError('missing last-prompt marker') } const leaf = nodes.get(leafUuid) - if (!leaf || leaf.sessionId !== input.providerSessionId) { + if (!leaf || leaf.sessionId !== input.providerSessionId || leaf.disallowedLeaf) { throw transcriptError('marker leaf is missing from the session graph') } + if (leaf.lineIndex > leafMarkerLineIndex) { + throw transcriptError('marker precedes its leaf record') + } const previousLeafUuid = input.previousLeafUuid if (!previousLeafUuid) { + proveMainLineAncestry(nodes, leafUuid, input.providerSessionId) + // A branch proof is based on an append-only snapshot. A child that appears + // before its claimed parent is not a post-snapshot descendant observation; + // accepting that graph would turn reordered/torn rows into durable ancestry. + proveAppendOrder(nodes) return { leafUuid, relation: 'initial' } } const previous = nodes.get(previousLeafUuid) - if (!previous || previous.sessionId !== input.providerSessionId) { - throw transcriptError('previous cursor is missing from the session graph') + if (!previous) { + throw new ClaudeTranscriptPreviousCursorMissingError() } + if (previous.sessionId !== input.providerSessionId || previous.disallowedLeaf) { + throw transcriptError('previous cursor is not on the main transcript') + } + // The latest marker can be equal to, or descend from, a sampled cursor. In + // either case prove the sampled cursor's own ancestry before accepting it; + // otherwise a cursor that descended through a parent-tool-use sidechain + // could be persisted and resumed as if it were on the main transcript. + proveMainLineAncestry(nodes, previousLeafUuid, input.providerSessionId) if (leafUuid === previousLeafUuid) { + proveAppendOrder(nodes) return { leafUuid, relation: 'same' } } const visited = new Set() @@ -104,8 +191,12 @@ export function proveClaudeTranscriptBranchFromJsonl(input: { if (!node || node.sessionId !== input.providerSessionId) { throw transcriptError(`missing ancestor ${cursor}`) } + if (node.disallowedLeaf) { + throw transcriptError(`ancestor ${cursor} is not on the main transcript`) + } cursor = node.parentUuid if (cursor === previousLeafUuid) { + proveAppendOrder(nodes) return { leafUuid, relation: 'descendant' } } } @@ -126,3 +217,37 @@ export async function proveClaudeTranscriptBranch(input: { previousLeafUuid: input.previousLeafUuid }) } + +/** Re-run a durable branch proof from the transcript root when a sampled cursor is stale. */ +export async function readClaudeTranscriptLeafWithReproof(input: { + readTranscriptLeaf: (input: { + providerSessionId: string + previousLeafUuid: string | null + claudeConfigDir: string + }) => Promise + claudeConfigDir: string + providerSessionId: string + previousLeafUuid: string | null +}): Promise { + try { + return await input.readTranscriptLeaf({ + providerSessionId: input.providerSessionId, + previousLeafUuid: input.previousLeafUuid, + claudeConfigDir: input.claudeConfigDir + }) + } catch (error) { + // A missing cursor can be stale after compaction and is safe to re-prove from the root. A torn + // tail is still being written; dropping the cursor would make a later sibling look admissible. + if ( + input.previousLeafUuid === null || + !(error instanceof ClaudeTranscriptPreviousCursorMissingError) + ) { + throw error + } + return input.readTranscriptLeaf({ + providerSessionId: input.providerSessionId, + previousLeafUuid: null, + claudeConfigDir: input.claudeConfigDir + }) + } +} diff --git a/src/main/claude/claude-tui-exit.test.ts b/src/main/claude/claude-tui-exit.test.ts new file mode 100644 index 00000000000..6e1140b0f4d --- /dev/null +++ b/src/main/claude/claude-tui-exit.test.ts @@ -0,0 +1,159 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + completeClaudeTuiExit, + readClaudeTranscriptEntryUuid, + readClaudeTranscriptLeafUuid +} from './claude-tui-exit' + +const roots: string[] = [] + +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe('Claude TUI exit', () => { + it('does not sample UUIDs from subagent stdout frames with a parent tool use', () => { + expect( + readClaudeTranscriptEntryUuid({ + type: 'assistant', + uuid: 'subagent-assistant', + parent_tool_use_id: 'parent-tool' + }) + ).toBeNull() + expect( + readClaudeTranscriptEntryUuid({ + type: 'assistant', + uuid: 'main-assistant', + parent_tool_use_id: null + }) + ).toBe('main-assistant') + }) + + it('reads the authoritative last-prompt leaf from a transcript tail', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-claude-tui-exit-')) + roots.push(root) + const transcriptPath = join(root, 'session.jsonl') + await writeFile( + transcriptPath, + [ + { type: 'user', uuid: 'user-one' }, + { type: 'assistant', uuid: 'assistant-one' }, + { type: 'last-prompt', leafUuid: 'chain-head' }, + { type: 'file-history-snapshot', snapshot: {} } + ] + .map((entry) => JSON.stringify(entry)) + .join('\n') + ) + + await expect(readClaudeTranscriptLeafUuid(transcriptPath)).resolves.toBe('chain-head') + }) + + it('falls back to the last persisted message when last-prompt metadata is absent', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-claude-tui-exit-')) + roots.push(root) + const transcriptPath = join(root, 'session.jsonl') + await writeFile( + transcriptPath, + [ + { type: 'user', uuid: 'user-one' }, + { type: 'assistant', uuid: 'assistant-one' }, + { type: 'system', subtype: 'init', uuid: 'init-frame' }, + { type: 'result', uuid: 'result-frame' }, + { type: 'stream_event', uuid: 'stream-event-frame' } + ] + .map((entry) => JSON.stringify(entry)) + .join('\n') + ) + + await expect(readClaudeTranscriptLeafUuid(transcriptPath)).resolves.toBe('assistant-one') + }) + + it('ignores sidechain messages when selecting a fallback transcript leaf', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-claude-tui-sidechain-leaf-')) + const transcriptPath = join(root, 'session.jsonl') + await writeFile( + transcriptPath, + [ + { type: 'assistant', uuid: 'main-assistant' }, + { type: 'assistant', uuid: 'subagent-assistant', isSidechain: true }, + { type: 'result', uuid: 'result-frame' } + ] + .map((entry) => JSON.stringify(entry)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcriptPath)).resolves.toBe('main-assistant') + }) + + it('persists the resumed chain head only after the exact Claude child exits', async () => { + let resolveExit!: (exit: { + pid: number + exitCode: number | null + signal: string | null + }) => void + const exitPromise = new Promise<{ + pid: number + exitCode: number | null + signal: string | null + }>((resolve) => { + resolveExit = resolve + }) + const persistHandle = vi.fn(async () => undefined) + const completion = completeClaudeTuiExit({ + childPid: 4210, + waitForChildExit: () => exitPromise, + sessionId: 'provider-session', + transcriptPath: '/accounts/claude/session.jsonl', + fence: 7, + persistHandle, + readLeafUuid: async () => 'tui-leaf', + linkId: 'tui-resumed-link', + now: () => 12 + }) + + expect(persistHandle).not.toHaveBeenCalled() + resolveExit({ pid: 4210, exitCode: 0, signal: null }) + + await expect(completion).resolves.toMatchObject({ + link: { + linkId: 'tui-resumed-link', + handle: { provider: 'claude', sessionId: 'provider-session', leafUuid: 'tui-leaf' }, + origin: 'resumed', + mintedAtFence: 7, + observedAt: 12 + } + }) + expect(persistHandle).toHaveBeenCalledTimes(1) + }) + + it('refuses another process exit and a missing transcript leaf', async () => { + const persistHandle = vi.fn(async () => undefined) + await expect( + completeClaudeTuiExit({ + childPid: 4210, + waitForChildExit: async () => ({ pid: 4211, exitCode: 0, signal: null }), + sessionId: 'provider-session', + transcriptPath: '/session.jsonl', + fence: 2, + persistHandle, + readLeafUuid: async () => 'leaf' + }) + ).rejects.toThrow(/did not belong to the Claude child/) + await expect( + completeClaudeTuiExit({ + childPid: 4210, + waitForChildExit: async () => ({ pid: 4210, exitCode: 1, signal: null }), + sessionId: 'provider-session', + transcriptPath: '/session.jsonl', + fence: 2, + persistHandle, + readLeafUuid: async () => null + }) + ).rejects.toThrow(/resumable transcript leaf/) + expect(persistHandle).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/claude/claude-tui-exit.ts b/src/main/claude/claude-tui-exit.ts new file mode 100644 index 00000000000..3772e9c5e52 --- /dev/null +++ b/src/main/claude/claude-tui-exit.ts @@ -0,0 +1,119 @@ +import { open } from 'node:fs/promises' +import type { AgentSessionProviderHandleLink } from '../../shared/agent-session-provider-handle' +import { claudeProviderHandleLink } from './claude-structured-owner-identity' + +const TRANSCRIPT_TAIL_CHUNK_BYTES = 64 * 1024 +const TRANSCRIPT_TAIL_READ_LIMIT_BYTES = 4 * 1024 * 1024 + +type TranscriptLeafCandidate = { leafUuid: string; authoritative: boolean } + +function validLeafUuid(value: unknown): string | null { + if (typeof value !== 'string' || value.length === 0 || value.length > 512) { + return null + } + const hasControlCharacter = [...value].some((character) => { + const code = character.codePointAt(0) ?? 0 + return code <= 0x1f || code === 0x7f + }) + return value === value.trim() && !hasControlCharacter ? value : null +} + +export function readClaudeTranscriptEntryUuid(value: Record): string | null { + return value.isSidechain === true || + value.parent_tool_use_id != null || + (value.type !== 'user' && value.type !== 'assistant') + ? null + : validLeafUuid(value.uuid) +} + +function readLeafCandidate(line: string): TranscriptLeafCandidate | null { + try { + const value = JSON.parse(line) as Record + const lastPromptLeaf = value.type === 'last-prompt' ? validLeafUuid(value.leafUuid) : null + if (lastPromptLeaf) { + return { leafUuid: lastPromptLeaf, authoritative: true } + } + const messageLeaf = readClaudeTranscriptEntryUuid(value) + return messageLeaf ? { leafUuid: messageLeaf, authoritative: false } : null + } catch { + return null + } +} + +export async function readClaudeTranscriptLeafUuid(transcriptPath: string): Promise { + const file = await open(transcriptPath, 'r') + try { + const { size } = await file.stat() + let position = size + let suffix = '' + let fallback: string | null = null + let scanned = 0 + while (position > 0 && scanned < TRANSCRIPT_TAIL_READ_LIMIT_BYTES) { + const length = Math.min(TRANSCRIPT_TAIL_CHUNK_BYTES, position) + position -= length + scanned += length + const buffer = Buffer.alloc(length) + await file.read(buffer, 0, length, position) + const lines = `${buffer.toString('utf8')}${suffix}`.split(/\r?\n/) + suffix = position > 0 ? (lines.shift() ?? '') : '' + for (let index = lines.length - 1; index >= 0; index -= 1) { + const line = lines[index]?.trim() + if (!line) { + continue + } + const candidate = readLeafCandidate(line) + if (!candidate) { + continue + } + if (candidate.authoritative) { + return candidate.leafUuid + } + fallback ??= candidate.leafUuid + } + } + return fallback + } finally { + await file.close() + } +} + +export type ClaudeTuiChildExit = { + pid: number + exitCode: number | null + signal: string | null +} + +export async function completeClaudeTuiExit(input: { + childPid: number + waitForChildExit: () => Promise + sessionId: string + transcriptPath: string + fence: number + persistHandle: (link: AgentSessionProviderHandleLink) => Promise + readLeafUuid?: (transcriptPath: string) => Promise + linkId?: string + now?: () => number +}): Promise<{ + exit: ClaudeTuiChildExit + transcriptPath: string + link: AgentSessionProviderHandleLink +}> { + const exit = await input.waitForChildExit() + if (exit.pid !== input.childPid) { + throw new Error('The observed process exit did not belong to the Claude child.') + } + const leafUuid = await (input.readLeafUuid ?? readClaudeTranscriptLeafUuid)(input.transcriptPath) + if (!leafUuid) { + throw new Error('The exited Claude TUI did not persist a resumable transcript leaf.') + } + const link = claudeProviderHandleLink({ + sessionId: input.sessionId, + leafUuid, + resumed: true, + fence: input.fence, + ...(input.linkId ? { linkId: input.linkId } : {}), + observedAt: input.now?.() ?? Date.now() + }) + await input.persistHandle(link) + return { exit, transcriptPath: input.transcriptPath, link } +} diff --git a/src/main/claude/claude-tui-resume-launch.test.ts b/src/main/claude/claude-tui-resume-launch.test.ts new file mode 100644 index 00000000000..f907d1fcb4e --- /dev/null +++ b/src/main/claude/claude-tui-resume-launch.test.ts @@ -0,0 +1,227 @@ +import { chmodSync, mkdtempSync, mkdirSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { delimiter, join } from 'node:path' +import { describe, expect, it } from 'vitest' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { CLAUDE_AUTH_ENV_CONFLICT_MESSAGE } from '../claude-accounts/environment' +import { CLAUDE_DEFAULT_SETTING_SOURCES } from './claude-structured-launch-resolution' +import { CLAUDE_SPAWN_TOKEN_ENV } from './claude-structured-owner-identity' +import { createClaudeTuiResumeLaunchBuilder } from './claude-tui-resume-launch' + +function record(overrides: Partial = {}): AgentSessionRecord { + return { + sessionId: 'orca-session-1', + provider: 'claude', + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-folder', + workspaceKind: 'folder' + }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/accounts/claude-one' }, + providerHandleChain: [ + { + linkId: 'created', + handle: { provider: 'claude', sessionId: 'provider-session', leafUuid: 'leaf-one' }, + origin: 'created', + mintedAtFence: 1, + observedAt: 1 + } + ], + ...overrides + } as AgentSessionRecord +} + +function makeExecutable(path: string): void { + mkdirSync(join(path, '..'), { recursive: true }) + writeFileSync(path, '') + if (process.platform !== 'win32') { + chmodSync(path, 0o755) + } +} + +describe('Claude TUI resume launch', () => { + it('pins the workspace, account home, setting sources, and launch identity', async () => { + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async (workspaceId) => `/workspaces/${workspaceId}`, + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + resolveEnv: () => ({ + SELECTED_ACCOUNT: 'one', + ANTHROPIC_AUTH_TOKEN: 'selected-account-token' + }), + inheritedEnv: { + ANTHROPIC_API_KEY: 'inherited-gateway-key', + ANTHROPIC_BASE_URL: 'https://inherited-gateway.invalid', + CLAUDE_CODE_SESSION_ID: 'parent-session', + SAFE_PARENT: 'kept' + } + }) + + const launch = await build({ record: record(), spawnToken: 'spawn-one' }) + + expect(launch).toMatchObject({ + command: '/usr/local/bin/claude', + args: [ + '--setting-sources', + CLAUDE_DEFAULT_SETTING_SOURCES.join(','), + '--resume', + 'provider-session' + ], + cwd: '/workspaces/workspace-folder', + providerSessionId: 'provider-session', + resumeLeafUuid: 'leaf-one' + }) + expect(launch.env).toMatchObject({ + SAFE_PARENT: 'kept', + SELECTED_ACCOUNT: 'one', + CLAUDE_CONFIG_DIR: '/accounts/claude-one', + ORCA_AGENT_LAUNCH_TOKEN: 'spawn-one', + [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-one', + ANTHROPIC_AUTH_TOKEN: 'selected-account-token' + }) + // System auth (the only state an explicit ANTHROPIC_AUTH_TOKEN overlay is legal in): + // the user's own inherited key is their sign-in and survives. The managed-account + // half — where it is stripped — is covered by 'structured-to-TUI handoff auth'. + expect(launch.env.ANTHROPIC_API_KEY).toBe('inherited-gateway-key') + // Endpoint selection is not credential material; the existing adapter pinning preserves it. + expect(launch.env.ANTHROPIC_BASE_URL).toBe('https://inherited-gateway.invalid') + expect(launch.env.CLAUDE_CODE_SESSION_ID).toBeUndefined() + }) + + it('pairs the resumed Claude CLI with its sibling Node runtime', async () => { + const root = mkdtempSync(join(tmpdir(), 'orca-claude-resume-')) + const binDir = join(root, 'bin') + const claudeCommand = join(binDir, process.platform === 'win32' ? 'claude.cmd' : 'claude') + const nodeCommand = join(binDir, process.platform === 'win32' ? 'node.cmd' : 'node') + makeExecutable(claudeCommand) + makeExecutable(nodeCommand) + + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => claudeCommand, + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + resolveEnv: () => ({ PATH: '/usr/bin' }), + inheritedEnv: {} + }) + + const launch = await build({ record: record(), spawnToken: 'spawn' }) + + expect((launch.env.PATH ?? launch.env.Path)?.split(delimiter)[0]).toBe(binDir) + }) + + it('uses the durable session environment instead of current account settings', async () => { + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => 'claude', + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + resolveEnv: () => ({ ANTHROPIC_AUTH_TOKEN: 'pinned-token' }), + inheritedEnv: {} + }) + + const launch = await build({ record: record(), spawnToken: 'spawn' }) + + expect(launch.env.ANTHROPIC_AUTH_TOKEN).toBe('pinned-token') + }) + + it('resolves the durable chain head instead of an earlier Claude leaf', async () => { + const nextRecord = record({ + providerHandleChain: [ + ...record().providerHandleChain, + { + linkId: 'resumed', + handle: { provider: 'claude', sessionId: 'provider-session', leafUuid: 'leaf-two' }, + origin: 'resumed', + mintedAtFence: 2, + observedAt: 2 + } + ] + }) + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => 'claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + inheritedEnv: {} + }) + + await expect(build({ record: nextRecord, spawnToken: 'spawn-two' })).resolves.toMatchObject({ + providerSessionId: 'provider-session', + resumeLeafUuid: 'leaf-two' + }) + }) + + it('preserves durable Claude launch arguments before resume defaults', async () => { + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => 'claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + inheritedEnv: {} + }) + + const launch = await build({ + record: record({ launchArgs: ['--model', 'claude-sonnet-4-5'] }), + spawnToken: 'spawn' + }) + + expect(launch.args.slice(0, 3)).toEqual(['--model', 'claude-sonnet-4-5', '--setting-sources']) + }) + + it('rejects missing Claude handles and unpinned account homes', async () => { + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => 'claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + inheritedEnv: {} + }) + + await expect( + build({ record: record({ providerHandleChain: [] }), spawnToken: 'spawn' }) + ).rejects.toThrow('claude_tui_resume_handle_required') + await expect( + build({ + record: record({ accountHome: { variable: 'CODEX_HOME', path: '/wrong' } }), + spawnToken: 'spawn' + }) + ).rejects.toThrow(/CLAUDE_CONFIG_DIR/) + }) +}) + +// buildClaudeChildProcessEnv strips its inherited half unconditionally, so this module +// would have signed a system-auth user out of the session the structured path had just +// honoured. It is not wired up yet; the required policy is what stops the next caller +// from inheriting that. +describe('structured-to-TUI handoff auth', () => { + it('carries a system-auth user their own inherited credential', async () => { + const launch = await createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/repos/workspace-1', + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + inheritedEnv: { ANTHROPIC_API_KEY: 'sk-ant-SHELL', PATH: '/usr/bin' } + })({ record: record(), spawnToken: 'token-1' }) + + expect(launch.env.ANTHROPIC_API_KEY).toBe('sk-ant-SHELL') + }) + + it('still strips it once a managed account owns the credential', async () => { + const launch = await createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/repos/workspace-1', + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + inheritedEnv: { ANTHROPIC_API_KEY: 'sk-ant-SHELL', PATH: '/usr/bin' } + })({ record: record(), spawnToken: 'token-1' }) + + expect(launch.env.ANTHROPIC_API_KEY).toBeUndefined() + }) + + it('refuses a configured override of a pinned managed account, as the terminal path does', async () => { + await expect( + createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/repos/workspace-1', + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + resolveEnv: () => ({ ANTHROPIC_API_KEY: 'sk-ant-CONFIGURED' }), + inheritedEnv: {} + })({ record: record(), spawnToken: 'token-1' }) + ).rejects.toThrow(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + }) +}) diff --git a/src/main/claude/claude-tui-resume-launch.ts b/src/main/claude/claude-tui-resume-launch.ts new file mode 100644 index 00000000000..8c09335fd9d --- /dev/null +++ b/src/main/claude/claude-tui-resume-launch.ts @@ -0,0 +1,101 @@ +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { agentSessionProviderHandleChainHead } from '../../shared/agent-session-provider-handle' +import { withCliRuntimeOnPath } from '../../shared/node-cli-command-resolution' +import { resolveClaudeCommand } from '../codex-cli/command' +import { getSpawnArgsForWindows } from '../win32-utils' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + claudeAuthEnvCarriedForward, + hasClaudeAuthEnvConflict +} from '../claude-accounts/environment' +import { buildClaudeChildProcessEnv } from './claude-child-process-environment' +import { claudeConfigDirEnvPatch } from './claude-config-dir-pin' +import { CLAUDE_DEFAULT_SETTING_SOURCES } from './claude-structured-launch-resolution' +import { CLAUDE_SPAWN_TOKEN_ENV } from './claude-structured-owner-identity' + +export const CLAUDE_TUI_RESUME_BASE_ARGS = [ + '--setting-sources', + CLAUDE_DEFAULT_SETTING_SOURCES.join(',') +] as const + +export type ClaudeTuiResumeLaunch = { + command: string + args: string[] + cwd: string + env: Record + providerSessionId: string + resumeLeafUuid: string | null +} + +export type ClaudeTuiResumeLaunchBuilderDeps = { + resolveWorkspacePath: (workspaceId: string) => Promise + resolveCommand?: () => string + resolveEnv?: () => Record + inheritedEnv?: NodeJS.ProcessEnv + /** + * Required so whoever wires this module up has to answer the question rather than + * inherit the wrong default: buildClaudeChildProcessEnv strips its inherited half + * unconditionally, which would sign out a system-auth user whose own ANTHROPIC_* + * is their only credential. Build it with claudeStructuredAuthPolicyForSettings. + */ + resolveAuthPolicy: () => Promise | ClaudeStructuredAuthPolicy +} + +export function createClaudeTuiResumeLaunchBuilder( + deps: ClaudeTuiResumeLaunchBuilderDeps +): (input: { record: AgentSessionRecord; spawnToken: string }) => Promise { + return async ({ record, spawnToken }) => { + if (record.provider !== 'claude') { + throw new Error(`session ${record.sessionId} is a ${record.provider} session`) + } + if (record.accountHome.variable !== 'CLAUDE_CONFIG_DIR') { + throw new Error(`claude sessions pin CLAUDE_CONFIG_DIR, not ${record.accountHome.variable}`) + } + const head = agentSessionProviderHandleChainHead(record.providerHandleChain) + if (head?.handle.provider !== 'claude') { + throw new Error('claude_tui_resume_handle_required') + } + + const command = (deps.resolveCommand ?? resolveClaudeCommand)() + const { spawnCmd, spawnArgs } = getSpawnArgsForWindows(command, [ + ...(record.launchArgs ?? []), + ...CLAUDE_TUI_RESUME_BASE_ARGS, + '--resume', + head.handle.sessionId + ]) + const auth = await deps.resolveAuthPolicy() + const configuredEnv = deps.resolveEnv?.() ?? {} + if (auth.stripAuthEnv && hasClaudeAuthEnvConflict(configuredEnv)) { + throw new Error(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + } + // The inherited half is always stripped downstream, so a system-auth user's own + // credential only reaches the resumed TUI if it is carried in the configured half. + const carriedAuth = auth.stripAuthEnv + ? {} + : claudeAuthEnvCarriedForward(deps.inheritedEnv ?? process.env) + // Compared against what the child would otherwise inherit, so the record's account + // home still wins over a diverging overlay without a needless pin. + const inheritedEnv = { ...(deps.inheritedEnv ?? process.env), ...configuredEnv } + const env = buildClaudeChildProcessEnv( + { + ...carriedAuth, + ...configuredEnv, + ...claudeConfigDirEnvPatch(record.accountHome.path, { env: inheritedEnv }), + ORCA_AGENT_LAUNCH_TOKEN: spawnToken, + [CLAUDE_SPAWN_TOKEN_ENV]: spawnToken + }, + { inheritedEnv: deps.inheritedEnv } + ) + const pairedEnv = withCliRuntimeOnPath(command, env, { platform: process.platform }) + + return { + command: spawnCmd, + args: spawnArgs, + cwd: await deps.resolveWorkspacePath(record.location.workspaceId), + env: pairedEnv, + providerSessionId: head.handle.sessionId, + resumeLeafUuid: head.handle.leafUuid + } + } +} diff --git a/src/main/claude/claude-tui-resume-proof.test.ts b/src/main/claude/claude-tui-resume-proof.test.ts new file mode 100644 index 00000000000..6d2e99d4b00 --- /dev/null +++ b/src/main/claude/claude-tui-resume-proof.test.ts @@ -0,0 +1,80 @@ +import { describe, expect, it } from 'vitest' +import { proveClaudeTuiResume, readClaudeTuiSessionStartEvidence } from './claude-tui-resume-proof' + +const SESSION = '91deba8d-a398-4b69-a05d-35041536fe8e' +const TRANSCRIPT = '/accounts/claude/projects/workspace/transcript.jsonl' + +function envelope(overrides: Record = {}): Record { + return { + launchToken: 'spawn-one', + payload: JSON.stringify({ + hook_event_name: 'SessionStart', + source: 'resume', + session_id: SESSION, + transcript_path: TRANSCRIPT, + ...overrides + }) + } +} + +describe('Claude TUI resume proof', () => { + it('reads SessionStart identity from the hook envelope', () => { + expect(readClaudeTuiSessionStartEvidence(envelope())).toEqual({ + hookEventName: 'SessionStart', + source: 'resume', + sessionId: SESSION, + transcriptPath: TRANSCRIPT, + launchToken: 'spawn-one' + }) + }) + + it('proves the exact launched session and transcript without terminal output', async () => { + await expect( + proveClaudeTuiResume({ + expectedSessionId: SESSION, + expectedTranscriptPath: TRANSCRIPT, + expectedLaunchToken: 'spawn-one', + waitForSessionStart: async () => envelope() + }) + ).resolves.toMatchObject({ sessionId: SESSION, transcriptPath: TRANSCRIPT }) + }) + + it.each([ + ['source', { source: 'startup' }, /resume SessionStart/], + ['session', { session_id: 'other-session' }, /different Claude session/], + ['transcript', { transcript_path: '/other/transcript.jsonl' }, /different Claude transcript/] + ])('rejects a mismatched %s', async (_name, overrides, expected) => { + await expect( + proveClaudeTuiResume({ + expectedSessionId: SESSION, + expectedTranscriptPath: TRANSCRIPT, + expectedLaunchToken: 'spawn-one', + waitForSessionStart: async () => envelope(overrides) + }) + ).rejects.toThrow(expected) + }) + + it('rejects a SessionStart from another launched process', async () => { + await expect( + proveClaudeTuiResume({ + expectedSessionId: SESSION, + expectedTranscriptPath: TRANSCRIPT, + expectedLaunchToken: 'spawn-two', + waitForSessionStart: async () => envelope() + }) + ).rejects.toThrow(/different launched process/) + }) + + it('compares Windows paths using host path semantics', async () => { + await expect( + proveClaudeTuiResume({ + expectedSessionId: SESSION, + expectedTranscriptPath: 'C:\\Users\\Dev\\session.jsonl', + expectedLaunchToken: 'spawn-one', + platform: 'win32', + waitForSessionStart: async () => + envelope({ transcript_path: 'c:\\users\\dev\\session.jsonl' }) + }) + ).resolves.toMatchObject({ sessionId: SESSION }) + }) +}) diff --git a/src/main/claude/claude-tui-resume-proof.ts b/src/main/claude/claude-tui-resume-proof.ts new file mode 100644 index 00000000000..f352423116e --- /dev/null +++ b/src/main/claude/claude-tui-resume-proof.ts @@ -0,0 +1,111 @@ +import { posix, win32 } from 'node:path' + +export type ClaudeTuiSessionStartEvidence = { + hookEventName: 'SessionStart' + source: 'resume' + sessionId: string + transcriptPath: string + launchToken: string +} + +function record(value: unknown): Record | null { + return typeof value === 'object' && value !== null && !Array.isArray(value) + ? (value as Record) + : null +} + +function nonEmptyString(value: unknown): string | null { + return typeof value === 'string' && value.length > 0 ? value : null +} + +function hookPayload(envelope: Record): Record | null { + if (typeof envelope.payload === 'string') { + try { + return record(JSON.parse(envelope.payload)) + } catch { + return null + } + } + return record(envelope.payload) ?? envelope +} + +export function readClaudeTuiSessionStartEvidence( + value: unknown +): ClaudeTuiSessionStartEvidence | null { + const envelope = record(value) + if (!envelope) { + return null + } + const payload = hookPayload(envelope) + if (!payload) { + return null + } + const hookEventName = nonEmptyString(payload.hook_event_name ?? payload.hookEventName) + const source = nonEmptyString(payload.source) + const sessionId = nonEmptyString(payload.session_id ?? payload.sessionId) + const transcriptPath = nonEmptyString(payload.transcript_path ?? payload.transcriptPath) + const launchToken = nonEmptyString(envelope.launchToken ?? payload.launchToken) + return hookEventName === 'SessionStart' && + source === 'resume' && + sessionId && + transcriptPath && + launchToken + ? { hookEventName, source, sessionId, transcriptPath, launchToken } + : null +} + +function comparablePath(value: string, platform: NodeJS.Platform): string | null { + if (value.includes('\0')) { + return null + } + const path = platform === 'win32' ? win32 : posix + if (!path.isAbsolute(value)) { + return null + } + const normalized = path.normalize(value) + return platform === 'win32' ? normalized.toLowerCase() : normalized +} + +export async function proveClaudeTuiResume(input: { + expectedSessionId: string + expectedTranscriptPath: string + expectedLaunchToken: string + waitForSessionStart: () => Promise + timeoutMs?: number + platform?: NodeJS.Platform +}): Promise { + const timeoutMs = input.timeoutMs ?? 15_000 + let timer: ReturnType | undefined + try { + const evidence = readClaudeTuiSessionStartEvidence( + await Promise.race([ + input.waitForSessionStart(), + new Promise((_resolve, reject) => { + timer = setTimeout( + () => reject(new Error('The agent terminal did not prove the expected Claude resume.')), + timeoutMs + ) + timer.unref?.() + }) + ]) + ) + if (!evidence) { + throw new Error('The agent terminal did not emit a Claude resume SessionStart proof.') + } + if (evidence.launchToken !== input.expectedLaunchToken) { + throw new Error('The Claude resume proof came from a different launched process.') + } + if (evidence.sessionId !== input.expectedSessionId) { + throw new Error('The agent terminal resumed a different Claude session.') + } + const platform = input.platform ?? process.platform + const expectedPath = comparablePath(input.expectedTranscriptPath, platform) + const observedPath = comparablePath(evidence.transcriptPath, platform) + if (!expectedPath || !observedPath || observedPath !== expectedPath) { + throw new Error('The agent terminal resumed a different Claude transcript.') + } + return evidence + } finally { + clearTimeout(timer) + } +} diff --git a/src/main/claude/claude-tui-resume-real-binary.integration.test.ts b/src/main/claude/claude-tui-resume-real-binary.integration.test.ts new file mode 100644 index 00000000000..9ba3daf2285 --- /dev/null +++ b/src/main/claude/claude-tui-resume-real-binary.integration.test.ts @@ -0,0 +1,279 @@ +import { spawnSync } from 'node:child_process' +import { randomUUID } from 'node:crypto' +import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises' +import { homedir, tmpdir } from 'node:os' +import { join } from 'node:path' +import * as pty from 'node-pty' +import { afterEach, describe, expect, it } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { resolveClaudeCommand } from '../codex-cli/command' +import { readStructuredTuiProcessIdentity } from '../runtime/structured-tui-process-identity' +import { getSpawnArgsForWindows } from '../win32-utils' +import { CLAUDE_STRUCTURED_BASE_OPTIONS } from './claude-structured-launch-resolution' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' +import { createClaudeTuiResumeLaunchBuilder } from './claude-tui-resume-launch' +import { proveClaudeTuiResume } from './claude-tui-resume-proof' + +const command = resolveClaudeCommand() +const claudeAvailable = + spawnSync(command, ['--version'], { stdio: 'ignore', timeout: 5_000 }).status === 0 +const authStatusLaunch = getSpawnArgsForWindows(command, ['auth', 'status', '--json']) +const claudeAuthenticated = (() => { + if (!claudeAvailable) { + return false + } + const result = spawnSync(authStatusLaunch.spawnCmd, authStatusLaunch.spawnArgs, { + encoding: 'utf8', + windowsHide: true, + timeout: 5_000 + }) + return result.status === 0 && /"loggedIn"\s*:\s*true/.test(result.stdout) +})() +const roots: string[] = [] +const transcripts: string[] = [] + +function shellQuote(value: string): string { + return process.platform === 'win32' + ? `"${value.replace(/"/g, '""')}"` + : `'${value.replace(/'/g, `'"'"'`)}'` +} + +async function installCaptureHook( + root: string +): Promise<{ eventsPath: string; settingsPath: string }> { + const scriptPath = join(root, 'capture-session-start.cjs') + const eventsPath = join(root, 'session-start.jsonl') + const settingsPath = join(root, 'settings.json') + await writeFile( + scriptPath, + [ + "const { appendFileSync } = require('node:fs')", + "let input = ''", + "process.stdin.setEncoding('utf8')", + "process.stdin.on('data', (chunk) => { input += chunk })", + "process.stdin.on('end', () => {", + ' const payload = JSON.parse(input)', + ' payload.launchToken = process.env.ORCA_AGENT_LAUNCH_TOKEN', + ' appendFileSync(process.argv[2], `${JSON.stringify(payload)}\\n`)', + '})', + '' + ].join('\n') + ) + await writeFile( + settingsPath, + JSON.stringify({ + theme: 'dark', + hooks: { + SessionStart: [ + { + hooks: [ + { + type: 'command', + command: [process.execPath, scriptPath, eventsPath].map(shellQuote).join(' ') + } + ] + } + ] + } + }) + ) + return { eventsPath, settingsPath } +} + +async function waitForHook( + eventsPath: string, + source: 'startup' | 'resume' +): Promise> { + const deadline = Date.now() + 15_000 + while (Date.now() < deadline) { + const contents = await readFile(eventsPath, 'utf8').catch(() => '') + for (const line of contents.split(/\r?\n/)) { + if (!line.trim()) { + continue + } + const event = JSON.parse(line) as Record + if (event.hook_event_name === 'SessionStart' && event.source === source) { + return event + } + } + await new Promise((resolve) => setTimeout(resolve, 50)) + } + throw new Error(`Claude did not emit a ${source} SessionStart hook`) +} + +type RunningTui = { proc: pty.IPty; exited: Promise } + +function spawnResumeTui(args: string[], env: Record): RunningTui { + const direct = process.platform === 'win32' + const proc = pty.spawn( + direct ? command : process.env.SHELL || '/bin/zsh', + direct ? args : ['-l'], + { + name: 'xterm-256color', + cols: 100, + rows: 30, + cwd: process.cwd(), + env: { ...env, TERM: 'xterm-256color' } + } + ) + if (!direct) { + setTimeout(() => { + proc.write(`${[command, ...args].map(shellQuote).join(' ')}\r`) + }, 100).unref() + } + return { proc, exited: new Promise((resolve) => proc.onExit(() => resolve())) } +} + +function structuredIdentity(providerSessionId: string): AgentSessionJournalIdentity { + return { + sessionId: 'orca-real-claude-resume', + workspaceId: 'workspace-real', + hostId: 'local', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: providerSessionId, leafUuid: null } + } +} + +async function waitForStructuredResult(events: ClaudeStructuredSessionEvent[]): Promise { + const deadline = Date.now() + 30_000 + while (Date.now() < deadline) { + if (events.some((event) => event.type === 'message' && event.message.type === 'result')) { + return + } + await new Promise((resolve) => setTimeout(resolve, 50)) + } + throw new Error('Claude structured session did not finish its product-path turn') +} + +async function stopTui(tui: RunningTui): Promise { + try { + tui.proc.kill('SIGKILL') + } catch { + return + } + await Promise.race([ + tui.exited, + new Promise((_resolve, reject) => + setTimeout(() => reject(new Error('Claude TUI did not exit after cleanup')), 5_000) + ) + ]) +} + +afterEach(async () => { + await Promise.all(transcripts.splice(0).map((path) => rm(path, { force: true }))) + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe.skipIf(!claudeAuthenticated)('real Claude TUI resume proof', () => { + it('resumes a product-created structured session and proves its exact child', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-claude-tui-resume-')) + roots.push(root) + const { eventsPath, settingsPath } = await installCaptureHook(root) + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: command, + options: { + ...CLAUDE_STRUCTURED_BASE_OPTIONS, + extraArgs: { ...CLAUDE_STRUCTURED_BASE_OPTIONS.extraArgs, settings: settingsPath }, + sessionId: providerSessionId + }, + cwd: process.cwd(), + claudeConfigDir, + providerSessionId, + resumeLeafUuid: null, + resumed: false + }), + onEvent: (event) => events.push(event), + readProcessStartTime: async () => 1 + }) + let resumed: RunningTui | null = null + try { + const acquisition = await adapter.acquire({ + identity: structuredIdentity(providerSessionId), + fence: 1, + spawnToken: 'real-create' + }) + await expect( + adapter.dispatch({ + sessionId: 'orca-real-claude-resume', + clientMessageId: 'real-product-turn', + fence: 1, + body: { + kind: 'message', + role: 'user', + blocks: [{ type: 'text', text: 'Reply only with ORCA_RESUME_READY.' }] + } + }) + ).resolves.toMatchObject({ state: 'accepted' }) + await waitForStructuredResult(events) + const started = await waitForHook(eventsPath, 'startup') + const transcriptPath = String(started.transcript_path) + transcripts.push(transcriptPath) + expect(started.session_id).toBe(providerSessionId) + await adapter.closeAll() + + const record = { + sessionId: 'orca-real-claude-resume', + provider: 'claude', + location: { workspaceId: 'workspace-real' }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: claudeConfigDir }, + providerHandleChain: [ + { + linkId: 'created-real', + handle: acquisition.link.handle, + origin: 'created', + mintedAtFence: 1, + observedAt: 1 + } + ] + } as AgentSessionRecord + const launch = await createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => process.cwd(), + resolveCommand: () => command, + // The real binary authenticates from the developer's own environment here, + // which is the system-auth case: stripping it would sign the resume out. + resolveAuthPolicy: () => ({ stripAuthEnv: false }) + })({ record, spawnToken: 'real-resume' }) + resumed = spawnResumeTui([...launch.args, '--settings', settingsPath], launch.env) + let resumedOutput = '' + resumed.proc.onData((data) => { + resumedOutput = `${resumedOutput}${data}`.slice(-4_000) + }) + + const [processIdentity, proof] = await Promise.all([ + readStructuredTuiProcessIdentity({ + hostId: 'local', + rootPid: resumed.proc.pid, + spawnToken: 'real-resume', + agent: 'claude' + }), + proveClaudeTuiResume({ + expectedSessionId: providerSessionId, + expectedTranscriptPath: transcriptPath, + expectedLaunchToken: 'real-resume', + waitForSessionStart: () => waitForHook(eventsPath, 'resume') + }).catch((error) => { + throw new Error(`${String(error)}\nClaude output: ${resumedOutput}`) + }) + ]) + expect(processIdentity).toMatchObject({ + hostId: 'local', + spawnToken: 'real-resume', + pid: expect.any(Number) + }) + expect(proof).toMatchObject({ sessionId: providerSessionId, transcriptPath }) + } finally { + await adapter.closeAll() + if (resumed) { + await stopTui(resumed) + } + } + }, 30_000) +}) diff --git a/src/main/codex/codex-app-server-session.ts b/src/main/codex/codex-app-server-session.ts index 2e176e13ae2..35537f59d0d 100644 --- a/src/main/codex/codex-app-server-session.ts +++ b/src/main/codex/codex-app-server-session.ts @@ -2,6 +2,7 @@ import { spawn, type ChildProcess, type ChildProcessWithoutNullStreams } from 'n import { waitForProcessExitUntil } from './codex-process-exit-deadline' import { stderrIndicatesMissingAppServer } from './codex-app-server-capability-signal' import { withCliRuntimeOnPath } from '../../shared/node-cli-command-resolution' +import { admitProcessTreeKill } from '../../shared/child-process/process-tree-kill-gate' // Why: `codex app-server` is Orca's sanctioned RPC surface into Codex-owned // state (hook trust hashes, the sqlite thread index). This module owns the @@ -74,6 +75,18 @@ export function killCodexAppServerProcessTree( const platform = options.platform ?? process.platform const spawnImpl = options.spawnImpl ?? spawn if (platform === 'win32' && child.pid) { + if ( + !admitProcessTreeKill({ + pid: child.pid, + site: 'codex-app-server-session-deadline', + scope: 'win-taskkill-tree' + }) + ) { + // Refusal blocks the tree walk, not the termination: the root kill is + // handle-addressed, so it cannot reach the recycled pid we refused. + child.kill('SIGKILL') + return + } try { // Why: npm-installed Codex runs behind cmd.exe; killing only that wrapper // leaves the app-server child alive after a timeout or failed shutdown. diff --git a/src/main/codex/codex-structured-session-close.test.ts b/src/main/codex/codex-structured-session-close.test.ts index 4238c75de9e..b04e7bc2540 100644 --- a/src/main/codex/codex-structured-session-close.test.ts +++ b/src/main/codex/codex-structured-session-close.test.ts @@ -11,6 +11,8 @@ import { } from './codex-structured-session-adapter' import { handleCodexSessionExit } from './codex-structured-session-close' import type { CodexSession } from './codex-structured-session-state' +import type { StructuredAgentSessionAdapter } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { StructuredAgentSessionAdapterRouter } from '../native-chat/agent-session-wire/structured-agent-session-adapter-router' const THREAD = 'thread-1' @@ -60,6 +62,16 @@ function adapterFixture() { return { adapter, connections, events } } +function claudeAdapterStub(): StructuredAgentSessionAdapter { + return { + acquire: vi.fn(async () => ({ process: { pid: 1 } }) as never), + dispatch: vi.fn(), + cancelTurn: vi.fn(), + answerPrompt: vi.fn(), + setOption: vi.fn() + } +} + describe('Codex structured session close lifecycle', () => { it('forwards a one-shot exit when lifecycle admission is rejected', () => { const connection: CodexAppServerConnection = { @@ -164,4 +176,27 @@ describe('Codex structured session close lifecycle', () => { { cause: 'unexpected-exit', reason: 'sink failed', fence: 7 } ]) }) + + it('routes Codex sink-failure recovery through force-close and preserves unexpected-exit settlement', async () => { + const { adapter, connections, events } = adapterFixture() + const router = new StructuredAgentSessionAdapterRouter( + { claude: claudeAdapterStub(), codex: adapter }, + async () => {} + ) + await router.acquire({ identity: identity('session-1'), fence: 7, spawnToken: 'spawn-1' }) + const current = connections[0] + if (!current) { + throw new Error('missing connection') + } + current.connection.close = async () => { + current.handlers.onExit?.(new Error('journal sink failed')) + return true + } + + const forceCloseSession = router.forceCloseSession + await expect(forceCloseSession('session-1')).resolves.toBe(true) + expect(events.filter((event) => event.type === 'ended')).toMatchObject([ + { cause: 'unexpected-exit', reason: 'journal sink failed', fence: 7 } + ]) + }) }) diff --git a/src/main/codex/codex-structured-turn-processes.ts b/src/main/codex/codex-structured-turn-processes.ts index 6bb4b960a1b..163cbb098fb 100644 --- a/src/main/codex/codex-structured-turn-processes.ts +++ b/src/main/codex/codex-structured-turn-processes.ts @@ -57,6 +57,8 @@ async function terminateWindowsAddedProcesses( const added = current.filter((row) => baseline.get(row.pid) !== windowsIdentity(row)) const addedPids = new Set(added.map((row) => row.pid)) const roots = added.filter((row) => !addedPids.has(row.ppid)) + // Added roots come from a table walk, not a spawn, so a refused tree walk has + // no handle to fall back to: the row stays in `remaining` and this reports false. await Promise.all( roots.map((row) => terminateWindowsProcessTree(row.pid, { site: 'codex-turn-added-roots' })) ) diff --git a/src/main/crash-reporting/expected-teardown-state.ts b/src/main/crash-reporting/expected-teardown-state.ts index 1480ecbbfe2..c2993769633 100644 --- a/src/main/crash-reporting/expected-teardown-state.ts +++ b/src/main/crash-reporting/expected-teardown-state.ts @@ -10,9 +10,17 @@ type Clock = () => number const monotonicNow = (): number => performance.now() let now: Clock = monotonicNow let systemSessionEndedAt: number | null = null +let systemSessionEnded = false export function markSystemSessionEnding(): void { systemSessionEndedAt = now() + systemSessionEnded = true +} + +// Why latched, unlike the 5s crash-suppression window below: a native dialog or a recovery verdict is never +// right once the OS is tearing the session down, however long the process outlives the signal. +export function isSystemSessionEnding(): boolean { + return systemSessionEnded } function isRecentSystemSessionEnd(): boolean { @@ -55,4 +63,5 @@ export function resolveExpectedTeardownScope({ export function resetExpectedTeardownStateForTest(clock: Clock = monotonicNow): void { now = clock systemSessionEndedAt = null + systemSessionEnded = false } diff --git a/src/main/crash-reporting/self-initiated-tree-kill-log.test.ts b/src/main/crash-reporting/self-initiated-tree-kill-log.test.ts index c57c1796a35..e958dac42f8 100644 --- a/src/main/crash-reporting/self-initiated-tree-kill-log.test.ts +++ b/src/main/crash-reporting/self-initiated-tree-kill-log.test.ts @@ -17,17 +17,17 @@ import { recordProcessGoneCrash, type ProcessGoneCrashEvent } from './process-go import { resetProcessGoneSiblingCorrelationForTest } from './process-gone-sibling-correlation' import { findSelfInitiatedTreeKills, - installProcessTreeKillBreadcrumbObserver, recordRefusedOwnChromiumTreeKill, recordSelfInitiatedTreeKill, resetSelfInitiatedTreeKillLogForTest, selfInitiatedTreeKillDetails } from './self-initiated-tree-kill-log' import { - notifyProcessTreeKill, - setProcessTreeKillObserver -} from '../../shared/child-process/process-tree-kill-observer' + admitProcessTreeKill, + setProcessTreeKillGate +} from '../../shared/child-process/process-tree-kill-gate' import { terminateWindowsProcessTree } from '../windows-process-tree-kill' +import { installMainProcessTreeKillGate } from '../own-chromium-tree-kill-guard' import { _resetTracerForTests, setActiveSink } from '../observability/tracer' /** The field shape: renderer, `reason=killed exitCode=1`, win32 (#G2). */ @@ -252,12 +252,90 @@ describe('self-initiated tree kill breadcrumb', () => { expect(String(details.selfInitiatedKills)).toContain('more)') }) - it('records a kill issued through the shared runProcess choke point', () => { - installProcessTreeKillBreadcrumbObserver() + it('keeps the pid-addressed kill when a window-close burst overruns the ring', () => { + // Review probe: one taskkill, then 32 routine Job Object teardowns. Under + // plain FIFO the discriminating entry is evicted and the persisted detail + // becomes byte-identical to the external-kill arm. + const goneAt = 5_000_000 + recordSelfInitiatedTreeKill({ + pid: 4242, + site: 'pty-descendant-sweep', + scope: 'win-taskkill-tree', + at: goneAt - 4_000 + }) + for (let index = 0; index < 32; index += 1) { + recordSelfInitiatedTreeKill({ + pid: 6000 + index, + site: 'windows-pty-job-teardown', + scope: 'win-pty-job', + at: goneAt - 100 + }) + } - notifyProcessTreeKill({ pid: 3131, site: 'run-process-tree', scope: 'posix-process-group' }) + const details = selfInitiatedTreeKillDetails(goneAt) + + expect(details.selfInitiatedTreeKillCount).toBe(1) + expect(details.selfInitiatedGroupKillCount).toBe(31) + expect(String(details.selfInitiatedKills)).toMatch( + /^win-taskkill-tree\/pty-descendant-sweep\/pid4242 -4000ms/ + ) + }) + + it('keeps the newest teardown when a session has saturated the ring with pid kills', () => { + // Review probe, the mirror of the case above: 32 session-old taskkills (six + // routine families feed them) then the Job Object teardown 50ms before the + // death. A scope-preference eviction with no floor splices the entry it just + // pushed, and `{}` is byte-identical to the external-kill arm. + const goneAt = 5_000_000 + for (let index = 0; index < 32; index += 1) { + recordSelfInitiatedTreeKill({ + pid: 6000 + index, + site: 'pty-descendant-sweep', + scope: 'win-taskkill-tree', + at: goneAt - 600_000 + index * 1_000 + }) + } + recordSelfInitiatedTreeKill({ + pid: 7777, + site: 'windows-pty-job-teardown', + scope: 'win-pty-job', + at: goneAt - 50 + }) + + const details = selfInitiatedTreeKillDetails(goneAt) + + expect(details.selfInitiatedGroupKillCount).toBe(1) + expect(String(details.selfInitiatedKills)).toContain( + 'win-pty-job/windows-pty-job-teardown/pid7777 -50ms' + ) + }) + + it('evicts the oldest pid kill, not the newest, once every candidate is pid-addressed', () => { + const goneAt = 5_000_000 + for (let index = 0; index < 33; index += 1) { + recordSelfInitiatedTreeKill({ + pid: 6000 + index, + site: 'git-command-tree-kill', + scope: 'win-taskkill-tree', + at: goneAt - 1_000 + }) + } + + const pids = findSelfInitiatedTreeKills(goneAt).map((kill) => kill.pid) + + expect(pids).toHaveLength(32) + expect(pids).toContain(6032) + expect(pids).not.toContain(6000) + }) + + it('records a kill issued through the shared runProcess choke point', () => { + installMainProcessTreeKillGate() + + expect( + admitProcessTreeKill({ pid: 3131, site: 'run-process-tree', scope: 'posix-process-group' }) + ).toBe(true) expect(findSelfInitiatedTreeKills(Date.now()).map((kill) => kill.pid)).toEqual([3131]) - setProcessTreeKillObserver(null) + setProcessTreeKillGate(null) }) }) diff --git a/src/main/crash-reporting/self-initiated-tree-kill-log.ts b/src/main/crash-reporting/self-initiated-tree-kill-log.ts index e819795074f..809d421ac61 100644 --- a/src/main/crash-reporting/self-initiated-tree-kill-log.ts +++ b/src/main/crash-reporting/self-initiated-tree-kill-log.ts @@ -1,8 +1,5 @@ import type { CrashReportDetailValue } from '../../shared/crash-reporting' -import { - setProcessTreeKillObserver, - type ProcessTreeKillScope -} from '../../shared/child-process/process-tree-kill-observer' +import type { ProcessTreeKillScope } from '../../shared/child-process/process-tree-kill-gate' import { recordCoalescedDurableCrashBreadcrumb } from './durable-crash-breadcrumb' /** @@ -19,24 +16,38 @@ import { recordCoalescedDurableCrashBreadcrumb } from './durable-crash-breadcrum * The ring is per-process and its only reader is `process-gone-recorder`, which * exists in Electron main. So a count reported on a `render-process-gone` covers * kills issued *from Electron main*, and nothing else: - * - Main only: the three `taskkill /T /F` families that gate on - * `admitSelfInitiatedTreeKill` (`terminateWindowsProcessTree` and the codex / - * claude account-login teardowns) and the codex app-server POSIX group + * - Main only: the families that import the gate directly — + * `terminateWindowsProcessTree`, the codex and claude account-login + * teardowns, the git command-runner abort, the notebook-cell and + * automation-precheck timeouts — plus the codex app-server POSIX group * teardowns. - * - Main *and* other hosts: `signalProcessTree` (the `runProcess` choke point, - * reached from the CLI, relay and daemon too — a fourth pid-addressed - * `taskkill` family, gated on the child not being reaped rather than on the - * Chromium set it cannot read), the POSIX PTY process-group sweep and the - * Windows PTY Job Object (relay `pty-handler`, daemon - * `subprocess-handle`). When those run outside main they record into that - * process's own ring, which nothing reads — no observer is installed there, - * and the tracer sink is a no-op. - * - Never instrumented: the direct `process.kill(-pid)` calls in the browser - * routes, notebooks, automation prechecks and ephemeral-VM recipes. + * - Main *and* other hosts, through the `process-tree-kill-gate` seam main + * installs the same guard into: `signalProcessTree` (the `runProcess` choke + * point, reached from the CLI, relay and daemon too), the codex app-server + * deadline kill (compiled into the CLI as well) and the ephemeral-VM recipe + * kill. Also host-spanning but recording directly: the POSIX PTY + * process-group sweep and the Windows PTY Job Object (relay `pty-handler`, + * daemon `subprocess-handle`). When any of these run outside main they record + * into that process's own ring, which nothing reads — no gate is installed + * there, and the tracer sink is a no-op. + * - Never instrumented, and none of them a pid-addressed kill issued from main: + * the POSIX `process.kill(-pid, …)` group arms of the notebook, precheck, + * browser-route and ephemeral-VM kills, plus the macOS keyboard-input-source + * probe's group kill in `ipc/app.ts`; the relay's own + * `subprocess-tree-termination` taskkill and the CLI's login-interruption + * taskkill (neither runs in main); and the browser-route Electron probes, + * which are reached only from `*.electron.test.ts`. + * + * `main-process-tree-kill-gate.test.ts` is the ratchet that keeps that list + * closed: it counts `/pid` call sites against gate admissions per file, so a new + * pid-addressed kill fails it whether it lands in a new file or inside a family + * that already asks the gate. It does not see a `/pid` argument built from a + * variable. * * A daemon or relay kill missing from the count is a diagnostics gap, not a * missed suspect: those hosts cannot reach a Chromium pid in the first place - * (see `orca-chromium-process-pids.ts`). Absence is evidence, not proof. + * (see `orca-chromium-process-pids.ts`), and a group or Job-Object kill can + * only contain what Orca put in it. Absence is evidence, not proof. */ /** Which mechanism issued the kill; each has a different blast radius. */ @@ -81,6 +92,25 @@ function isPidAddressedTreeKill(scope: SelfInitiatedTreeKillScope): boolean { return scope === 'win-taskkill-tree' } +/** + * Drop one entry, newest-first-preserving. + * + * Two rules, in order. The entry just recorded is never a candidate: it is the + * one closest to any death that follows, and evicting it leaves a detail + * byte-identical to the external-kill arm. Among the rest, routine group/job + * teardown goes before a pid-addressed kill — a window-close burst is 30+ group + * kills and plain FIFO would drop the one entry that can explain the death — + * falling back to plain FIFO once every candidate is pid-addressed, which is + * what an ordinary session saturates the ring with. + */ +function evictOneSelfInitiatedTreeKill(): void { + const lastCandidate = selfInitiatedKills.length - 1 + const oldestGroupKill = selfInitiatedKills.findIndex( + (kill, index) => index < lastCandidate && !isPidAddressedTreeKill(kill.scope) + ) + selfInitiatedKills.splice(Math.max(oldestGroupKill, 0), 1) +} + export function recordSelfInitiatedTreeKill({ pid, site, @@ -96,13 +126,15 @@ export function recordSelfInitiatedTreeKill({ return } selfInitiatedKills.push({ pid, site, scope, at }) - if (selfInitiatedKills.length > MAX_TRACKED_SELF_KILLS) { - selfInitiatedKills = selfInitiatedKills.slice(-MAX_TRACKED_SELF_KILLS) + while (selfInitiatedKills.length > MAX_TRACKED_SELF_KILLS) { + evictOneSelfInitiatedTreeKill() } // Durable so it survives into the diagnostic bundle even when the kill takes // the reporting renderer with it; coalesced because the crash detail above is // the primary record and a teardown burst must not cost 30 ring slots plus a - // forced disk flush each. The newest pid still rides the emitted crumb. + // forced disk flush each. The retained ring crumb carries the newest pid, but + // the span trail emits only the first of a coalesced burst — read + // `selfInitiatedKills` for the rest. recordCoalescedDurableCrashBreadcrumb({ name: 'self_tree_kill', data: { pid, site, scope }, @@ -133,11 +165,6 @@ export function recordRefusedOwnChromiumTreeKill(target: { }) } -/** Routes the `runProcess` choke point's kills here; shared code cannot import us. */ -export function installProcessTreeKillBreadcrumbObserver(): void { - setProcessTreeKillObserver((kill) => recordSelfInitiatedTreeKill(kill)) -} - export function findSelfInitiatedTreeKills(at: number): SelfInitiatedTreeKill[] { return selfInitiatedKills.filter((kill) => { const offsetMs = kill.at - at diff --git a/src/main/daemon/agent-startup-prompt-latency.node-pty.test.ts b/src/main/daemon/agent-startup-prompt-latency.node-pty.test.ts new file mode 100644 index 00000000000..cfa01fed8e5 --- /dev/null +++ b/src/main/daemon/agent-startup-prompt-latency.node-pty.test.ts @@ -0,0 +1,123 @@ +import { appendFileSync, existsSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { createPtySubprocess } from './pty-subprocess' +import { Session } from './session' + +const SHELLS = process.platform === 'win32' ? [] : ['/bin/bash', '/bin/zsh'].filter(existsSync) +const COMMAND = "printf 'AGENT_%s\\n' STARTED" + +async function launch( + shell: string, + slow: boolean, + legacy = false +): Promise<{ output: string; ms: number }> { + const root = mkdtempSync(join(tmpdir(), 'orca-startup-latency-')) + const bash = shell.endsWith('bash') + const pause = slow ? 'sleep 0.6\n' : '' + const prompt = slow ? "PS1='$(sleep 0.3)prompt> '\n" : "PS1='prompt> '\n" + writeFileSync( + join(root, bash ? '.bash_profile' : '.zshrc'), + `${pause}${bash ? '' : 'setopt PROMPT_SUBST\n'}${prompt}` + ) + vi.stubEnv('HOME', root) + vi.stubEnv('ZDOTDIR', root) + vi.stubEnv('ORCA_ORIG_ZDOTDIR', root) + let session: Session | undefined + let timer: ReturnType | undefined + let legacyTimer: ReturnType | undefined + const readinessEvents: string[] = [] + const started = performance.now() + try { + const subprocess = await createPtySubprocess({ + sessionId: 'startup-latency', + cols: 120, + rows: 30, + cwd: root, + shellOverride: shell, + command: COMMAND, + env: { HOME: root, SHELL: shell, TERM: 'xterm-256color' } + }) + session = new Session({ + sessionId: 'startup-latency', + cols: 120, + rows: 30, + subprocess, + shellReadySupported: !legacy, + reportReadinessEvent: (event) => readinessEvents.push(event) + }) + const active = session + return await new Promise((resolve, reject) => { + let output = '' + timer = setTimeout( + () => reject(new Error(`Startup timed out: ${JSON.stringify(output)}`)), + 5000 + ) + active.attachClient({ + onExit: () => {}, + onData: (data) => { + output += data + if (output.includes('AGENT_STARTED')) { + resolve({ output, ms: performance.now() - started }) + } + } + }) + if (legacy) { + legacyTimer = setTimeout(() => active.write(`${COMMAND}\n`), 300) + } else { + active.write(`${COMMAND}\n`) + } + }) + } finally { + clearTimeout(timer) + clearTimeout(legacyTimer) + if (session) { + await session.forceKillAndWaitForExit(3000) + session.dispose() + } + vi.unstubAllEnvs() + rmSync(root, { recursive: true, force: true }) + expect(readinessEvents).toEqual([]) + } +} + +describe('agent startup at the rendered shell prompt', () => { + afterEach(() => vi.unstubAllEnvs()) + it.each(SHELLS)( + '%s displays the command once after slow startup and prompt expansion', + async (shell) => { + const before = await launch(shell, true, true) + expect(before.output.split(COMMAND)).toHaveLength(3) + const result = await launch(shell, true) + expect(result.output).not.toContain('orca-shell-ready') + expect(result.output.split(COMMAND)).toHaveLength(2) + expect(result.output.indexOf('prompt> ')).toBeLessThan(result.output.indexOf(COMMAND)) + } + ) + + it.skipIf(!process.env.ORCA_STARTUP_BENCH || SHELLS.length === 0)( + 'compares legacy input timing with prompt delivery', + async () => { + for (const shell of SHELLS) { + for (const slow of [false, true]) { + const legacy: number[] = [] + const current: number[] = [] + for (let i = 0; i < 5; i++) { + legacy.push((await launch(shell, slow, true)).ms) + const result = await launch(shell, slow) + expect(result.output).not.toContain('orca-shell-ready') + expect(result.output.split(COMMAND)).toHaveLength(2) + current.push(result.ms) + } + const result = JSON.stringify({ shell, slow, legacy, current }) + if (process.env.ORCA_STARTUP_BENCH_OUTPUT) { + appendFileSync(process.env.ORCA_STARTUP_BENCH_OUTPUT, `${result}\n`) + } + console.log(result) + } + } + }, + 60_000 + ) +}) diff --git a/src/main/daemon/daemon-bash-shell-ready-rcfile.ts b/src/main/daemon/daemon-bash-shell-ready-rcfile.ts index 142c0c131d9..f7834dba9d7 100644 --- a/src/main/daemon/daemon-bash-shell-ready-rcfile.ts +++ b/src/main/daemon/daemon-bash-shell-ready-rcfile.ts @@ -2,7 +2,6 @@ import { getPosixOmpShellWrapper } from '../pty/omp-shell-wrapper' import { getPosixCodexShellLaunchPreflight } from '../pty/codex-shell-launch-preflight' import { BASH_PROMPT_COMMAND_COMPOSITION_BLOCK } from '../bash-prompt-command-composition' import { BASH_FEATURE_CHANNEL_BLOCK, SHELL_STARTUP_IDENTITY_MARKER_BLOCK } from '../shell-templates' -import { SHELL_READY_MARKER } from './daemon-shell-ready-marker' export function getDaemonBashShellReadyRcfileContent(): string { return `# Orca daemon bash shell-ready wrapper @@ -56,10 +55,6 @@ __orca_osc133_precmd() { unset __orca_in_command fi printf "\\033]133;A\\007" - # Why: emit the shell-ready marker here (not a trailing PROMPT_COMMAND entry) - # so a framework that must be last in PROMPT_COMMAND — bash-preexec — is not - # displaced by one of Orca's own hooks. - [[ -n "$__orca_ready_marker" ]] && printf "${SHELL_READY_MARKER}" return "$exit_code" } __orca_osc133_preexec() { @@ -117,6 +112,11 @@ __orca_osc133_epilogue() { unset __orca_in_prompt_command __orca_adopt_outer_debug_trap trap '__orca_osc133_preexec' DEBUG + # Readline renders PS1 after entering raw mode; prompt hooks still run in cooked mode. + if [[ -n "$__orca_ready_marker" ]]; then + PS1="\${PS1-}"'\\[\\e]777;orca-shell-ready\\a\\]' + __orca_ready_marker="" + fi } ${BASH_PROMPT_COMMAND_COMPOSITION_BLOCK} __orca_prepend_prompt_command "__orca_osc133_precmd" diff --git a/src/main/daemon/daemon-foreground-process-protocol.ts b/src/main/daemon/daemon-foreground-process-protocol.ts index 25c6113057c..5c29e8a9e31 100644 --- a/src/main/daemon/daemon-foreground-process-protocol.ts +++ b/src/main/daemon/daemon-foreground-process-protocol.ts @@ -18,5 +18,7 @@ export type InspectProcessRequest = Omit & type: 'inspectProcess' payload: GetForegroundProcessRequest['payload'] & { expectedIncarnationId?: string + /** Optional; a daemon that predates it answers with the full capture as it always did. */ + steadyState?: boolean } } diff --git a/src/main/daemon/daemon-pty-adapter-steady-state-compat.test.ts b/src/main/daemon/daemon-pty-adapter-steady-state-compat.test.ts new file mode 100644 index 00000000000..ae7c29847c9 --- /dev/null +++ b/src/main/daemon/daemon-pty-adapter-steady-state-compat.test.ts @@ -0,0 +1,78 @@ +import { describe, expect, it, vi } from 'vitest' +import { DaemonPtyAdapter } from './daemon-pty-adapter' +import { COMPLETION_PROCESS_INSPECTION_PROTOCOL_VERSION, PROTOCOL_VERSION } from './types' + +type ClientInternals = { + client: { request: ReturnType; disconnect: ReturnType } +} + +function createAdapter( + protocolVersion: number, + request: ReturnType +): DaemonPtyAdapter { + const adapter = new DaemonPtyAdapter({ + socketPath: '/tmp/orca-steady-state-compat.sock', + tokenPath: '/tmp/orca-steady-state-compat.token', + protocolVersion + }) + ;(adapter as unknown as ClientInternals).client = { request, disconnect: vi.fn() } + return adapter +} + +describe('steadyState across daemon versions', () => { + it('sends steadyState as an additive optional field on the existing inspectProcess request', async () => { + const request = vi.fn(async () => ({ foregroundProcess: 'claude', hasChildProcesses: true })) + const adapter = createAdapter(PROTOCOL_VERSION, request) + await adapter.inspectProcess('sess-a', { steadyState: true }) + expect(request).toHaveBeenCalledWith('inspectProcess', { + sessionId: 'sess-a', + steadyState: true + }) + adapter.dispose() + }) + + it('omits the field entirely when not requested, so the wire is byte-identical to before', async () => { + const request = vi.fn(async () => ({ foregroundProcess: null, hasChildProcesses: false })) + const adapter = createAdapter(PROTOCOL_VERSION, request) + await adapter.inspectProcess('sess-a', { expectedIncarnationId: 'inc-1', steadyState: false }) + expect(request).toHaveBeenCalledWith('inspectProcess', { + sessionId: 'sess-a', + expectedIncarnationId: 'inc-1' + }) + adapter.dispose() + }) + + it('an old daemon that ignores steadyState still answers with the full-capture shape, and the client accepts it', async () => { + // A pre-field daemon returns exactly what it always did: name + evidence, never a cheap answer. + const oldDaemonAnswer = { + foregroundProcess: 'claude', + hasChildProcesses: true, + foregroundProcessEvidence: { + verdict: 'live', + processName: 'claude', + authorityGeneration: 'gen', + observationEpoch: 1, + capturedAgeMs: 0, + ptyId: 'sess-a', + ptyIncarnationId: 'inc-1' + } + } + const request = vi.fn(async () => oldDaemonAnswer) + const adapter = createAdapter(COMPLETION_PROCESS_INSPECTION_PROTOCOL_VERSION, request) + await expect(adapter.inspectProcess('sess-a', { steadyState: true })).resolves.toEqual( + oldDaemonAnswer + ) + adapter.dispose() + }) + + it('a pre-inspection daemon never sees the field: the client composes from getForegroundProcess as before', async () => { + const request = vi.fn(async () => ({ foregroundProcess: 'codex' })) + const adapter = createAdapter(COMPLETION_PROCESS_INSPECTION_PROTOCOL_VERSION - 1, request) + await expect(adapter.inspectProcess('sess-a', { steadyState: true })).resolves.toEqual({ + foregroundProcess: 'codex', + hasChildProcesses: true + }) + expect(request).toHaveBeenCalledWith('getForegroundProcess', { sessionId: 'sess-a' }) + adapter.dispose() + }) +}) diff --git a/src/main/daemon/daemon-pty-adapter.test.ts b/src/main/daemon/daemon-pty-adapter.test.ts index fc2f9365f23..5eba73d8ad2 100644 --- a/src/main/daemon/daemon-pty-adapter.test.ts +++ b/src/main/daemon/daemon-pty-adapter.test.ts @@ -27,8 +27,6 @@ const { isDaemonStaleForCurrentBundleMock: vi.fn(async () => false) })) -const itOnPosix = process.platform === 'win32' ? it.skip : it - vi.mock('./daemon-health', async (importOriginal) => { const actual = await importOriginal() return { @@ -343,37 +341,6 @@ describe('DaemonPtyAdapter (IPtyProvider)', () => { } } }) - - itOnPosix('keeps plain Codex startup on the short daemon shell-ready timeout', async () => { - await adapter.spawn({ - cols: 80, - rows: 24, - command: 'codex', - env: { SHELL: '/bin/zsh' } - }) - - await waitFor(() => vi.mocked(lastSubprocess.write).mock.calls.length > 0) - expect(lastSubprocess.write).toHaveBeenCalledWith('codex\n') - }) - - itOnPosix('waits for shell-ready for delivery-hinted Codex startup', async () => { - await adapter.spawn({ - cols: 80, - rows: 24, - command: "codex 'linked issue context'", - startupCommandDelivery: 'shell-ready', - env: { SHELL: '/bin/zsh' } - }) - - await new Promise((resolve) => setTimeout(resolve, 350)) - expect(lastSubprocess.write).not.toHaveBeenCalled() - - lastSubprocess._simulateData('\x1b]777;orca-shell-ready\x07') - lastSubprocess._simulateData('\r\nuser@host $ ') - - await waitFor(() => vi.mocked(lastSubprocess.write).mock.calls.length > 0) - expect(lastSubprocess.write).toHaveBeenCalledWith("codex 'linked issue context'\n") - }) }) describe('write', () => { diff --git a/src/main/daemon/daemon-pty-process-inspection.ts b/src/main/daemon/daemon-pty-process-inspection.ts index b05a8a6c6b6..335c641da62 100644 --- a/src/main/daemon/daemon-pty-process-inspection.ts +++ b/src/main/daemon/daemon-pty-process-inspection.ts @@ -25,7 +25,7 @@ export abstract class DaemonPtyProcessInspection extends DaemonPtyBufferSnapshot async inspectProcess( id: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; steadyState?: boolean } ): Promise { if (this.protocolVersion < GET_FOREGROUND_PROCESS_PROTOCOL_VERSION) { return clientOnlyUnverifiableInspection('old_host') @@ -47,7 +47,9 @@ export abstract class DaemonPtyProcessInspection extends DaemonPtyBufferSnapshot sessionId: id, ...(options?.expectedIncarnationId ? { expectedIncarnationId: options.expectedIncarnationId } - : {}) + : {}), + // Additive: an older daemon ignores it and pays for the full capture. + ...(options?.steadyState === true ? { steadyState: true } : {}) }) } diff --git a/src/main/daemon/daemon-pty-router.ts b/src/main/daemon/daemon-pty-router.ts index 78e12215504..962cde6760e 100644 --- a/src/main/daemon/daemon-pty-router.ts +++ b/src/main/daemon/daemon-pty-router.ts @@ -179,7 +179,7 @@ export class DaemonPtyRouter implements IPtyProvider { async inspectProcess( id: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; steadyState?: boolean } ): Promise { return this.adapterForInspection(id).inspectProcess(id, options) } diff --git a/src/main/daemon/daemon-pty-session-spawn.ts b/src/main/daemon/daemon-pty-session-spawn.ts index 235b286b0bc..bbb899b64a3 100644 --- a/src/main/daemon/daemon-pty-session-spawn.ts +++ b/src/main/daemon/daemon-pty-session-spawn.ts @@ -1,5 +1,6 @@ import { recognizeAgentProcessFromCommandLine } from '../../shared/agent-process-recognition' import { shouldUseShellReadyStartupDelivery } from '../../shared/codex-startup-delivery' +import { CODEX_SHELL_READY_TIMEOUT_MS } from './session-shell-ready-barrier' import type { HistoryRecoveryContext, PendingDaemonSpawnOperation @@ -11,8 +12,8 @@ import { DaemonPtySpawnResult } from './daemon-pty-spawn-result' import type { DaemonPtySpawnContext } from './daemon-pty-spawn-request' import type { ColdRestoreInfo } from './history-reader' import { mintPtySessionId } from './pty-session-id' -import { CODEX_SHELL_READY_TIMEOUT_MS } from './session-shell-ready-barrier' -import { supportsPtyStartupBarrier } from './shell-ready' +import { shellPathSupportsPtyStartupBarrier, resolvePtyShellPath } from './shell-ready' +import { shellReadyMarkerComesFromLineEditor } from '../../shared/shell-ready-marker-timing' import { getRecoveredHistorySeedSegments } from './terminal-history-seed-segments' import { AGENT_SESSION_CLAIM_DAEMON_PROTOCOL_VERSION, type CreateOrAttachResult } from './types' import { normalizeWslColdRestoreCwd } from './wsl-cold-restore-cwd' @@ -212,22 +213,26 @@ export abstract class DaemonPtySessionSpawn extends DaemonPtySpawnResult { let effectiveCols = restoreInfo?.cols ?? opts.cols let effectiveRows = restoreInfo?.rows ?? opts.rows - const shellReadySupported = opts.command ? supportsPtyStartupBarrier(opts.env ?? {}) : false - const isCodexStartupCommand = - recognizeAgentProcessFromCommandLine(opts.command)?.agent === 'codex' - const shouldWaitForShellReady = - isCodexStartupCommand && - shouldUseShellReadyStartupDelivery({ + const effectiveShellPath = + process.platform !== 'win32' && opts.command + ? resolveUnixShellPath(opts.shellOverride || resolvePtyShellPath(opts.env ?? {})) + : '' + const shellReadySupported = shellPathSupportsPtyStartupBarrier(effectiveShellPath) + const immediateMarker = shellReadyMarkerComesFromLineEditor(effectiveShellPath) + const shellReadyTimeoutMs = + shellReadySupported && + !immediateMarker && + recognizeAgentProcessFromCommandLine(opts.command)?.agent === 'codex' && + !shouldUseShellReadyStartupDelivery({ command: opts.command, startupCommandDelivery: opts.startupCommandDelivery }) - const shellReadyTimeoutMs = - shellReadySupported && isCodexStartupCommand && !shouldWaitForShellReady ? CODEX_SHELL_READY_TIMEOUT_MS : undefined - const context: DaemonPtySpawnContext = { - opts, + // Older daemons also need the existing hint to enable their ready marker. + opts: + opts.command && immediateMarker ? { ...opts, startupCommandDelivery: 'shell-ready' } : opts, operation, historyRecovery, requestedSessionId, diff --git a/src/main/daemon/daemon-pty-startup-delivery.test.ts b/src/main/daemon/daemon-pty-startup-delivery.test.ts new file mode 100644 index 00000000000..4b72070bb42 --- /dev/null +++ b/src/main/daemon/daemon-pty-startup-delivery.test.ts @@ -0,0 +1,107 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { rmSync } from 'node:fs' +import { basename, join } from 'node:path' +import * as localPtyUtils from '../providers/local-pty-utils' +import { + createMockSubprocess, + startDaemonAdapterHarness, + waitFor, + type DaemonAdapterHarness, + type SpawnSubprocess +} from './daemon-pty-adapter-test-harness' + +const itOnPosix = process.platform === 'win32' ? it.skip : it + +describe('DaemonPtyAdapter startup delivery', () => { + let harness: DaemonAdapterHarness + let adapter: DaemonAdapterHarness['adapter'] + let dir: string + let lastSubprocess: ReturnType + let lastSpawnOpts: Parameters[0] | null + + beforeEach(async () => { + lastSpawnOpts = null + harness = await startDaemonAdapterHarness((opts) => { + lastSpawnOpts = opts + lastSubprocess = createMockSubprocess() + return lastSubprocess + }) + adapter = harness.adapter + dir = harness.dir + }) + + afterEach(async () => { + adapter.dispose() + await harness.server.shutdown() + rmSync(dir, { recursive: true, force: true }) + }) + + itOnPosix('preserves the existing fast-start timing for fish', async () => { + // The mock subprocess represents installed fish even on hosts without it. + const resolveShell = vi + .spyOn(localPtyUtils, 'resolveUnixShellPath') + .mockReturnValue('/usr/bin/fish') + vi.useFakeTimers({ toFake: ['setTimeout', 'clearTimeout'] }) + try { + await adapter.spawn({ + cols: 80, + rows: 24, + command: 'codex', + env: { SHELL: '/usr/bin/fish' } + }) + await vi.advanceTimersByTimeAsync(299) + expect(lastSubprocess.write).not.toHaveBeenCalled() + await vi.advanceTimersByTimeAsync(1) + expect(lastSubprocess.write).toHaveBeenCalledExactlyOnceWith('codex\n') + expect(lastSpawnOpts).not.toEqual( + expect.objectContaining({ startupCommandDelivery: 'shell-ready' }) + ) + } finally { + vi.useRealTimers() + resolveShell.mockRestore() + } + }) + + itOnPosix.for(['environment', 'override'] as const)( + 'waits for the fallback shell when the %s shell is missing', + async (source, context) => { + const missingShell = join(dir, 'missing-fish') + const fallbackName = basename(localPtyUtils.resolveUnixShellPath(missingShell)) + context.skip(!['bash', 'zsh'].includes(fallbackName), 'Requires a Bash/zsh fallback') + await adapter.spawn({ + cols: 80, + rows: 24, + command: 'codex', + env: { SHELL: source === 'environment' ? missingShell : '/bin/sh' }, + ...(source === 'override' ? { shellOverride: missingShell } : {}) + }) + await new Promise((resolve) => setTimeout(resolve, 350)) + expect(lastSubprocess.write).not.toHaveBeenCalled() + expect(lastSpawnOpts).toEqual( + expect.objectContaining({ startupCommandDelivery: 'shell-ready' }) + ) + lastSubprocess._simulateData('\x1b]777;orca-shell-ready\x07\r\nuser@host $ ') + await waitFor(() => vi.mocked(lastSubprocess.write).mock.calls.length > 0) + expect(lastSubprocess.write).toHaveBeenCalledExactlyOnceWith('codex\n') + } + ) + + itOnPosix.each([ + { command: 'codex' }, + { command: 'codex', startupCommandDelivery: 'fast' as const }, + { command: "codex 'linked issue context'", startupCommandDelivery: 'shell-ready' as const } + ])('waits past 300ms and submits once after readiness: %j', async (startup) => { + await adapter.spawn({ cols: 80, rows: 24, ...startup, env: { SHELL: '/bin/zsh' } }) + + await new Promise((resolve) => setTimeout(resolve, 350)) + expect(lastSubprocess.write).not.toHaveBeenCalled() + expect(lastSpawnOpts).toEqual( + expect.objectContaining({ startupCommandDelivery: 'shell-ready' }) + ) + lastSubprocess._simulateData('\x1b]777;orca-shell-ready\x07') + lastSubprocess._simulateData('\r\nuser@host $ ') + + await waitFor(() => vi.mocked(lastSubprocess.write).mock.calls.length > 0) + expect(lastSubprocess.write).toHaveBeenCalledExactlyOnceWith(`${startup.command}\n`) + }) +}) diff --git a/src/main/daemon/daemon-request-router.ts b/src/main/daemon/daemon-request-router.ts index e081281fa8a..bb7d0d1a256 100644 --- a/src/main/daemon/daemon-request-router.ts +++ b/src/main/daemon/daemon-request-router.ts @@ -105,12 +105,17 @@ export class DaemonRequestRouter { return { foregroundProcess: this.options.host.getForegroundProcess(request.payload.sessionId) } - case 'inspectProcess': - return request.payload.expectedIncarnationId - ? this.options.host.inspectProcess(request.payload.sessionId, { - expectedIncarnationId: request.payload.expectedIncarnationId - }) + case 'inspectProcess': { + const options = { + ...(request.payload.expectedIncarnationId + ? { expectedIncarnationId: request.payload.expectedIncarnationId } + : {}), + ...(request.payload.steadyState === true ? { steadyState: true } : {}) + } + return Object.keys(options).length > 0 + ? this.options.host.inspectProcess(request.payload.sessionId, options) : this.options.host.inspectProcess(request.payload.sessionId) + } case 'confirmForegroundProcess': return { foregroundProcess: await this.options.host.confirmForegroundProcess( diff --git a/src/main/daemon/post-ready-flush-gate.test.ts b/src/main/daemon/post-ready-flush-gate.test.ts index f8e58ba7c7d..06ff64db99f 100644 --- a/src/main/daemon/post-ready-flush-gate.test.ts +++ b/src/main/daemon/post-ready-flush-gate.test.ts @@ -20,6 +20,14 @@ describe('PostReadyFlushGate', () => { vi.useRealTimers() }) + it('flushes synchronously when the marker comes from the line editor', () => { + gate = new PostReadyFlushGate(onFlush, true) + gate.arm() + expect(onFlush).toHaveBeenCalledTimes(1) + expect(gate.isPending).toBe(false) + expect(vi.getTimerCount()).toBe(0) + }) + it('does not flush immediately when armed', () => { gate.arm() expect(onFlush).not.toHaveBeenCalled() diff --git a/src/main/daemon/post-ready-flush-gate.ts b/src/main/daemon/post-ready-flush-gate.ts index 199bb601024..f7f7a2f1869 100644 --- a/src/main/daemon/post-ready-flush-gate.ts +++ b/src/main/daemon/post-ready-flush-gate.ts @@ -1,23 +1,5 @@ -/** - * Defers a flush callback until after the shell has drawn its prompt and - * switched the PTY into raw mode. - * - * Why: the OSC 777 shell-ready marker fires from zsh's precmd_functions / - * bash's PROMPT_COMMAND — before the shell draws its prompt and before - * zle/readline flips the PTY into raw mode. Flushing queued input then lets - * the kernel (ECHO still on) echo the command once, and the line editor - * redraws it under the prompt — producing a visible duplicate (e.g. "claude" - * appears twice on agent launch). - * - * Strategy: after arm() is called, wait for prompt bytes plus a short delay - * for the tcsetattr() that enables raw mode. If the marker-completing scan - * already saw post-marker bytes, use that same short path immediately. - * A conservative wall-clock fallback covers ambiguous marker-only cases. - * - * Mirrors the gate in local-pty-shell-ready.ts::writeStartupCommandWhenShellReady, - * which solves the same race on the non-daemon path. - */ - +// Bash's prompt and zsh's line-init marker are ready for input immediately. +// Other shells retain the existing settling delay. export const POST_READY_FLUSH_DELAY_MS = 30 export const POST_READY_FLUSH_FALLBACK_MS = 200 @@ -26,7 +8,10 @@ export class PostReadyFlushGate { private postDataTimer: ReturnType | null = null private fallbackTimer: ReturnType | null = null - constructor(private readonly onFlush: () => void) {} + constructor( + private readonly onFlush: () => void, + private readonly markerIsLineEditorReady = false + ) {} /** True between arm() and the actual flush firing. Callers should treat * input as still-queued during this window to preserve ordering. */ @@ -38,6 +23,10 @@ export class PostReadyFlushGate { * wall-clock fallback unless the marker scan already observed post-marker * bytes, in which case the short post-data settle path is enough. */ arm(postMarkerBytesObserved = false): void { + if (this.markerIsLineEditorReady) { + this.onFlush() + return + } this.awaitingPromptDraw = true if (postMarkerBytesObserved) { this.notifyData() diff --git a/src/main/daemon/pty-subprocess-managed-agent-env.test.ts b/src/main/daemon/pty-subprocess-managed-agent-env.test.ts index 5267778088c..d5dbc33246c 100644 --- a/src/main/daemon/pty-subprocess-managed-agent-env.test.ts +++ b/src/main/daemon/pty-subprocess-managed-agent-env.test.ts @@ -261,7 +261,7 @@ describe('createPtySubprocess', () => { expect(lastCall[2].env.ORCA_SHELL_FEATURES).not.toContain('ready') }) - it('keeps plain Codex startup commands on the no-marker wrapper', async () => { + it('enables readiness and shell identity for plain Codex startup', async () => { const proc = mockPtyProcess() spawnMock.mockReturnValue(proc) const platform = Object.getOwnPropertyDescriptor(process, 'platform') @@ -285,7 +285,8 @@ describe('createPtySubprocess', () => { const lastCall = spawnMock.mock.calls.at(-1)! expect(lastCall[1]).toEqual(['-l']) expect(lastCall[2].env.ZDOTDIR).toMatch(ZSH_SHELL_READY_DIR) - expect(lastCall[2].env.ORCA_SHELL_FEATURES).not.toContain('ready') + expect(lastCall[2].env.ORCA_SHELL_FEATURES).toContain('ready') + expect(lastCall[2].env.ORCA_SHELL_FEATURES).toContain('identity') }) it('uses shell-ready wrapper for delivery-hinted Codex startup commands', async () => { diff --git a/src/main/daemon/pty-subprocess/foreground-process-tracker.ts b/src/main/daemon/pty-subprocess/foreground-process-tracker.ts index 115b795202d..8726dc87281 100644 --- a/src/main/daemon/pty-subprocess/foreground-process-tracker.ts +++ b/src/main/daemon/pty-subprocess/foreground-process-tracker.ts @@ -36,7 +36,9 @@ type CachedAgentForeground = { processName: string; pid: number | null; refreshe export type PtyForegroundProcessTracker = { recordOutput(data: string): void markDead(): void - getForegroundProcess(): string | null + /** `rawFallback`: node-pty's own name only, with no identity cache and no background + * process-table refresh -- the cheap-tier tick must not fork a full `ps` as a side effect. */ + getForegroundProcess(options?: { rawFallback?: boolean }): string | null confirmForegroundProcess(): Promise confirmShellForeground(): Promise } @@ -213,10 +215,13 @@ export function createPtyForegroundProcessTracker(args: { cachedAgentForeground = null startupAgentForeground = null }, - getForegroundProcess: () => { + getForegroundProcess: (options) => { if (args.isDead()) { return null } + if (options?.rawFallback === true) { + return getFallbackProcess() + } try { const fallbackProcess = getFallbackProcess() const fallbackRecognition = recognizeAgentProcess(fallbackProcess) diff --git a/src/main/daemon/pty-subprocess/shell-launch-plan.ts b/src/main/daemon/pty-subprocess/shell-launch-plan.ts index ed8cd5e6a22..ba60e592743 100644 --- a/src/main/daemon/pty-subprocess/shell-launch-plan.ts +++ b/src/main/daemon/pty-subprocess/shell-launch-plan.ts @@ -1,3 +1,4 @@ +import { shouldUseShellReadyStartupDelivery } from '../../../shared/codex-startup-delivery' import { win32 as pathWin32 } from 'node:path' import { isWindowsGitBashShellPath, resolveWindowsGitBashShellPath } from '../../git-bash' import { isPwshAvailable } from '../../pwsh' @@ -32,7 +33,6 @@ import { recognizeAgentProcessFromCommandLine, type RecognizedAgentProcess } from '../../../shared/agent-process-recognition' -import { shouldUseShellReadyStartupDelivery } from '../../../shared/codex-startup-delivery' import { ORCA_HERMES_STARTUP_QUERY_ENV } from '../../../shared/hermes-startup-query' import { WINDOWS_GIT_BASH_SHELL } from '../../../shared/windows-terminal-shell' import { getShellLaunchConfig, resolvePtyShellPath } from '../shell-ready' @@ -60,7 +60,6 @@ export function createPtyShellLaunchPlan( let startupCommandDeliveredInShellArgs = false let windowsFallbackAttempts: WindowsShellSpawnAttempt[] = [] const startupAgentRecognition = recognizeAgentProcessFromCommandLine(opts.command) - const isCodexStartupCommand = startupAgentRecognition?.agent === 'codex' const requestedCwd = opts.cwd || resolveSafePtyDefaultCwd() if (opts.command && startupAgentRecognition) { assertSafeAgentStartupCwd(requestedCwd, opts.command) @@ -192,10 +191,11 @@ export function createPtyShellLaunchPlan( } const waitsForShellReady = Boolean(opts.command) && - (!isCodexStartupCommand || + (startupAgentRecognition?.agent !== 'codex' || shouldUseShellReadyStartupDelivery({ - command: opts.command as string, - startupCommandDelivery: opts.startupCommandDelivery + command: opts.command, + startupCommandDelivery: opts.startupCommandDelivery, + shellPath })) delete env.ORCA_SHELL_FEATURES const shellLaunch = getShellLaunchConfig( diff --git a/src/main/daemon/repro-13767-shell-ready-marker-lost-to-exec.test.ts b/src/main/daemon/repro-13767-shell-ready-marker-lost-to-exec.test.ts index 52462e3a2eb..a1db9461070 100644 --- a/src/main/daemon/repro-13767-shell-ready-marker-lost-to-exec.test.ts +++ b/src/main/daemon/repro-13767-shell-ready-marker-lost-to-exec.test.ts @@ -1,7 +1,7 @@ import { spawnSync } from 'node:child_process' -import { existsSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { existsSync, mkdtempSync, realpathSync, rmSync, writeFileSync } from 'node:fs' import { tmpdir } from 'node:os' -import { join } from 'node:path' +import { dirname, join } from 'node:path' import { describe, expect, it, vi } from 'vitest' import { createPtySubprocess } from './pty-subprocess' import { Session } from './session' @@ -10,6 +10,20 @@ const describePosix = process.platform === 'win32' ? describe.skip : describe const hasZsh = process.platform !== 'win32' && spawnSync('/bin/zsh', ['--version']).status === 0 const hasBash = process.platform !== 'win32' && spawnSync('/bin/bash', ['--version']).status === 0 const COMMAND_OUTPUT = 'ORCA_STARTUP_COMMAND_RAN' +// A second Bash install with its own canonical path -- the shape a login profile +// switches to (`exec /opt/homebrew/bin/bash`) and the one #18768 stalled on. A +// symlink cannot stand in: both sides are realpath'd before they are compared. +const alternateBashPath = ['/opt/homebrew/bin/bash', '/usr/local/bin/bash', '/usr/bin/bash'].find( + (candidate) => + hasBash && existsSync(candidate) && realpathSync(candidate) !== realpathSync('/bin/bash') +) +if (process.platform !== 'win32' && !alternateBashPath) { + // Why announced: usrmerge hosts resolve /usr/bin/bash back to /bin/bash, so these + // two skip on most Linux CI. A silent skip reads as coverage that does not exist. + console.warn( + '[repro-13767] no second Bash install with a distinct realpath; skipping the alternate-install recovery tests' + ) +} const READ_STARTED_FILE = '.orca-read-started' type ShellFixture = { @@ -122,7 +136,8 @@ type RunningFixture = { async function startFixture( fixture: ShellFixture, startupContent: string, - extraFiles: Record = {} + extraFiles: Record = {}, + pathEnv: string = process.env.PATH ?? '/usr/bin:/bin' ): Promise { const tempHome = mkdtempSync(join(tmpdir(), 'orca-shell-ready-exec-')) const previousHome = process.env.HOME @@ -150,7 +165,7 @@ async function startFixture( shellOverride: fixture.shellPath, env: { HOME: tempHome, - PATH: process.env.PATH ?? '/usr/bin:/bin', + PATH: pathEnv, SHELL: fixture.shellPath, TERM: 'xterm-256color' }, @@ -416,4 +431,49 @@ fi }, 10_000 ) + + const bashFixture = FIXTURES[2] as ShellFixture + const alternateBashTest = alternateBashPath ? it : it.skip + const alternateBashProfile = `if [[ -z "\${ORCA_EXEC_REPRO_DONE:-}" ]]; then + export ORCA_EXEC_REPRO_DONE=1 + exec ${alternateBashPath ?? '/bin/bash'} --noprofile --norc -l -i +fi +` + + alternateBashTest( + 'releases at the prompt of a second Bash install the pane PATH resolves', + async () => { + const running = await startFixture( + bashFixture, + alternateBashProfile, + {}, + `${dirname(alternateBashPath ?? '/bin/bash')}:/usr/bin:/bin` + ) + try { + await waitForOutput(running.subscribe, () => running.output().includes(COMMAND_OUTPUT)) + expect(running.session.shellState).toBe('ready') + expect(count(running.output(), COMMAND_OUTPUT)).toBe(1) + expect(running.output()).not.toContain('orca-shell-start') + } finally { + await running.cleanup() + } + }, + 10_000 + ) + + alternateBashTest( + 'does not trust a Bash install that the pane PATH cannot reach', + async () => { + const running = await startFixture(bashFixture, alternateBashProfile, {}, '/usr/bin:/bin') + try { + await waitForOutput(running.subscribe, () => running.output().includes('$')) + await new Promise((resolve) => setTimeout(resolve, 500)) + expect(running.session.shellState).toBe('pending') + expect(running.output()).not.toContain(COMMAND_OUTPUT) + } finally { + await running.cleanup() + } + }, + 10_000 + ) }) diff --git a/src/main/daemon/session-shell-ready-barrier.ts b/src/main/daemon/session-shell-ready-barrier.ts index e7562e23224..6fce89af89e 100644 --- a/src/main/daemon/session-shell-ready-barrier.ts +++ b/src/main/daemon/session-shell-ready-barrier.ts @@ -1,3 +1,4 @@ +import { shellReadyMarkerComesFromLineEditor } from '../../shared/shell-ready-marker-timing' import { installDeviceAttributesResponder, STARTUP_DA1_RESPONSE @@ -19,7 +20,6 @@ import { basename } from 'node:path' import type { ShellReadyState } from './types' const SHELL_READY_TIMEOUT_MS = 15_000 -// Why: Codex skips marker-gated command delivery; this only bounds older daemon/local paths that still report shell-ready for Codex. export const CODEX_SHELL_READY_TIMEOUT_MS = 300 export type SessionShellReadyBarrierDeps = { @@ -69,7 +69,10 @@ export class SessionShellReadyBarrier { this._state = 'unsupported' } - this.postReadyFlushGate = new PostReadyFlushGate(() => this.flushPreReadyQueue()) + this.postReadyFlushGate = new PostReadyFlushGate( + () => this.flushPreReadyQueue(), + shellReadyMarkerComesFromLineEditor(deps.subprocess.shellPath ?? '') + ) } get state(): ShellReadyState { diff --git a/src/main/daemon/session-subprocess-handle.ts b/src/main/daemon/session-subprocess-handle.ts index f14469afbb4..9268686d78e 100644 --- a/src/main/daemon/session-subprocess-handle.ts +++ b/src/main/daemon/session-subprocess-handle.ts @@ -6,7 +6,7 @@ export type SubprocessHandle = { pid: number /** Live foreground process name of the PTY (node-pty's `.process`), e.g. * 'claude' / 'codex' / 'zsh'. Null once the child has exited. */ - getForegroundProcess(): string | null + getForegroundProcess(options?: { rawFallback?: boolean }): string | null /** Await process-table evidence captured after this confirmation request. */ confirmForegroundProcess?(): Promise /** Proves a fresh post-boundary PTY process tree contains only the shell. */ diff --git a/src/main/daemon/session.ts b/src/main/daemon/session.ts index b0f1dfa538d..9265b2acfef 100644 --- a/src/main/daemon/session.ts +++ b/src/main/daemon/session.ts @@ -252,8 +252,8 @@ export class Session { return this.output.getCwd() } - getForegroundProcess(): string | null { - return this.subprocess.getForegroundProcess() + getForegroundProcess(options?: { rawFallback?: boolean }): string | null { + return this.subprocess.getForegroundProcess(options) } async confirmForegroundProcess(): Promise { diff --git a/src/main/daemon/shell-ready.test.ts b/src/main/daemon/shell-ready.test.ts index 6772a52d46e..01cdad10be9 100644 --- a/src/main/daemon/shell-ready.test.ts +++ b/src/main/daemon/shell-ready.test.ts @@ -198,11 +198,9 @@ describePosix('daemon shell-ready launch config', () => { }) it('extends the startup barrier to fish so launch commands queue until the prompt', async () => { - const { shellPathSupportsPtyStartupBarrier, supportsPtyStartupBarrier } = - await importFreshShellReady() + const { shellPathSupportsPtyStartupBarrier } = await importFreshShellReady() expect(shellPathSupportsPtyStartupBarrier('/opt/homebrew/bin/fish')).toBe(true) - expect(supportsPtyStartupBarrier({ SHELL: '/usr/local/bin/fish' })).toBe(true) // Why: unwrapped shells must stay off the barrier or their first command queues forever. expect(shellPathSupportsPtyStartupBarrier('/usr/bin/tcsh')).toBe(false) }) diff --git a/src/main/daemon/shell-ready.ts b/src/main/daemon/shell-ready.ts index 9dc60b7c980..5208f9ced6b 100644 --- a/src/main/daemon/shell-ready.ts +++ b/src/main/daemon/shell-ready.ts @@ -108,13 +108,6 @@ export function shellPathSupportsPtyStartupBarrier(shellPath: string): boolean { return shellName === 'zsh' || shellName === 'bash' || shellName === 'fish' } -export function supportsPtyStartupBarrier(env: Record): boolean { - if (process.platform === 'win32') { - return false - } - return shellPathSupportsPtyStartupBarrier(resolvePtyShellPath(env)) -} - export type ShellLaunchConfig = { args: string[] | null env: Record diff --git a/src/main/daemon/terminal-host-cheap-tier-ps-scan-volume.test.ts b/src/main/daemon/terminal-host-cheap-tier-ps-scan-volume.test.ts new file mode 100644 index 00000000000..b02eec33773 --- /dev/null +++ b/src/main/daemon/terminal-host-cheap-tier-ps-scan-volume.test.ts @@ -0,0 +1,217 @@ +// Measurement for the cheap-tier process inspection. Drives the REAL daemon inspection +// entrypoint (`inspectTerminalHostProcess`) for 8 idle agent panes over a simulated 60s idle +// cadence (POLL_TIER_INTERVAL_MS.idle = 2,000ms) and counts `ps` forks BY COLUMN SET: a fork +// asking for `command=` is the full capture (measured 0.34-0.50s on a 1,900-process Mac, 1.15s +// on Linux), one without it is the cheap capture (0.03s on both). CI cannot time a real `ps` +// portably, so fork counts by column set are what this test measures; the per-fork costs above +// are the numbers measured by hand on the reference hosts. +// +// The second test is the zero-trade-off proof: the same tick sequence, including an agent exit +// and a restart, produces the identical foregroundProcess series with the cheap tier on and off. +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +// Two seams because the two tiers spawn differently: the full evidence reader still forks +// through node:child_process, the cheap reader through Orca's runProcess entry point. +const { execFileMock, runProcessMock } = vi.hoisted(() => ({ + execFileMock: vi.fn(), + runProcessMock: vi.fn() +})) +vi.mock('node:child_process', () => ({ execFile: execFileMock })) +vi.mock('../../shared/child-process/run-process', () => ({ runProcess: runProcessMock })) + +import { resetCheapProcessTableSnapshotForTests } from '../../shared/cheap-process-table-snapshot-reader' +import { resetProcessTableSnapshotForTests } from '../../shared/process-table-snapshot-reader' +import { createPtyForegroundProcessTracker } from './pty-subprocess/foreground-process-tracker' +import { inspectTerminalHostProcess } from './terminal-host-process-inspection' +import type { Session } from './session' + +const PANE_COUNT = 8 +const IDLE_POLL_INTERVAL_MS = 2_000 // POLL_TIER_INTERVAL_MS.idle +const WINDOW_SECONDS = 60 +const TICKS = Math.floor((WINDOW_SECONDS * 1000) / IDLE_POLL_INTERVAL_MS) + +const shellPid = (pane: number): number => 1000 + pane * 100 +const agentPid = (pane: number): number => shellPid(pane) + 1 + +type PaneState = { agent: boolean; agentStart: string } +const panes: PaneState[] = Array.from({ length: PANE_COUNT }, () => ({ + agent: true, + agentStart: 'Thu Sep 3 16:02:05 2026' +})) + +const forks = { full: 0, cheap: 0 } + +function renderRows(): { full: string; cheap: string } { + const full: string[] = [] + const cheap: string[] = [] + panes.forEach((pane, i) => { + const s = shellPid(i) + const a = agentPid(i) + const tpgid = pane.agent ? a : s + const shellStat = pane.agent ? 'Ss' : 'Ss+' + cheap.push(`${s} 1 ${s} ${tpgid} ${shellStat} Thu Sep 3 16:02:01 2026`) + full.push(`${s} 1 ${s} ${tpgid} ${shellStat} ttys00${i} Thu Sep 3 16:02:01 2026 -zsh`) + if (pane.agent) { + cheap.push(`${a} ${s} ${a} ${a} S+ ${pane.agentStart}`) + full.push(`${a} ${s} ${a} ${a} S+ ttys00${i} ${pane.agentStart} node /usr/local/bin/claude`) + } + }) + return { full: `${full.join('\n')}\n`, cheap: `${cheap.join('\n')}\n` } +} + +function installCountingPs(): void { + execFileMock.mockImplementation((_cmd: string, args: string[], _opts: unknown, cb: unknown) => { + expect(args[1]).toContain('command=') + forks.full += 1 + ;(cb as (err: unknown, r: { stdout: string; stderr: string }) => void)(null, { + stdout: renderRows().full, + stderr: '' + }) + }) + runProcessMock.mockImplementation(async (spec: { args: readonly string[] }) => { + expect(spec.args[1]).not.toContain('command=') + forks.cheap += 1 + return { code: 0, signal: null, stdout: renderRows().cheap, stderr: '', timedOut: false } + }) +} + +function createSession(pane: number): Session { + const tracker = createPtyForegroundProcessTracker({ + process: { + pid: shellPid(pane), + get process() { + return panes[pane].agent ? 'node' : 'zsh' + } + } as never, + shellPath: '/bin/zsh', + sessionId: `wt:pane-${pane}`, + startupAgentRecognition: null, + isDead: () => false + }) + return { + pid: shellPid(pane), + incarnationId: `inc-${pane}`, + isAlive: true, + getForegroundProcess: (options?: { rawFallback?: boolean }) => + tracker.getForegroundProcess(options) + } as unknown as Session +} + +async function settle(): Promise { + for (let i = 0; i < 8; i += 1) { + await Promise.resolve() + } +} + +async function runTick(sessions: Session[], steadyState: boolean): Promise<(string | null)[]> { + const results = await Promise.all( + sessions.map((session, pane) => + inspectTerminalHostProcess({ + sessionId: `wt:pane-${pane}`, + session, + ...(steadyState ? { steadyState: true } : {}), + authorityGeneration: 'gen', + nextObservationEpoch: () => 1 + }) + ) + ) + await settle() + return results.map((r) => r.foregroundProcess) +} + +describe('cheap-tier ps scan volume at 8 idle agent panes over 60s', () => { + let platform: PropertyDescriptor | undefined + + beforeEach(() => { + execFileMock.mockReset() + runProcessMock.mockReset() + resetProcessTableSnapshotForTests() + resetCheapProcessTableSnapshotForTests() + forks.full = 0 + forks.cheap = 0 + panes.forEach((pane) => { + pane.agent = true + pane.agentStart = 'Thu Sep 3 16:02:05 2026' + }) + platform = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: 'darwin' }) + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(1_000_000) + installCountingPs() + }) + + afterEach(() => { + vi.useRealTimers() + if (platform) { + Object.defineProperty(process, 'platform', platform) + } + }) + + it('replaces ~all full captures with cheap ones once every pane holds an anchor', async () => { + const sessions = Array.from({ length: PANE_COUNT }, (_, pane) => createSession(pane)) + for (let tick = 0; tick < TICKS; tick += 1) { + vi.setSystemTime(1_000_000 + tick * IDLE_POLL_INTERVAL_MS) + const names = await runTick(sessions, true) + expect(names.every((name) => name === 'claude')).toBe(true) + } + // Baseline today: one full capture per tick (TTL-shared across the 8 panes) = TICKS. + // Now: the first tick establishes every anchor from one full capture; every later tick is + // one TTL-shared cheap capture. Published numbers, from this run: + // before: 30 full (~0.34-0.50s each on macOS, 1.15s Linux) + 0 cheap + // after: 1 full + 29 cheap (~0.03s each) + expect(forks.full).toBe(1) + expect(forks.cheap).toBe(TICKS - 1) + expect(forks.full + forks.cheap).toBe(TICKS) + }) + + it('keeps today’s cost when the caller does not opt in (old client / remote / restore)', async () => { + const sessions = Array.from({ length: PANE_COUNT }, (_, pane) => createSession(pane)) + for (let tick = 0; tick < TICKS; tick += 1) { + vi.setSystemTime(1_000_000 + tick * IDLE_POLL_INTERVAL_MS) + await runTick(sessions, false) + } + expect(forks.cheap).toBe(0) + expect(forks.full).toBe(TICKS) + }) + + it('completion detection is byte-for-byte unchanged: exit, idle, and restart resolve identically with and without the cheap tier', async () => { + const script = async (steadyState: boolean): Promise<(string | null)[][]> => { + resetProcessTableSnapshotForTests() + resetCheapProcessTableSnapshotForTests() + panes.forEach((pane) => { + pane.agent = true + pane.agentStart = 'Thu Sep 3 16:02:05 2026' + }) + const sessions = Array.from({ length: PANE_COUNT }, (_, pane) => createSession(pane)) + const series: (string | null)[][] = [] + for (let tick = 0; tick < TICKS; tick += 1) { + vi.setSystemTime(1_000_000 + tick * IDLE_POLL_INTERVAL_MS) + if (tick === 5) { + panes[2].agent = false // pane 2's agent exits + } + if (tick === 12) { + panes[2].agent = true // ...and is restarted with a new start time + panes[2].agentStart = 'Thu Sep 3 16:30:00 2026' + } + if (tick === 20) { + panes[6].agent = false + } + series.push(await runTick(sessions, steadyState)) + } + return series + } + const withCheapTier = await script(true) + const cheapForks = forks.cheap + forks.cheap = 0 + forks.full = 0 + const fullOnly = await script(false) + expect(withCheapTier).toEqual(fullOnly) + // And the exit was seen on the very tick it happened, in both modes. + expect(withCheapTier[4][2]).toBe('claude') + expect(withCheapTier[5][2]).not.toBe('claude') + expect(withCheapTier[12][2]).toBe('claude') + expect(withCheapTier[19][6]).toBe('claude') + expect(withCheapTier[20][6]).not.toBe('claude') + expect(cheapForks).toBeGreaterThan(0) + }) +}) diff --git a/src/main/daemon/terminal-host-process-inspection-cheap-tier.test.ts b/src/main/daemon/terminal-host-process-inspection-cheap-tier.test.ts new file mode 100644 index 00000000000..abf086e6e46 --- /dev/null +++ b/src/main/daemon/terminal-host-process-inspection-cheap-tier.test.ts @@ -0,0 +1,297 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +// Two seams because the two tiers spawn differently: the full evidence reader still forks +// through node:child_process, the cheap reader through Orca's runProcess entry point. +const { execFileMock, runProcessMock } = vi.hoisted(() => ({ + execFileMock: vi.fn(), + runProcessMock: vi.fn() +})) +vi.mock('node:child_process', () => ({ execFile: execFileMock })) +vi.mock('../../shared/child-process/run-process', () => ({ runProcess: runProcessMock })) + +import { resetCheapProcessTableSnapshotForTests } from '../../shared/cheap-process-table-snapshot-reader' +import { resetProcessTableSnapshotForTests } from '../../shared/process-table-snapshot-reader' +import { createPtyForegroundProcessTracker } from './pty-subprocess/foreground-process-tracker' +import { + inspectTerminalHostProcess, + type TerminalHostInspectionTier +} from './terminal-host-process-inspection' +import { getSteadyStateAnchor } from './terminal-host-steady-state-anchor' +import type { Session } from './session' + +const SHELL_PID = 4242 +const AGENT_PID = 4300 +const START_SHELL = 'Thu Sep 3 16:02:01 2026' +const START_AGENT = 'Thu Sep 3 16:02:05 2026' + +type Table = { agent: 'claude' | 'stopped' | 'gone' | 'replaced'; children?: number } + +/** One host table rendered in both column sets, so each fork answers by the args it asked for. */ +function renderTable(table: Table): { full: string; cheap: string } { + const shellTpgid = table.agent === 'claude' || table.agent === 'replaced' ? AGENT_PID : SHELL_PID + const shellStat = shellTpgid === SHELL_PID ? 'Ss+' : 'Ss' + const rows: { cheap: string; full: string }[] = [ + { + cheap: `${SHELL_PID} 1 ${SHELL_PID} ${shellTpgid} ${shellStat} ${START_SHELL}`, + full: `${SHELL_PID} 1 ${SHELL_PID} ${shellTpgid} ${shellStat} ttys004 ${START_SHELL} -zsh` + }, + { + cheap: `9000 1 9000 9000 Ss+ Thu Sep 3 12:00:00 2026`, + full: `9000 1 9000 9000 Ss+ ttys009 Thu Sep 3 12:00:00 2026 -zsh` + } + ] + if (table.agent !== 'gone') { + const stat = table.agent === 'stopped' ? 'T' : 'S+' + const start = table.agent === 'replaced' ? 'Thu Sep 3 16:30:00 2026' : START_AGENT + rows.push({ + cheap: `${AGENT_PID} ${SHELL_PID} ${AGENT_PID} ${shellTpgid} ${stat} ${start}`, + full: `${AGENT_PID} ${SHELL_PID} ${AGENT_PID} ${shellTpgid} ${stat} ttys004 ${start} node /usr/local/bin/claude` + }) + for (let i = 0; i < (table.children ?? 0); i += 1) { + const pid = AGENT_PID + 10 + i + rows.push({ + cheap: `${pid} ${AGENT_PID} ${AGENT_PID} ${shellTpgid} S+ Thu Sep 3 16:05:0${i} 2026`, + full: `${pid} ${AGENT_PID} ${AGENT_PID} ${shellTpgid} S+ ttys004 Thu Sep 3 16:05:0${i} 2026 rg --files` + }) + } + } + return { + full: `${rows.map((r) => r.full).join('\n')}\n`, + cheap: `${rows.map((r) => r.cheap).join('\n')}\n` + } +} + +const forks = { full: 0, cheap: 0 } +let table: Table = { agent: 'claude' } + +function installPs(): void { + execFileMock.mockImplementation((_cmd: string, args: string[], _opts: unknown, cb: unknown) => { + expect(args[1]).toContain('command=') + forks.full += 1 + ;(cb as (err: unknown, r: { stdout: string; stderr: string }) => void)(null, { + stdout: renderTable(table).full, + stderr: '' + }) + }) + runProcessMock.mockImplementation(async (spec: { args: readonly string[] }) => { + expect(spec.args[1]).not.toContain('command=') + forks.cheap += 1 + return { + code: 0, + signal: null, + stdout: renderTable(table).cheap, + stderr: '', + timedOut: false + } + }) +} + +function createSession(processName: () => string): Session { + let dead = false + const tracker = createPtyForegroundProcessTracker({ + process: { + pid: SHELL_PID, + get process() { + return processName() + } + } as never, + shellPath: '/bin/zsh', + sessionId: 'wt-1:pane-1', + startupAgentRecognition: null, + isDead: () => dead + }) + return { + pid: SHELL_PID, + incarnationId: 'inc-1', + get isAlive() { + return !dead + }, + getForegroundProcess: (options?: { rawFallback?: boolean }) => + tracker.getForegroundProcess(options), + markDead: () => { + dead = true + tracker.markDead() + } + } as unknown as Session & { markDead(): void } +} + +async function inspect( + session: Session, + options: { steadyState?: boolean; expectedIncarnationId?: string } = {} +): Promise<{ + tier: TerminalHostInspectionTier + result: Awaited> +}> { + let tier: TerminalHostInspectionTier = 'full' + const result = await inspectTerminalHostProcess({ + sessionId: 'wt-1:pane-1', + session, + ...options, + authorityGeneration: 'gen-1', + nextObservationEpoch: () => 1, + onTier: (t) => { + tier = t + } + }) + return { tier, result } +} + +async function settle(): Promise { + // The tracker's recognizing refresh runs off the same TTL-shared capture; let it land. + for (let i = 0; i < 8; i += 1) { + await Promise.resolve() + } +} + +async function advance(ms: number): Promise { + vi.setSystemTime(Date.now() + ms) +} + +describe('daemon cheap-tier process inspection', () => { + let platform: PropertyDescriptor | undefined + + beforeEach(() => { + execFileMock.mockReset() + runProcessMock.mockReset() + resetProcessTableSnapshotForTests() + resetCheapProcessTableSnapshotForTests() + forks.full = 0 + forks.cheap = 0 + table = { agent: 'claude' } + platform = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: 'darwin' }) + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(1_000_000) + installPs() + }) + + afterEach(() => { + vi.useRealTimers() + if (platform) { + Object.defineProperty(process, 'platform', platform) + } + }) + + /** Bring a session to a recognized anchor the way production does: one full cadence tick. */ + async function anchoredSession(): Promise { + const session = createSession(() => 'node') + const first = await inspect(session, { steadyState: true }) + await settle() + expect(first.tier).toBe('full') + expect(first.result.foregroundProcess).toBe('claude') + expect(getSteadyStateAnchor(session)?.agentName).toBe('claude') + return session + } + + it('a pane with NO recognized anchor never takes the cheap path, even when asked', async () => { + table = { agent: 'gone' } + const session = createSession(() => 'zsh') + for (let tick = 0; tick < 5; tick += 1) { + await advance(2_000) + const { tier, result } = await inspect(session, { steadyState: true }) + expect(tier).toBe('full') + expect(result.foregroundProcessEvidence).toBeDefined() + } + expect(forks.cheap).toBe(0) + expect(forks.full).toBe(5) + }) + + it('serves an unchanged anchored pane from the cheap tier and OMITS evidence rather than faking it', async () => { + const session = await anchoredSession() + const fullBefore = forks.full + for (let tick = 0; tick < 4; tick += 1) { + await advance(2_000) + const { tier, result } = await inspect(session, { steadyState: true }) + expect(tier).toBe('cheap') + expect(result.foregroundProcess).toBe('claude') + expect(result.hasChildProcesses).toBe(true) + expect(result).not.toHaveProperty('foregroundProcessEvidence') + } + expect(forks.cheap).toBe(4) + expect(forks.full).toBe(fullBefore) + }) + + it('a request without steadyState (old client, remote client, restore path) always gets the full capture with evidence', async () => { + const session = await anchoredSession() + await advance(2_000) + const { tier, result } = await inspect(session) + expect(tier).toBe('full') + expect(result.foregroundProcessEvidence).toMatchObject({ + verdict: 'live', + processName: 'claude' + }) + expect(forks.cheap).toBe(0) + }) + + it('escalates to the full capture the moment the agent exits, and reports the exit', async () => { + const session = await anchoredSession() + await advance(2_000) + expect((await inspect(session, { steadyState: true })).tier).toBe('cheap') + table = { agent: 'gone' } + await advance(2_000) + const { tier, result } = await inspect(session, { steadyState: true }) + expect(tier).toBe('full') + expect(result.foregroundProcessEvidence).toMatchObject({ verdict: 'live', processName: null }) + }) + + it.each<[string, Table]>([ + ['Ctrl-Z stops the agent', { agent: 'stopped' }], + ['exit-and-replace reuses the pid', { agent: 'replaced' }], + ['a child spawns under the agent', { agent: 'claude', children: 1 }] + ])('escalates when %s', async (_name, next) => { + const session = await anchoredSession() + await advance(2_000) + expect((await inspect(session, { steadyState: true })).tier).toBe('cheap') + table = next + await advance(2_000) + expect((await inspect(session, { steadyState: true })).tier).toBe('full') + }) + + it('escalates when node-pty reports a different foreground name, without waiting on ps', async () => { + let name = 'node' + const session = createSession(() => name) + await inspect(session, { steadyState: true }) + await settle() + await advance(2_000) + expect((await inspect(session, { steadyState: true })).tier).toBe('cheap') + name = 'zsh' + await advance(2_000) + const cheapBefore = forks.cheap + expect((await inspect(session, { steadyState: true })).tier).toBe('full') + expect(forks.cheap).toBe(cheapBefore) + }) + + it('falls through to the full capture when the cheap fork fails, and after an incarnation mismatch', async () => { + const session = await anchoredSession() + await advance(2_000) + runProcessMock.mockRejectedValueOnce(new Error('ps died')) + expect((await inspect(session, { steadyState: true })).tier).toBe('full') + await advance(2_000) + const mismatched = await inspect(session, { steadyState: true, expectedIncarnationId: 'other' }) + expect(mismatched.tier).toBe('full') + expect(mismatched.result.foregroundProcessEvidence).toMatchObject({ + reason: 'incarnation_mismatch' + }) + }) + + it('a dead session is never served from its anchor', async () => { + const session = (await anchoredSession()) as Session & { markDead(): void } + session.markDead() + await expect(inspect(session, { steadyState: true })).rejects.toThrow('not found') + expect(forks.cheap).toBe(0) + }) + + it('an anchor is dropped when a full capture stops naming a recognized agent', async () => { + const session = await anchoredSession() + table = { agent: 'gone' } + await advance(2_000) + await inspect(session, { steadyState: true }) + expect(getSteadyStateAnchor(session)).toBeNull() + // Back with a new agent, but the pane must re-anchor via a FULL capture first. + table = { agent: 'claude' } + await advance(2_000) + const cheapBefore = forks.cheap + expect((await inspect(session, { steadyState: true })).tier).toBe('full') + expect(forks.cheap).toBe(cheapBefore) + }) +}) diff --git a/src/main/daemon/terminal-host-process-inspection.ts b/src/main/daemon/terminal-host-process-inspection.ts index 181ca7266d7..6810687631e 100644 --- a/src/main/daemon/terminal-host-process-inspection.ts +++ b/src/main/daemon/terminal-host-process-inspection.ts @@ -1,8 +1,15 @@ import { isShellProcess } from '../../shared/agent-detection' import type { RemoteForegroundEvidence } from '../../shared/foreground-process-evidence' +import { getCheapProcessTableSnapshot } from '../../shared/cheap-process-table-snapshot-reader' import { getStrictProcessTableSnapshotWithAge } from '../../shared/process-table-snapshot-reader' import { resolveRemoteForegroundEvidence } from '../providers/agent-foreground-process' +import { buildPaneProcessFingerprint } from '../providers/posix-pane-foreground-fingerprint' import type { Session } from './session' +import { + clearSteadyStateAnchor, + getSteadyStateAnchor, + rememberSteadyStateAnchor +} from './terminal-host-steady-state-anchor' import { SessionNotFoundError } from './types' export type TerminalHostProcessInspection = { @@ -13,13 +20,23 @@ export type TerminalHostProcessInspection = { type RetiredIncarnation = { incarnationId: string; code: number; expiresAt: number } +/** + * Tick tiers for a POSIX pane. `cheap` forks `ps` without `tty=`/`command=` (11-38x cheaper) + * and answers from the anchored identity when the pane fingerprint is unchanged; anything it + * cannot prove escalates to `full`, today's evidence capture. + */ +export type TerminalHostInspectionTier = 'full' | 'cheap' + export async function inspectTerminalHostProcess(args: { sessionId: string session: Session | null expectedIncarnationId?: string + /** The caller is a self-correcting poll that only reads the process name, never evidence. */ + steadyState?: boolean retiredIncarnation?: RetiredIncarnation authorityGeneration: string nextObservationEpoch: () => number + onTier?: (tier: TerminalHostInspectionTier) => void }): Promise { const { sessionId, session, expectedIncarnationId, retiredIncarnation } = args if (!session || !session.isAlive) { @@ -45,9 +62,22 @@ export async function inspectTerminalHostProcess(args: { throw new SessionNotFoundError(sessionId) } + const incarnationMatches = + !expectedIncarnationId || expectedIncarnationId === session.incarnationId + if (args.steadyState === true && incarnationMatches) { + const anchored = await readAnchoredForeground(session) + if (anchored !== null) { + args.onTier?.('cheap') + // No evidence member on purpose: a tty-less capture cannot fence anything, and a + // fabricated fence would be read by remote/restore consumers as an observation. + return { foregroundProcess: anchored, hasChildProcesses: true } + } + } + args.onTier?.('full') + const foregroundProcess = session.getForegroundProcess() let evidence: RemoteForegroundEvidence - if (expectedIncarnationId && expectedIncarnationId !== session.incarnationId) { + if (!incarnationMatches) { evidence = unverifiableEvidence(args, session, 'incarnation_mismatch') } else { try { @@ -64,8 +94,10 @@ export async function inspectTerminalHostProcess(args: { }, snapshot.rows ) + await rememberSteadyStateAnchor(session, evidence, snapshot.rows) } catch { evidence = unverifiableEvidence(args, session, 'process_table_unreadable') + clearSteadyStateAnchor(session) } } return { @@ -75,6 +107,32 @@ export async function inspectTerminalHostProcess(args: { } } +/** + * Cheap tier, gated on an anchor the last full capture established. Start discovery therefore + * keeps today's exact behaviour: a pane with no anchor never gets here. A recognized agent's + * exit is a pid vanishing from the subtree, which the fingerprint always sees, so completion + * detection is unaffected. Any mismatch, unreadable capture, changed node-pty name, or non-POSIX + * host answers null -> full tier. + */ +async function readAnchoredForeground(session: Session): Promise { + const anchor = getSteadyStateAnchor(session) + if (process.platform === 'win32' || !anchor) { + return null + } + if (session.getForegroundProcess({ rawFallback: true }) !== anchor.rawFallback) { + return null + } + try { + const observed = await buildPaneProcessFingerprint( + await getCheapProcessTableSnapshot(), + session.pid + ) + return observed !== null && observed === anchor.fingerprint ? anchor.agentName : null + } catch { + return null + } +} + function unverifiableEvidence( args: { sessionId: string diff --git a/src/main/daemon/terminal-host-steady-state-anchor.ts b/src/main/daemon/terminal-host-steady-state-anchor.ts new file mode 100644 index 00000000000..562224f856f --- /dev/null +++ b/src/main/daemon/terminal-host-steady-state-anchor.ts @@ -0,0 +1,58 @@ +import { recognizeAgentProcess } from '../../shared/agent-process-recognition' +import type { RemoteForegroundEvidence } from '../../shared/foreground-process-evidence' +import { buildPaneProcessFingerprint } from '../providers/posix-pane-foreground-fingerprint' +import type { Session } from './session' + +/** + * What the last FULL capture proved about a pane: a recognized agent name, the pane subtree + * fingerprint at that moment, and node-pty's raw foreground name at that moment. A later cheap + * tick may re-serve `agentName` only while both of the latter still match. + */ +export type SteadyStateAnchor = { + agentName: string + fingerprint: string + rawFallback: string | null +} + +// Weakly keyed: an anchor dies with its Session, and a recycled pid under a new Session can +// never inherit one. Retired sessions fail `isAlive` before any read gets here regardless. +const anchors = new WeakMap() + +export function getSteadyStateAnchor(session: Session): SteadyStateAnchor | null { + return anchors.get(session) ?? null +} + +export function clearSteadyStateAnchor(session: Session): void { + anchors.delete(session) +} + +/** + * Record (or drop) the anchor after a full capture. Only a `live` verdict naming a recognized + * agent establishes one: the cheap tier is licensed by proven identity, never by a fallback name + * or an unverifiable read, so a pane without one always pays for the full capture. + */ +export async function rememberSteadyStateAnchor( + session: Session, + evidence: RemoteForegroundEvidence, + rows: Parameters[0] +): Promise { + if (evidence.verdict !== 'live' || !recognizeAgentProcess(evidence.processName)) { + anchors.delete(session) + return + } + let fingerprint: string | null + try { + fingerprint = await buildPaneProcessFingerprint(rows, session.pid) + } catch { + fingerprint = null + } + if (fingerprint === null || evidence.processName === null) { + anchors.delete(session) + return + } + anchors.set(session, { + agentName: evidence.processName, + fingerprint, + rawFallback: session.getForegroundProcess({ rawFallback: true }) + }) +} diff --git a/src/main/daemon/terminal-host.ts b/src/main/daemon/terminal-host.ts index 18f7b82a90f..95bedd1a7fd 100644 --- a/src/main/daemon/terminal-host.ts +++ b/src/main/daemon/terminal-host.ts @@ -233,7 +233,7 @@ export class TerminalHost { inspectProcess( sessionId: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; steadyState?: boolean } ): Promise { pruneRetiredPtyIncarnations(this.retiredIncarnations) const session = this.sessions.get(sessionId) @@ -253,6 +253,7 @@ export class TerminalHost { ...(options?.expectedIncarnationId ? { expectedIncarnationId: options.expectedIncarnationId } : {}), + ...(options?.steadyState === true ? { steadyState: true } : {}), retiredIncarnation: this.retiredIncarnations.get(sessionId), authorityGeneration: this.authorityGeneration, nextObservationEpoch: () => ++this.observationEpoch diff --git a/src/main/git/command-runner/git-exec-file.ts b/src/main/git/command-runner/git-exec-file.ts index dbd861d4897..acfeb9f8a3e 100644 --- a/src/main/git/command-runner/git-exec-file.ts +++ b/src/main/git/command-runner/git-exec-file.ts @@ -10,6 +10,7 @@ import { prepareWslLinkedWorktreeGitRouting } from '../wsl-linked-worktree-git-routing' import { resolveCommand, type ResolvedCommand } from './wsl-command-resolution' +import { annotateWslHostFailure } from './wsl-host-failure' import type { GitAdmissionTier, GitExecOptions } from './git-exec-options' import { execFileCapture, execFileCaptureToTermination } from './exec-file-capture' import { @@ -96,7 +97,7 @@ async function gitExecFileAsyncUnlocked( ? {} : { createTimeoutError: () => new GitCommandTimeoutError(timeoutMs) }) } - return options.terminationBarrier + const captured = options.terminationBarrier ? execFileCaptureToTermination( command.binary, command.args, @@ -104,6 +105,10 @@ async function gitExecFileAsyncUnlocked( command.termination ) : execFileCapture(command.binary, command.args, captureOptions) + // Why: a dead WSL distro fails with an empty stderr, so the span would carry no cause at all. + return captured.catch((error: unknown) => { + throw annotateWslHostFailure(error, command) + }) } const runCapturedCommand = async (): Promise<{ stdout: string; stderr: string }> => { let result: { stdout: string | Buffer; stderr: string | Buffer } diff --git a/src/main/git/command-runner/git-process-env.ts b/src/main/git/command-runner/git-process-env.ts index 6154364e5ab..9064f5ed900 100644 --- a/src/main/git/command-runner/git-process-env.ts +++ b/src/main/git/command-runner/git-process-env.ts @@ -63,6 +63,11 @@ export function nonInteractiveGitEnv( platform: NodeJS.Platform = process.platform ): NodeJS.ProcessEnv { const next = promptGuardGitEnv(env, platform) + if (platform === 'win32') { + // Why: without it wsl.exe writes its OWN failures ("no distribution with the supplied name") as + // UTF-16LE (#9010), which is how a dead distro reached telemetry as an error with no text. + next.WSL_UTF8 = '1' + } if (!next.GIT_SSH_COMMAND) { next.GIT_SSH_COMMAND = 'ssh -o BatchMode=yes' if (platform === 'win32') { diff --git a/src/main/git/command-runner/spawned-command-tree-kill.ts b/src/main/git/command-runner/spawned-command-tree-kill.ts index c2fefcba1bc..324e04f1db3 100644 --- a/src/main/git/command-runner/spawned-command-tree-kill.ts +++ b/src/main/git/command-runner/spawned-command-tree-kill.ts @@ -1,4 +1,5 @@ import { spawn, type ChildProcess } from 'node:child_process' +import { admitSelfInitiatedTreeKill } from '../../own-chromium-tree-kill-guard' const WINDOWS_TREE_KILL_WAIT_MS = 2_000 @@ -8,6 +9,14 @@ export function killSpawnedCommandTree(child: ChildProcess): Promise { child.kill() return Promise.resolve() } + if ( + !admitSelfInitiatedTreeKill({ pid, site: 'git-command-tree-kill', scope: 'win-taskkill-tree' }) + ) { + // Refusal blocks the pid-addressed tree walk, never the termination: the + // handle-addressed root kill cannot reach a recycled pid. + child.kill() + return Promise.resolve() + } return new Promise((resolve) => { let killer: ChildProcess try { diff --git a/src/main/git/command-runner/wsl-host-failure.test.ts b/src/main/git/command-runner/wsl-host-failure.test.ts new file mode 100644 index 00000000000..4b495817f6f --- /dev/null +++ b/src/main/git/command-runner/wsl-host-failure.test.ts @@ -0,0 +1,143 @@ +import { EventEmitter } from 'node:events' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { execFileMock } = vi.hoisted(() => ({ execFileMock: vi.fn() })) + +vi.mock('node:child_process', () => ({ + execFile: execFileMock, + execFileSync: vi.fn(), + spawn: vi.fn() +})) +vi.mock('../../observability/instrumentation', () => ({ + withGitSpan: (_attributes: unknown, run: (span: unknown) => unknown) => + run({ setAttribute: () => {} }) +})) +vi.mock('../../diagnostics/main-thread-churn-probe', () => ({ recordSubprocessSpawn: vi.fn() })) + +import { gitExecFileAsync } from '../runner' +import { _resetGitAdmissionForTests } from './git-subprocess-admission' +import { nonInteractiveGitEnv } from './git-process-env' +import { annotateWslHostFailure, readWslHostFailureDiagnostic } from './wsl-host-failure' +import type { ResolvedCommand } from './wsl-command-resolution' + +const WSL_COMMAND: ResolvedCommand = { + binary: 'wsl.exe', + args: ['-d', 'kali-linux', '--exec', 'sh', '-lc', 'git worktree list'], + cwd: 'C:\\Users\\paulius', + wsl: { distro: 'kali-linux', linuxPath: '/home/paulius/bugbounty' }, + wslMode: 'login-shell' +} + +const WSL_DIAGNOSTIC = + 'There is no distribution with the supplied name.\r\nError code: Wsl/Service/WSL_E_DISTRO_NOT_FOUND\r\n' + +/** wsl.exe without WSL_UTF8 writes its own diagnostic as UTF-16LE, which reaches Node as NUL-riddled text. */ +function asUtf16Mojibake(text: string): string { + return [...text].map((character) => `${character}\u0000`).join('') +} + +function hostFailure(stdout: string): Error { + return Object.assign(new Error('Command failed: wsl.exe -d kali-linux --exec sh -lc ...'), { + code: 4294967295, + stdout, + stderr: '' + }) +} + +async function withPlatform(platform: NodeJS.Platform, run: () => Promise): Promise { + const original = process.platform + Object.defineProperty(process, 'platform', { configurable: true, value: platform }) + try { + return await run() + } finally { + Object.defineProperty(process, 'platform', { configurable: true, value: original }) + } +} + +describe('wsl.exe host failure classification', () => { + it('reads the diagnostic wsl.exe left on stdout, including UTF-16 output', () => { + expect(readWslHostFailureDiagnostic(hostFailure(WSL_DIAGNOSTIC), WSL_COMMAND)).toContain( + 'Wsl/Service/WSL_E_DISTRO_NOT_FOUND' + ) + expect( + readWslHostFailureDiagnostic(hostFailure(asUtf16Mojibake(WSL_DIAGNOSTIC)), WSL_COMMAND) + ).toContain('Wsl/Service/WSL_E_DISTRO_NOT_FOUND') + }) + + it('leaves a guest failure and a non-WSL command alone', () => { + const guestFailure = Object.assign(new Error('Command failed'), { + code: 1, + stdout: '', + stderr: 'fatal: not a git repository\n' + }) + expect(readWslHostFailureDiagnostic(guestFailure, WSL_COMMAND)).toBeNull() + // Same exit code, but wsl.exe was never involved. + expect( + readWslHostFailureDiagnostic(hostFailure(WSL_DIAGNOSTIC), { + binary: 'git', + args: ['status'], + cwd: '/repo', + wsl: null, + wslMode: null + }) + ).toBeNull() + }) + + it('moves the diagnostic into the message the span records', () => { + const error = annotateWslHostFailure(hostFailure(WSL_DIAGNOSTIC), WSL_COMMAND) as Error & { + wslHostFailure?: boolean + wslDistro?: string + code?: number + } + expect(error.message).toContain('Wsl/Service/WSL_E_DISTRO_NOT_FOUND') + expect(error.message).toContain('kali-linux') + expect(error.wslHostFailure).toBe(true) + expect(error.wslDistro).toBe('kali-linux') + // The original failure detail must survive for callers that classify on it. + expect(error.code).toBe(4294967295) + expect(error.message).toContain('Command failed: wsl.exe') + }) +}) + +describe('WSL-routed git subprocess', () => { + beforeEach(() => { + execFileMock.mockReset() + }) + + afterEach(() => { + _resetGitAdmissionForTests() + }) + + it('sets WSL_UTF8 so wsl.exe explains itself in UTF-8', () => { + expect(nonInteractiveGitEnv({}, 'win32').WSL_UTF8).toBe('1') + expect(nonInteractiveGitEnv({}, 'darwin').WSL_UTF8).toBeUndefined() + }) + + it('reports a dead distro instead of an empty git error', async () => { + execFileMock.mockImplementation((_command, _args, _options, callback) => { + const child = new EventEmitter() as EventEmitter & { pid: number; kill: () => void } + child.pid = 4321 + child.kill = () => {} + queueMicrotask(() => + callback?.( + hostFailure(asUtf16Mojibake(WSL_DIAGNOSTIC)), + asUtf16Mojibake(WSL_DIAGNOSTIC), + '' + ) + ) + return child + }) + + const failure = await withPlatform('win32', () => + gitExecFileAsync(['worktree', 'list', '--porcelain', '-z'], { + cwd: '\\\\wsl.localhost\\kali-linux\\home\\paulius\\bugbounty' + }).then( + () => null, + (error: unknown) => error as Error + ) + ) + + expect(failure?.message).toContain('Wsl/Service/WSL_E_DISTRO_NOT_FOUND') + expect(execFileMock.mock.calls.at(-1)?.[2]?.env?.WSL_UTF8).toBe('1') + }) +}) diff --git a/src/main/git/command-runner/wsl-host-failure.ts b/src/main/git/command-runner/wsl-host-failure.ts new file mode 100644 index 00000000000..74f98509d0b --- /dev/null +++ b/src/main/git/command-runner/wsl-host-failure.ts @@ -0,0 +1,55 @@ +import type { ResolvedCommand } from './wsl-command-resolution' + +/** wsl.exe's own launch-failure exit, distinct from any status the guest process can return. */ +export const WSL_HOST_FAILURE_EXIT_CODE = 0xffffffff + +function outputText(value: unknown): string { + if (typeof value === 'string') { + return value + } + return Buffer.isBuffer(value) ? value.toString('utf8') : '' +} + +/** + * The message wsl.exe prints when it — not the guest — failed: a distro that was renamed or + * removed, or a VM that would not start. + * + * Why this needs decoding at all: wsl.exe exits 0xFFFFFFFF, leaves stderr EMPTY, and writes + * `Error code: Wsl/Service/WSL_E_*` to stdout, so every caller that reports stderr reports nothing. + * NULs are stripped because a wsl.exe that ignores WSL_UTF8 writes that line as UTF-16LE (#9010). + */ +export function readWslHostFailureDiagnostic( + error: unknown, + command: ResolvedCommand +): string | null { + if (!command.wsl || !error || typeof error !== 'object') { + return null + } + const { code, status, stdout, stderr } = error as { + code?: unknown + status?: unknown + stdout?: unknown + stderr?: unknown + } + const exitCode = typeof code === 'number' ? code : typeof status === 'number' ? status : null + // A guest failure always explains itself on stderr; an empty one plus this exit is the host. + if (exitCode !== WSL_HOST_FAILURE_EXIT_CODE || outputText(stderr).trim().length > 0) { + return null + } + const diagnostic = outputText(stdout).replaceAll('\u0000', '').trim() + return diagnostic.length > 0 ? diagnostic : 'wsl.exe reported no diagnostic.' +} + +/** + * Move a wsl.exe host failure into the error's message, which is what `git.exec` spans record ahead + * of the stack. Left untouched when the failure came from git itself. + */ +export function annotateWslHostFailure(error: unknown, command: ResolvedCommand): unknown { + const diagnostic = readWslHostFailureDiagnostic(error, command) + if (diagnostic === null || !(error instanceof Error) || !command.wsl) { + return error + } + const distro = command.wsl.distro + error.message = `wsl.exe host failure (distro "${distro}"): ${diagnostic}\n${error.message}` + return Object.assign(error, { wslHostFailure: true, wslDistro: distro }) +} diff --git a/src/main/git/worktree-base-divergence-real-git.test.ts b/src/main/git/worktree-base-divergence-real-git.test.ts index 377b63b48f9..4aa1734c792 100644 --- a/src/main/git/worktree-base-divergence-real-git.test.ts +++ b/src/main/git/worktree-base-divergence-real-git.test.ts @@ -3,6 +3,7 @@ import { mkdtemp, rm, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, describe, expect, it } from 'vitest' +import { GIT_FETCH_SKIP_AUTO_MAINTENANCE_CONFIG_ARGS } from '../../shared/git-fetch-auto-maintenance' import { measureRetargetDivergence, RETARGET_MAX_COMMIT_DIVERGENCE @@ -10,8 +11,13 @@ import { const tempRoots: string[] = [] +// Why the maintenance suppression: `git commit` detaches `git maintenance run --auto`, and its +// commit-graph task arms once a fixture crosses 100 new commits — which the cap-sized histories +// below always do. That detached process keeps writing `.git/objects/info/commit-graphs` after the +// synchronous exec has returned, so it re-creates entries under a `.git/objects` the temp-root +// teardown is midway through deleting, and the recursive remove dies with ENOTEMPTY. function git(cwd: string, args: string[]): string { - return execFileSync('git', args, { + return execFileSync('git', [...GIT_FETCH_SKIP_AUTO_MAINTENANCE_CONFIG_ARGS, ...args], { cwd, encoding: 'utf8', stdio: ['pipe', 'pipe', 'pipe'] diff --git a/src/main/git/worktree-listing.ts b/src/main/git/worktree-listing.ts index a60819181d4..f027bac4bc1 100644 --- a/src/main/git/worktree-listing.ts +++ b/src/main/git/worktree-listing.ts @@ -35,17 +35,7 @@ export async function listWorktreeGraph( ? worktrees : worktrees.filter((worktree) => !isWorktreeCreatePreparation(worktree)) } catch (err) { - if (getErrorCode(err) === 'ENOENT') { - try { - await stat(repoPath) - } catch (statErr) { - if (getErrorCode(statErr) === 'ENOENT') { - console.warn(`[git/worktree] repo path missing; skipping worktree list: ${repoPath}`) - return [] - } - } - } - if (isNotGitRepositoryError(err)) { + if (await isTrueEmptyWorktreeListing(repoPath, err)) { return [] } console.warn(`[git/worktree] listWorktreeGraph failed for ${repoPath}:`, err) @@ -64,17 +54,7 @@ export async function listWorktreesUnshared( : worktrees.filter((worktree) => !isWorktreeCreatePreparation(worktree)) return annotateSparseCheckoutStatus(repoPath, visibleWorktrees, options) } catch (err) { - if (getErrorCode(err) === 'ENOENT') { - try { - await stat(repoPath) - } catch (statErr) { - if (getErrorCode(statErr) === 'ENOENT') { - console.warn(`[git/worktree] repo path missing; skipping worktree list: ${repoPath}`) - return [] - } - } - } - if (isNotGitRepositoryError(err)) { + if (await isTrueEmptyWorktreeListing(repoPath, err)) { return [] } // Why: don't swallow git-compat/repo-state failures — else they resurface as opaque "created but not found in listing" errors. @@ -83,6 +63,24 @@ export async function listWorktreesUnshared( } } +/** + * The two failures where an empty listing is the repo's true answer, not a broken scan: the repo + * path is gone, or it is not a Git repo. Every other failure means the scan could not read Git. + */ +async function isTrueEmptyWorktreeListing(repoPath: string, err: unknown): Promise { + if (getErrorCode(err) === 'ENOENT') { + try { + await stat(repoPath) + } catch (statErr) { + if (getErrorCode(statErr) === 'ENOENT') { + console.warn(`[git/worktree] repo path missing; skipping worktree list: ${repoPath}`) + return true + } + } + } + return isNotGitRepositoryError(err) +} + export async function listWorktreesStrict( repoPath: string, options: GitWorktreeExecOptions = {} @@ -97,6 +95,28 @@ export async function listWorktreesStrict( return annotateSparseCheckoutStatus(repoPath, visibleWorktrees, options) } +/** + * Strict except for the two true empties above. + * + * Why: a Git or host failure (dead WSL distro, hung mount) softened to `[]` reaches the detected + * listing as an *authoritative* empty scan, which then permanently prunes the repo's worktrees and + * the agent tabs attached to them. Rejecting keeps that listing non-authoritative, while a deleted + * repo still reports empty so real removals prune. + */ +export async function listWorktreesStrictAllowingTrueEmpty( + repoPath: string, + options: GitWorktreeExecOptions = {} +): Promise { + try { + return await listWorktreesStrict(repoPath, options) + } catch (err) { + if (await isTrueEmptyWorktreeListing(repoPath, err)) { + return [] + } + throw err + } +} + export async function annotateSparseCheckoutStatus( repoPath: string, worktrees: GitWorktreeInfo[], diff --git a/src/main/git/worktree-scan-cache.ts b/src/main/git/worktree-scan-cache.ts index 59a7691fe7f..a35f0a4f275 100644 --- a/src/main/git/worktree-scan-cache.ts +++ b/src/main/git/worktree-scan-cache.ts @@ -2,7 +2,8 @@ import type { GitWorktreeInfo } from '../../shared/worktree/types' import { annotateSparseCheckoutStatus, listWorktreeGraph as listWorktreeGraphUnshared, - listWorktreesStrict as listWorktreesStrictUnshared + listWorktreesStrict as listWorktreesStrictUnshared, + listWorktreesStrictAllowingTrueEmpty as listWorktreesStrictAllowingTrueEmptyUnshared } from './worktree-listing' import type { GitWorktreeExecOptions } from './worktree-operation-options' import { WORKTREE_LIST_TIMEOUT_MS } from './worktree-operation-options' @@ -10,7 +11,7 @@ import { WORKTREE_LIST_TIMEOUT_MS } from './worktree-operation-options' // Why: share concurrent `git worktree list` scans, which are expensive on Windows. const inFlightWorktreeScans = new Map>() -type WorktreeScanKind = 'graph' | 'lenient' | 'strict' +type WorktreeScanKind = 'graph' | 'lenient' | 'strict' | 'strict-true-empty' // Why: mutation generations prevent listings from joining stale scans. const worktreeScanGenerations = new Map() @@ -139,3 +140,20 @@ export function listWorktreesSharedStrict( ): Promise { return shareWorktreeScan(repoPath, options, 'strict', listWorktreesStrictUnshared) } + +/** + * The detected scan's discipline: reject a Git/host failure so it cannot publish as an + * authoritative empty listing, but still answer `[]` for a repo that is gone or not a repo. + * Its own kind because neither a strict nor a lenient joiner may inherit that middle contract. + */ +export function listWorktreesSharedStrictAllowingTrueEmpty( + repoPath: string, + options: GitWorktreeExecOptions = {} +): Promise { + return shareWorktreeScan( + repoPath, + options, + 'strict-true-empty', + listWorktreesStrictAllowingTrueEmptyUnshared + ) +} diff --git a/src/main/git/worktree.ts b/src/main/git/worktree.ts index 18e92e2873a..024485b6981 100644 --- a/src/main/git/worktree.ts +++ b/src/main/git/worktree.ts @@ -32,7 +32,8 @@ export { _resetWorktreeScanCacheForTests, listWorktreeGraph, listWorktrees, - listWorktreesSharedStrict + listWorktreesSharedStrict, + listWorktreesSharedStrictAllowingTrueEmpty } from './worktree-scan-cache' export { bumpWorktreeScanGeneration as notifyPreparedWorktreeMutation } from './worktree-scan-cache' export { addSparseWorktree } from './worktree-sparse-add' diff --git a/src/main/ipc/notebook.ts b/src/main/ipc/notebook.ts index 9255ab7383d..d90ab3616de 100644 --- a/src/main/ipc/notebook.ts +++ b/src/main/ipc/notebook.ts @@ -4,6 +4,7 @@ import { dirname } from 'node:path' import { ipcMain } from 'electron' import type { Store } from '../persistence' import { resolveAuthorizedPath } from './filesystem-auth' +import { admitSelfInitiatedTreeKill } from '../own-chromium-tree-kill-guard' export type NotebookRunResult = { stdout: string @@ -53,7 +54,8 @@ function appendBounded(capture: BoundedCapture, chunk: Buffer): void { capture.truncated = true } -function terminateNotebookProcessTree( +/** Exported for the refusal-fallback test; the timeout path is otherwise unreachable. */ +export function terminateNotebookProcessTree( child: ChildProcessWithoutNullStreams ): ReturnType | null { if (!child.pid) { @@ -62,6 +64,18 @@ function terminateNotebookProcessTree( } if (process.platform === 'win32') { + if ( + !admitSelfInitiatedTreeKill({ + pid: child.pid, + site: 'notebook-cell-timeout', + scope: 'win-taskkill-tree' + }) + ) { + // Refusal blocks the tree walk, not the termination: killing the root by + // handle cannot reach a recycled pid, and a timed-out cell must still stop. + child.kill() + return null + } try { // Why: a timed-out cell can spawn descendants. taskkill /T is the // Windows equivalent of terminating the whole process group. diff --git a/src/main/ipc/pty/delivery/attached-pty-size.test.ts b/src/main/ipc/pty/delivery/attached-pty-size.test.ts new file mode 100644 index 00000000000..90ff37c744d --- /dev/null +++ b/src/main/ipc/pty/delivery/attached-pty-size.test.ts @@ -0,0 +1,166 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + commitAttachedPtySize, + resolveCommittedPtySize, + shouldSeedPreAttachPtySize +} from './attached-pty-size' +import { ptySizes } from './visibility-state' + +const REQUESTED = { cols: 80, rows: 24 } +const CACHED = { cols: 180, rows: 50 } +const LIVE = { cols: 211, rows: 57 } + +describe('shouldSeedPreAttachPtySize', () => { + it('seeds a fresh session id even when the pane never measured itself', () => { + expect( + shouldSeedPreAttachPtySize({ + isFreshSessionId: true, + hasCachedSize: true, + requestIsUnmeasured: true + }) + ).toBe(true) + }) + + it('never overwrites a size main already holds for the session', () => { + expect( + shouldSeedPreAttachPtySize({ + isFreshSessionId: false, + hasCachedSize: true, + requestIsUnmeasured: false + }) + ).toBe(false) + }) + + it('refuses an unmeasured request on an attach even with nothing cached', () => { + expect( + shouldSeedPreAttachPtySize({ + isFreshSessionId: false, + hasCachedSize: false, + requestIsUnmeasured: true + }) + ).toBe(false) + }) + + it('seeds a measured attach request when main holds nothing better', () => { + expect( + shouldSeedPreAttachPtySize({ + isFreshSessionId: false, + hasCachedSize: false, + requestIsUnmeasured: false + }) + ).toBe(true) + }) +}) + +describe('resolveCommittedPtySize', () => { + it('records the requested grid for a fresh spawn, ignoring any stale cache', () => { + expect( + resolveCommittedPtySize({ + result: {}, + requested: REQUESTED, + cachedBeforeAttach: CACHED + }) + ).toEqual(REQUESTED) + }) + + it('prefers a grid the provider applied on attach', () => { + expect( + resolveCommittedPtySize({ + result: { + isReattach: true, + attachedGrid: { cols: 100, rows: 30 }, + snapshotCols: LIVE.cols, + snapshotRows: LIVE.rows + }, + requested: REQUESTED, + cachedBeforeAttach: CACHED + }) + ).toEqual({ cols: 100, rows: 30 }) + }) + + it('falls back to the reattach snapshot grid', () => { + expect( + resolveCommittedPtySize({ + result: { isReattach: true, snapshotCols: LIVE.cols, snapshotRows: LIVE.rows }, + requested: REQUESTED, + cachedBeforeAttach: CACHED + }) + ).toEqual(LIVE) + }) + + it('falls back to the size main held when the provider proves nothing', () => { + expect( + resolveCommittedPtySize({ + result: { isReattach: true }, + requested: REQUESTED, + cachedBeforeAttach: CACHED + }) + ).toEqual(CACHED) + }) + + it('rejects a non-integer provider grid as unproven', () => { + expect( + resolveCommittedPtySize({ + result: { isReattach: true, snapshotCols: 120.5, snapshotRows: 40 }, + requested: REQUESTED, + cachedBeforeAttach: CACHED + }) + ).toEqual(CACHED) + }) + + it('takes the request only when nothing better exists', () => { + expect( + resolveCommittedPtySize({ + result: { isReattach: true }, + requested: REQUESTED, + cachedBeforeAttach: undefined + }) + ).toEqual(REQUESTED) + }) + + it('rejects a non-positive provider grid rather than publishing a zero-width model', () => { + expect( + resolveCommittedPtySize({ + result: { isReattach: true, snapshotCols: 0, snapshotRows: 0 }, + requested: REQUESTED, + cachedBeforeAttach: CACHED + }) + ).toEqual(CACHED) + }) +}) + +describe('commitAttachedPtySize', () => { + afterEach(() => { + ptySizes.delete('pty-commit') + }) + + it('records the resolved grid and reflows the model onto it for a reattach', () => { + const reflow = vi.fn() + const committed = commitAttachedPtySize({ + result: { + id: 'pty-commit', + isReattach: true, + snapshotCols: LIVE.cols, + snapshotRows: LIVE.rows + }, + requested: REQUESTED, + cachedBeforeAttach: undefined, + reflowHeadlessTerminalToPtyGrid: reflow + }) + expect(committed).toEqual(LIVE) + expect(ptySizes.get('pty-commit')).toEqual(LIVE) + expect(reflow).toHaveBeenCalledWith('pty-commit', LIVE.cols, LIVE.rows) + }) + + it('reflows a fresh spawn onto the request too: bytes can create the model before the reply', () => { + const reflow = vi.fn() + commitAttachedPtySize({ + result: { id: 'pty-commit' }, + requested: REQUESTED, + cachedBeforeAttach: CACHED, + reflowHeadlessTerminalToPtyGrid: reflow + }) + expect(ptySizes.get('pty-commit')).toEqual(REQUESTED) + expect(reflow).toHaveBeenCalledWith('pty-commit', REQUESTED.cols, REQUESTED.rows) + }) +}) diff --git a/src/main/ipc/pty/delivery/attached-pty-size.ts b/src/main/ipc/pty/delivery/attached-pty-size.ts new file mode 100644 index 00000000000..d8b14267cb8 --- /dev/null +++ b/src/main/ipc/pty/delivery/attached-pty-size.ts @@ -0,0 +1,81 @@ +import type { PtySpawnResult } from '../../../providers/types' +import { ptySizes } from './visibility-state' + +export type PtyGrid = { cols: number; rows: number } + +function positiveGrid(cols: unknown, rows: unknown): PtyGrid | undefined { + return typeof cols === 'number' && + typeof rows === 'number' && + Number.isInteger(cols) && + Number.isInteger(rows) && + cols > 0 && + rows > 0 + ? { cols, rows } + : undefined +} + +/** Pre-attach seed for `ptySizes`. Daemon PTYs can emit before spawn() resolves, so a genuinely + * fresh session must record its geometry now or early bytes parse at xterm's 80x24 default. + * An attach must not seed: a pane that mounted while hidden reports xterm's unmeasured default, + * and the live PTY's real grid is either already cached or arrives with the attach result. */ +export function shouldSeedPreAttachPtySize(args: { + isFreshSessionId: boolean + hasCachedSize: boolean + requestIsUnmeasured: boolean +}): boolean { + return args.isFreshSessionId || (!args.hasCachedSize && !args.requestIsUnmeasured) +} + +/** Grid to record for a settled spawn. Daemon and relay attach never resize the session they hand + * back, so on a reattach the requested grid describes the pane, not the live process — take the + * provider's proven grid, then the size main already held, before trusting the request. */ +export function resolveCommittedPtySize(args: { + result: Pick + requested: PtyGrid + cachedBeforeAttach: PtyGrid | undefined +}): PtyGrid { + if (args.result.isReattach !== true) { + return args.requested + } + return ( + positiveGrid(args.result.attachedGrid?.cols, args.result.attachedGrid?.rows) ?? + positiveGrid(args.result.snapshotCols, args.result.snapshotRows) ?? + positiveGrid(args.cachedBeforeAttach?.cols, args.cachedBeforeAttach?.rows) ?? + args.requested + ) +} + +type HeadlessReflow = ((ptyId: string, cols: number, rows: number) => void) | undefined + +/** Reflow main's model onto the committed grid, whatever the spawn was. Why unconditional: live + * bytes can lazily create the model at the 80x24 default before the reply arrives, a seed skips an + * existing model, and the pre-attach seed is now withheld for unmeasured attaches, so a session the + * daemon re-created instead of attaching would otherwise keep the default forever. */ +export function reflowHeadlessTerminalToCommittedGrid(args: { + result: Pick + committedSize: PtyGrid + reflowHeadlessTerminalToPtyGrid: HeadlessReflow +}): void { + args.reflowHeadlessTerminalToPtyGrid?.( + args.result.id, + args.committedSize.cols, + args.committedSize.rows + ) +} + +/** Record the settled grid, then reflow the model onto it. Callers that seed the model between the + * two steps (ipc spawn commit) call the halves separately. */ +export function commitAttachedPtySize(args: { + result: Pick< + PtySpawnResult, + 'id' | 'isReattach' | 'attachedGrid' | 'snapshotCols' | 'snapshotRows' + > + requested: PtyGrid + cachedBeforeAttach: PtyGrid | undefined + reflowHeadlessTerminalToPtyGrid: HeadlessReflow +}): PtyGrid { + const committedSize = resolveCommittedPtySize(args) + ptySizes.set(args.result.id, committedSize) + reflowHeadlessTerminalToCommittedGrid({ ...args, committedSize }) + return committedSize +} diff --git a/src/main/ipc/pty/ipc/inspect.ts b/src/main/ipc/pty/ipc/inspect.ts index 041abaa7854..5a6d03d8855 100644 --- a/src/main/ipc/pty/ipc/inspect.ts +++ b/src/main/ipc/pty/ipc/inspect.ts @@ -172,7 +172,12 @@ export function installPtyInspectIpcHandlers(deps: { 'pty:inspectProcess', async ( _event, - args: { id: string; expectedIncarnationId?: string; scanChildProcesses?: boolean } + args: { + id: string + expectedIncarnationId?: string + scanChildProcesses?: boolean + steadyState?: boolean + } ) => { // Why: same routing hazard as pty:hasPty — an unroutable id must read as client-only unverifiable, not as a local-provider answer or a raised IPC error. if (typeof args?.id !== 'string' || !args.id || args.id.startsWith('remote:')) { @@ -189,7 +194,8 @@ export function installPtyInspectIpcHandlers(deps: { ...(args.expectedIncarnationId ? { expectedIncarnationId: args.expectedIncarnationId } : {}), - ...(args.scanChildProcesses === true ? { scanChildProcesses: true } : {}) + ...(args.scanChildProcesses === true ? { scanChildProcesses: true } : {}), + ...(args.steadyState === true ? { steadyState: true } : {}) } return Object.keys(options).length > 0 ? inspectPtyProviderProcessForRenderer(getProviderForPty(args.id), args.id, options) diff --git a/src/main/ipc/pty/ipc/spawn-commit-persist.ts b/src/main/ipc/pty/ipc/spawn-commit-persist.ts index 540ada9397d..d9bee3e6934 100644 --- a/src/main/ipc/pty/ipc/spawn-commit-persist.ts +++ b/src/main/ipc/pty/ipc/spawn-commit-persist.ts @@ -11,12 +11,14 @@ import { } from '../pane/serializer-state' import { ptyOwnership, ptyIncarnationById, deletePtyOwnership } from '../provider/ownership-state' import { ptySizes } from '../delivery/visibility-state' +import { resolveCommittedPtySize, type PtyGrid } from '../delivery/attached-pty-size' import { clearProviderPtyState } from '../provider/state-cleanup' import type { PtyIpcSpawnState } from './spawn-state' export async function persistPtyIpcSpawnCommit(ctx: PtyIpcSpawnState): Promise<{ rendererPreSignaled: boolean rendererAlreadyRegistered: boolean + committedSize: PtyGrid }> { const args = ctx.args try { @@ -89,7 +91,12 @@ export async function persistPtyIpcSpawnCommit(ctx: PtyIpcSpawnState): Promise<{ ctx.agentTeamsLeaderHandle = null } } - ptySizes.set(ctx.result.id, { cols: args.cols, rows: args.rows }) + const committedSize = resolveCommittedPtySize({ + result: ctx.result, + requested: { cols: args.cols, rows: args.rows }, + cachedBeforeAttach: ctx.sessionSizeBeforeAttach + }) + ptySizes.set(ctx.result.id, committedSize) if (ctx.effectiveSessionAppId !== undefined && ctx.effectiveSessionAppId !== ctx.result.id) { ptySizes.delete(ctx.effectiveSessionAppId) } @@ -157,5 +164,5 @@ export async function persistPtyIpcSpawnCommit(ctx: PtyIpcSpawnState): Promise<{ pendingPtyIdBySerializerGeneration.set(pending.gen, ctx.result.id) } } - return { rendererPreSignaled, rendererAlreadyRegistered } + return { rendererPreSignaled, rendererAlreadyRegistered, committedSize } } diff --git a/src/main/ipc/pty/ipc/spawn-commit.ts b/src/main/ipc/pty/ipc/spawn-commit.ts index f02eb215c87..90b700f24b2 100644 --- a/src/main/ipc/pty/ipc/spawn-commit.ts +++ b/src/main/ipc/pty/ipc/spawn-commit.ts @@ -24,10 +24,12 @@ import { } from '../pane/launch-authority' import type { PtyIpcSpawnState } from './spawn-state' import { persistPtyIpcSpawnCommit } from './spawn-commit-persist' +import { reflowHeadlessTerminalToCommittedGrid } from '../delivery/attached-pty-size' export async function commitPtyIpcSpawn(ctx: PtyIpcSpawnState): Promise { const args = ctx.args - const { rendererPreSignaled, rendererAlreadyRegistered } = await persistPtyIpcSpawnCommit(ctx) + const { rendererPreSignaled, rendererAlreadyRegistered, committedSize } = + await persistPtyIpcSpawnCommit(ctx) // Why: seed the headless emulator before registerPty so concurrent live PTY data lands on top of the seed, not replacing it (mobile keeps the daemon-restored scrollback). // Skip when the renderer will be authoritative — its xterm buffer is richer than the daemon snapshot. @@ -71,6 +73,15 @@ export async function commitPtyIpcSpawn(ctx: PtyIpcSpawnState): Promise 0 && diff --git a/src/main/ipc/pty/ipc/spawn-env.ts b/src/main/ipc/pty/ipc/spawn-env.ts index af5da3858bd..94f1acf363e 100644 --- a/src/main/ipc/pty/ipc/spawn-env.ts +++ b/src/main/ipc/pty/ipc/spawn-env.ts @@ -7,7 +7,11 @@ import { isRemoteAgentHooksEnabled } from '../../../../shared/agent-hook-relay' import { isOpaqueRemintedPaneKey } from '../../../../shared/pane-key-alias' import { isValidTerminalTabId } from '../../../../shared/terminal-tab-id' import { isClaudeAuthSwitchInProgress } from '../../../claude-accounts/live-pty-gate' -import { hasClaudeAuthEnvConflict } from '../../../claude-accounts/environment' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE, + hasClaudeAuthEnvConflict +} from '../../../claude-accounts/environment' import { LocalPtyProvider } from '../../../providers/local-pty-provider' import { resolvePathEnvKey } from '../../../pty/windows-environment-path' import { routesFreshSpawnsToLocalProvider } from '../host-env/fresh-spawn-routing' @@ -20,12 +24,10 @@ import { assemblePtyIpcSpawnCodexEnv } from './spawn-env-codex' export async function assemblePtyIpcSpawnEnv(ctx: PtyIpcSpawnState): Promise { const args = ctx.args if (ctx.isClaudeLaunch && isClaudeAuthSwitchInProgress()) { - throw new Error('A Claude account switch is in progress. Try again after it finishes.') + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) } if (ctx.claudeAuth?.stripAuthEnv && hasClaudeAuthEnvConflict(args.env)) { - throw new Error( - 'This Claude launch defines explicit Anthropic auth environment variables. Remove those overrides before using a managed Claude account.' - ) + throw new Error(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) } // Why: the daemon-backed provider skips LocalPtyProvider's buildSpawnEnv, so assemble the same host-local env here for parity. // Safety: skip entirely for SSH — every injection is a loopback secret or a local path that leaks or misleads on the remote host. diff --git a/src/main/ipc/pty/ipc/spawn-options.ts b/src/main/ipc/pty/ipc/spawn-options.ts index ed75b989f6d..78747e6a462 100644 --- a/src/main/ipc/pty/ipc/spawn-options.ts +++ b/src/main/ipc/pty/ipc/spawn-options.ts @@ -17,6 +17,7 @@ import { pendingRuntimePaneCreatesByOwnerKey } from '../pane/spawn-reservation' import { ptySizes } from '../delivery/visibility-state' +import { shouldSeedPreAttachPtySize } from '../delivery/attached-pty-size' import { getStartupTerminalColorQueryReplyColors } from '../../terminal-startup-color-query-replies' import type { PtyIpcSpawnState } from './spawn-state' @@ -97,7 +98,14 @@ export async function buildPtyIpcSpawnOptions( ctx.effectiveSessionAppId !== undefined ? ptySizes.has(ctx.effectiveSessionAppId) : false ctx.sessionSizeBeforeAttach = ctx.effectiveSessionAppId !== undefined ? ptySizes.get(ctx.effectiveSessionAppId) : undefined - if (ctx.effectiveSessionId !== undefined) { + if ( + ctx.effectiveSessionId !== undefined && + shouldSeedPreAttachPtySize({ + isFreshSessionId: ctx.isMintedSessionId, + hasCachedSize: ctx.hadSessionSizeBeforeAttach, + requestIsUnmeasured: args.initiallyHidden === true + }) + ) { // Why: daemon PTYs can emit before spawn() resolves; set real geometry now or early bytes default to 80x24 and wrap TUIs. ptySizes.set(ctx.effectiveSessionAppId ?? ctx.effectiveSessionId, { cols: args.cols, diff --git a/src/main/ipc/pty/ipc/spawn-preflight.ts b/src/main/ipc/pty/ipc/spawn-preflight.ts index f40b46e5dd5..f885d6e2f6f 100644 --- a/src/main/ipc/pty/ipc/spawn-preflight.ts +++ b/src/main/ipc/pty/ipc/spawn-preflight.ts @@ -4,6 +4,7 @@ import { } from '../../../../shared/local-windows-terminal-runtime' import { isWslUncPath, toWindowsWslPath } from '../../../../shared/wsl-paths' import { isClaudeAuthSwitchInProgress } from '../../../claude-accounts/live-pty-gate' +import { CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE } from '../../../claude-accounts/environment' import { mintPtySessionId } from '../../../daemon/pty-session-id' import { resolveWslSessionContext } from '../../../daemon/wsl-session-context' import { LocalPtyProvider } from '../../../providers/local-pty-provider' @@ -193,7 +194,7 @@ export async function preparePtyIpcSpawnPreflight(ctx: PtyIpcSpawnState): Promis ctx.isClaudeLaunch = !ctx.preAdoptedStablePane && !args.connectionId && isClaudeLaunchCommand(args.command) if (ctx.isClaudeLaunch && isClaudeAuthSwitchInProgress()) { - throw new Error('A Claude account switch is in progress. Try again after it finishes.') + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) } ctx.terminalRuntimeOptions = process.platform === 'win32' && !args.connectionId diff --git a/src/main/ipc/pty/ipc/spawn-reattach-size-cache.test.ts b/src/main/ipc/pty/ipc/spawn-reattach-size-cache.test.ts new file mode 100644 index 00000000000..53b361bc8fe --- /dev/null +++ b/src/main/ipc/pty/ipc/spawn-reattach-size-cache.test.ts @@ -0,0 +1,136 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { PtySpawnResult } from '../../../providers/types' +import { ptySizes } from '../delivery/visibility-state' +import { buildPtyIpcSpawnOptions } from './spawn-options' +import { commitPtyIpcSpawn } from './spawn-commit' +import { createPtyIpcSpawnState, type PtyIpcSpawnState } from './spawn-state' +import type { PtySpawnIpcArgs, PtySpawnIpcDeps } from './spawn-types' + +const SESSION_ID = 'orca-pty-session-1' +/** What a pane that mounted while `display:none` reports: xterm's unmeasured default. */ +const HIDDEN_PANE_REQUEST = { cols: 80, rows: 24 } +/** The grid the surviving daemon session is actually running at. */ +const LIVE_GRID = { cols: 211, rows: 57 } + +function makeRuntime() { + return { + seedHeadlessTerminal: vi.fn(), + reflowHeadlessTerminalToPtyGrid: vi.fn(), + registerPty: vi.fn(), + cancelPendingPtyRegistration: vi.fn(), + noteTerminalSpawnCommand: vi.fn(), + seedTerminalRestoreTail: vi.fn(), + registerPreAllocatedHandleForPty: vi.fn() + } +} + +function makeCtx(args: PtySpawnIpcArgs, runtime: ReturnType): PtyIpcSpawnState { + const deps = { + transitionSpawnHiddenRendererPtyDeliveryState: vi.fn(), + syncPtyBackgroundedDelivery: vi.fn(), + sendPtySpawnedToRenderer: vi.fn(), + runtime + } as unknown as PtySpawnIpcDeps + const ctx = createPtyIpcSpawnState(deps, args) + ctx.env = {} + ctx.isDaemonHostSpawn = true + ctx.effectiveSessionId = SESSION_ID + ctx.effectiveSessionAppId = SESSION_ID + // Mirrors spawn-preflight: a caller-supplied sessionId is an attach, never a fresh mint. + ctx.isMintedSessionId = args.sessionId === undefined + return ctx +} + +async function runSpawn( + args: PtySpawnIpcArgs, + result: PtySpawnResult +): Promise<{ + runtime: ReturnType + preAttachSize: { cols: number; rows: number } | undefined +}> { + const runtime = makeRuntime() + const ctx = makeCtx(args, runtime) + await buildPtyIpcSpawnOptions(ctx) + const preAttachSize = ptySizes.get(SESSION_ID) + ctx.result = result + await commitPtyIpcSpawn(ctx) + return { runtime, preAttachSize } +} + +describe('spawn size cache on reattach', () => { + afterEach(() => { + ptySizes.delete(SESSION_ID) + vi.restoreAllMocks() + }) + + it('records the reattached session real grid, not a hidden pane placeholder', async () => { + const { runtime, preAttachSize } = await runSpawn( + { ...HIDDEN_PANE_REQUEST, sessionId: SESSION_ID, initiallyHidden: true }, + { + id: SESSION_ID, + isReattach: true, + snapshotCols: LIVE_GRID.cols, + snapshotRows: LIVE_GRID.rows + } + ) + + // Pre-attach: an unmeasured request must not be published as the live PTY's size. + expect(preAttachSize).toBeUndefined() + expect(ptySizes.get(SESSION_ID)).toEqual(LIVE_GRID) + expect(runtime.reflowHeadlessTerminalToPtyGrid).toHaveBeenCalledWith( + SESSION_ID, + LIVE_GRID.cols, + LIVE_GRID.rows + ) + }) + + it('records the requested grid for a genuinely fresh spawn', async () => { + const { runtime, preAttachSize } = await runSpawn({ cols: 120, rows: 40 }, { id: SESSION_ID }) + + expect(preAttachSize).toEqual({ cols: 120, rows: 40 }) + expect(ptySizes.get(SESSION_ID)).toEqual({ cols: 120, rows: 40 }) + expect(runtime.reflowHeadlessTerminalToPtyGrid).toHaveBeenCalledWith(SESSION_ID, 120, 40) + }) + + // Why: the pre-attach seed is withheld for an unmeasured attach, and a daemon that restarted + // re-creates the session instead of attaching, so only the commit can size the model. + it('reflows a hidden attach the daemon answered with a fresh session onto the request', async () => { + const { runtime, preAttachSize } = await runSpawn( + { cols: 100, rows: 30, sessionId: SESSION_ID, initiallyHidden: true }, + { id: SESSION_ID } + ) + + expect(preAttachSize).toBeUndefined() + expect(ptySizes.get(SESSION_ID)).toEqual({ cols: 100, rows: 30 }) + expect(runtime.reflowHeadlessTerminalToPtyGrid).toHaveBeenCalledWith(SESSION_ID, 100, 30) + }) + + it('keeps the size main already held when the reattach carries no snapshot grid', async () => { + ptySizes.set(SESSION_ID, { cols: 180, rows: 50 }) + + const { preAttachSize } = await runSpawn( + { ...HIDDEN_PANE_REQUEST, sessionId: SESSION_ID, initiallyHidden: true }, + { id: SESSION_ID, isReattach: true } + ) + + expect(preAttachSize).toEqual({ cols: 180, rows: 50 }) + expect(ptySizes.get(SESSION_ID)).toEqual({ cols: 180, rows: 50 }) + }) + + it('prefers the grid the provider applied on attach over every other source', async () => { + ptySizes.set(SESSION_ID, { cols: 180, rows: 50 }) + + await runSpawn( + { ...HIDDEN_PANE_REQUEST, sessionId: SESSION_ID, initiallyHidden: true }, + { + id: SESSION_ID, + isReattach: true, + attachedGrid: { cols: 100, rows: 30 }, + snapshotCols: LIVE_GRID.cols, + snapshotRows: LIVE_GRID.rows + } + ) + + expect(ptySizes.get(SESSION_ID)).toEqual({ cols: 100, rows: 30 }) + }) +}) diff --git a/src/main/ipc/pty/runtime/spawn-commit-pty-size.test.ts b/src/main/ipc/pty/runtime/spawn-commit-pty-size.test.ts new file mode 100644 index 00000000000..84d87b75005 --- /dev/null +++ b/src/main/ipc/pty/runtime/spawn-commit-pty-size.test.ts @@ -0,0 +1,80 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { ptySizes } from '../delivery/visibility-state' +import { commitRuntimePtySpawn } from './spawn-commit' +import { createRuntimePtySpawnState, type RuntimePtySpawnArgs } from './spawn-state' +import type { PtyRuntimeControllerDeps } from './controller-deps' + +const PTY_ID = 'orca-pty-adopted' +const LIVE_GRID = { cols: 211, rows: 57 } + +function makeRuntime() { + return { + registerPreAllocatedHandleForPty: vi.fn(), + registerPty: vi.fn(), + reflowHeadlessTerminalToPtyGrid: vi.fn(), + seedHeadlessTerminal: vi.fn(), + noteTerminalSpawnCommand: vi.fn() + } +} + +describe('runtime spawn commit: adopted agent-session claim', () => { + afterEach(() => { + ptySizes.delete(PTY_ID) + }) + + function makeAdoptedCtx(result: Record) { + const runtime = makeRuntime() + const deps = { runtime, store: undefined, options: {} } as unknown as PtyRuntimeControllerDeps + const args = { cols: 120, rows: 40, worktreeId: 'wt-1' } as unknown as RuntimePtySpawnArgs + const ctx = createRuntimePtySpawnState(deps, args) + ctx.result = { + id: PTY_ID, + ...result, + agentSessionEnsure: { + disposition: 'adopted', + owner: { + claim: { kind: 'terminal' }, + generation: 'g1', + phase: 'live', + ptyId: PTY_ID, + surface: { worktreeId: 'wt-1', tabId: 'tab-1', leafId: 'leaf-1', terminalHandle: 'h1' } + } + } + } as unknown as typeof ctx.result + return { runtime, ctx } + } + + it('commits the live grid from the adoption reply before the early return', async () => { + const { runtime, ctx } = makeAdoptedCtx({ + isReattach: true, + snapshotCols: LIVE_GRID.cols, + snapshotRows: LIVE_GRID.rows + }) + + await commitRuntimePtySpawn(ctx) + + expect(ptySizes.get(PTY_ID)).toEqual(LIVE_GRID) + expect(runtime.reflowHeadlessTerminalToPtyGrid).toHaveBeenCalledWith( + PTY_ID, + LIVE_GRID.cols, + LIVE_GRID.rows + ) + }) + + // Why: the SSH relay's adopted reply carries neither isReattach nor snapshot dims; the + // adoption itself proves a live owner, so main keeps what it held rather than the request. + it('treats an adoption without a reattach flag as an attach and keeps the held size', async () => { + ptySizes.set(PTY_ID, LIVE_GRID) + const { runtime, ctx } = makeAdoptedCtx({}) + ctx.sessionSizeBeforeAttach = LIVE_GRID + + await commitRuntimePtySpawn(ctx) + + expect(ptySizes.get(PTY_ID)).toEqual(LIVE_GRID) + expect(runtime.reflowHeadlessTerminalToPtyGrid).toHaveBeenCalledWith( + PTY_ID, + LIVE_GRID.cols, + LIVE_GRID.rows + ) + }) +}) diff --git a/src/main/ipc/pty/runtime/spawn-commit-pty-size.ts b/src/main/ipc/pty/runtime/spawn-commit-pty-size.ts new file mode 100644 index 00000000000..9b73f093761 --- /dev/null +++ b/src/main/ipc/pty/runtime/spawn-commit-pty-size.ts @@ -0,0 +1,18 @@ +import { commitAttachedPtySize } from '../delivery/attached-pty-size' +import type { RuntimePtySpawnState } from './spawn-state' + +/** Record the settled grid for a runtime-path spawn; `result` is passed explicitly because the + * adopted-claim branch commits before it returns early. */ +export function commitRuntimePtySize( + ctx: RuntimePtySpawnState, + result: RuntimePtySpawnState['result'] +): void { + commitAttachedPtySize({ + result, + requested: { cols: ctx.args.cols, rows: ctx.args.rows }, + cachedBeforeAttach: ctx.sessionSizeBeforeAttach, + reflowHeadlessTerminalToPtyGrid: ctx.deps.runtime?.reflowHeadlessTerminalToPtyGrid?.bind( + ctx.deps.runtime + ) + }) +} diff --git a/src/main/ipc/pty/runtime/spawn-commit.ts b/src/main/ipc/pty/runtime/spawn-commit.ts index 23592604bea..7f8a9e38267 100644 --- a/src/main/ipc/pty/runtime/spawn-commit.ts +++ b/src/main/ipc/pty/runtime/spawn-commit.ts @@ -1,6 +1,7 @@ import { isValidTerminalTabId } from '../../../../shared/terminal-tab-id' import { ptyOwnership, ptyIncarnationById, deletePtyOwnership } from '../provider/ownership-state' import { ptySizes } from '../delivery/visibility-state' +import { commitRuntimePtySize } from './spawn-commit-pty-size' import { shouldSkipCodexHomeEnvForWindowsShell, recordCodexPaneAccountForSpawn, @@ -57,6 +58,9 @@ export async function commitRuntimePtySpawn(ctx: RuntimePtySpawnState) { }) } if (ctx.result.agentSessionEnsure?.disposition === 'adopted') { + // Why: an adoption is an attach to a live owner by definition, but the SSH relay's adopted + // reply omits isReattach; derive it once so the size commit and the reservation agree. + const adoptedResult = { ...ctx.result, isReattach: true } const owner = ctx.result.agentSessionEnsure.owner ptyOwnership.set(ctx.result.id, args.connectionId ?? ptyOwnership.get(ctx.result.id) ?? null) ctx.deps.runtime?.registerPreAllocatedHandleForPty(ctx.result.id, owner.surface.terminalHandle) @@ -86,13 +90,17 @@ export async function commitRuntimePtySpawn(ctx: RuntimePtySpawnState) { ...(ctx.env ? { launchEnv: ctx.env } : {}) }) } + // Why: this branch returns before the normal commit site; without this the cache keeps + // whatever the caller requested. + commitRuntimePtySize(ctx, adoptedResult) // Why: the adopted branch returns before the normal settle site, so the // reservation must be resolved here or every later spawn for this pane // awaits a promise that never settles. - resolvePaneSpawnReservation(ctx.paneSpawnReservationKey, ctx.paneSpawnReservation, { - ...ctx.result, - isReattach: true - }) + resolvePaneSpawnReservation( + ctx.paneSpawnReservationKey, + ctx.paneSpawnReservation, + adoptedResult + ) return { id: ctx.result.id, ...(ctx.result.incarnationId ? { incarnationId: ctx.result.incarnationId } : {}), @@ -125,7 +133,7 @@ export async function commitRuntimePtySpawn(ctx: RuntimePtySpawnState) { if (!ctx.hostSessionBinding) { persistSshLease() } - ptySizes.set(ctx.result.id, { cols: args.cols, rows: args.rows }) + commitRuntimePtySize(ctx, ctx.result) if (ctx.effectiveSessionAppId !== undefined && ctx.effectiveSessionAppId !== ctx.result.id) { ptySizes.delete(ctx.effectiveSessionAppId) } diff --git a/src/main/ipc/pty/runtime/spawn-options.ts b/src/main/ipc/pty/runtime/spawn-options.ts index fcba7770a39..2e81bdeafdb 100644 --- a/src/main/ipc/pty/runtime/spawn-options.ts +++ b/src/main/ipc/pty/runtime/spawn-options.ts @@ -3,6 +3,7 @@ import { LocalPtyProvider } from '../../../providers/local-pty-provider' import { makePaneKey, isTerminalLeafId } from '../../../../shared/stable-pane-id' import { isValidTerminalTabId } from '../../../../shared/terminal-tab-id' import { ptySizes } from '../delivery/visibility-state' +import { shouldSeedPreAttachPtySize } from '../delivery/attached-pty-size' import { CODEX_HOME_ENV_KEYS } from '../host-env/codex-home' import { mergePtyEnvDeletions, @@ -110,7 +111,17 @@ export async function buildRuntimePtySpawnOptions( ctx.effectiveSessionAppId !== undefined ? ptySizes.get(ctx.effectiveSessionAppId) : undefined if (ctx.sessionId !== undefined) { ctx.spawnOptions.sessionId = ctx.sessionId - ptySizes.set(ctx.effectiveSessionAppId ?? ctx.sessionId, { cols: args.cols, rows: args.rows }) + if ( + shouldSeedPreAttachPtySize({ + isFreshSessionId: ctx.isNewDaemonSession, + hasCachedSize: ctx.hadSessionSizeBeforeAttach, + // Why false: runtime callers (CLI, headless serve) have no hidden pane to report, so a + // cached size is the only source that can outrank their requested grid here. + requestIsUnmeasured: false + }) + ) { + ptySizes.set(ctx.effectiveSessionAppId ?? ctx.sessionId, { cols: args.cols, rows: args.rows }) + } } ctx.materializedPaneKey = ctx.hostSessionBinding ? makePaneKey(ctx.hostSessionBinding.tabId, ctx.hostSessionBinding.leafId) diff --git a/src/main/ipc/pty/runtime/spawn-preflight.ts b/src/main/ipc/pty/runtime/spawn-preflight.ts index aed89b44df8..43fd2778119 100644 --- a/src/main/ipc/pty/runtime/spawn-preflight.ts +++ b/src/main/ipc/pty/runtime/spawn-preflight.ts @@ -23,7 +23,11 @@ import { import { stripRemotePaneEnvWhenHooksDisabled } from '../provider/liveness' import { isTuiAgent } from '../../../../shared/tui-agent-config' import { isClaudeAuthSwitchInProgress } from '../../../claude-accounts/live-pty-gate' -import { hasClaudeAuthEnvConflict } from '../../../claude-accounts/environment' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE, + hasClaudeAuthEnvConflict +} from '../../../claude-accounts/environment' import { isSafePtySessionId, mintPtySessionId, @@ -65,7 +69,7 @@ export async function prepareRuntimePtySpawn( ctx.isClaudeLaunch = !ctx.preAdoptedStablePane && !args.connectionId && isClaudeLaunchCommand(args.command) if (ctx.isClaudeLaunch && isClaudeAuthSwitchInProgress()) { - throw new Error('A Claude account switch is in progress. Try again after it finishes.') + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) } // Why: runtime-created terminals carry no renderer-computed projectRuntime; resolve from worktreeId to honor the project's Windows runtime. ctx.terminalRuntimeOptions = @@ -134,12 +138,10 @@ export async function prepareRuntimePtySpawn( ? await ctx.deps.prepareClaudeAuth(ctx.codexSelectionTarget) : null if (ctx.isClaudeLaunch && isClaudeAuthSwitchInProgress()) { - throw new Error('A Claude account switch is in progress. Try again after it finishes.') + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) } if (ctx.claudeAuth?.stripAuthEnv && hasClaudeAuthEnvConflict(args.env)) { - throw new Error( - 'This Claude launch defines explicit Anthropic auth environment variables. Remove those overrides before using a managed Claude account.' - ) + throw new Error(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) } ctx.shouldPersistHostSessionBinding = args.persistHostSessionBinding === true diff --git a/src/main/ipc/runtime.test.ts b/src/main/ipc/runtime.test.ts index dc7f8a01cd7..07010087363 100644 --- a/src/main/ipc/runtime.test.ts +++ b/src/main/ipc/runtime.test.ts @@ -136,6 +136,40 @@ describe('registerRuntimeHandlers', () => { }) }) + it('projects Claude structured tabs to the same-version desktop client', async () => { + const claudeTab = { + type: 'agent-session', + id: 'agent-session:claude-1', + title: 'Claude Chat', + sessionId: 'claude-1', + agent: 'claude', + isActive: true + } + const runtime = { + getRuntimeId: vi.fn().mockReturnValue('runtime-1'), + restoreStructuredAgentSessionTabs: vi.fn(async () => undefined), + listMobileSessionTabs: vi.fn(async () => ({ + worktree: 'workspace-1', + publicationEpoch: 'epoch-1', + snapshotVersion: 1, + activeGroupId: 'group-1', + activeTabId: claudeTab.id, + activeTabType: 'agent-session', + tabGroups: [{ id: 'group-1', activeTabId: claudeTab.id, tabOrder: [claudeTab.id] }], + tabs: [claudeTab] + })) + } + + registerRuntimeHandlers(runtime as never) + const callRegistration = handleMock.mock.calls.find(([channel]) => channel === 'runtime:call') + const result = await callRegistration![1](runtimeCallEvent(), { + method: 'session.tabs.list', + params: { worktree: 'id:workspace-1' } + }) + + expect(result).toMatchObject({ ok: true, result: { tabs: [claudeTab] } }) + }) + it('registers project group runtime RPC methods for local desktop callers', async () => { const runtime = { syncWindowGraph: vi.fn(), diff --git a/src/main/ipc/runtime.ts b/src/main/ipc/runtime.ts index 901d14bfce6..3901d8b1ffa 100644 --- a/src/main/ipc/runtime.ts +++ b/src/main/ipc/runtime.ts @@ -10,7 +10,10 @@ import type { import type { RuntimeRpcResponse } from '../../shared/runtime-rpc-envelope' import type { ClientHostedBrowserRowsEvent } from '../../shared/client-hosted-browser-rows' import { TERMINAL_FIT_RESTORE_DEADLINE_MS } from '../../shared/terminal-fit-restore-deadline' -import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../shared/protocol-version' +import { + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY +} from '../../shared/protocol-version' import { RpcDispatcher } from '../runtime/rpc/dispatcher' import { ALL_RPC_METHODS } from '../runtime/rpc/methods' import { DesktopRuntimeSenderLifecycle } from './desktop-runtime-sender-lifecycle' @@ -76,7 +79,10 @@ export function registerRuntimeHandlers(runtime: OrcaRuntimeService): void { clientId: 'desktop-renderer', clientKind: 'runtime', connectionId: desktopSenders.connectionIdFor(event.sender), - clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + clientCapabilities: [ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ] } )) as RuntimeRpcResponse } @@ -121,7 +127,10 @@ export function registerRuntimeHandlers(runtime: OrcaRuntimeService): void { clientId: 'desktop-renderer', clientKind: 'runtime', connectionId, - clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + clientCapabilities: [ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ] } ) .finally(stop) diff --git a/src/main/ipc/worktrees-test-module-mocks.ts b/src/main/ipc/worktrees-test-module-mocks.ts index a2139c4c762..8d2787fcf2f 100644 --- a/src/main/ipc/worktrees-test-module-mocks.ts +++ b/src/main/ipc/worktrees-test-module-mocks.ts @@ -110,6 +110,7 @@ export const gitWorktreeModuleMock = () => ({ listWorktrees: listWorktreesMock, listWorktreesStrict: listWorktreesMock, listWorktreesSharedStrict: listWorktreesMock, + listWorktreesSharedStrictAllowingTrueEmpty: listWorktreesMock, describeCreatedWorktree: describeCreatedWorktreeMock, parseWorktreeList: parseWorktreeListMock, assertWorktreeCleanForRemoval: assertWorktreeCleanForRemovalMock, diff --git a/src/main/ipc/worktrees-windows.test.ts b/src/main/ipc/worktrees-windows.test.ts index 4d4115264a3..7a54200d9f8 100644 --- a/src/main/ipc/worktrees-windows.test.ts +++ b/src/main/ipc/worktrees-windows.test.ts @@ -74,6 +74,7 @@ vi.mock('../git/worktree', () => ({ listWorktrees: listWorktreesMock, listWorktreesStrict: listWorktreesMock, listWorktreesSharedStrict: listWorktreesMock, + listWorktreesSharedStrictAllowingTrueEmpty: listWorktreesMock, describeCreatedWorktree: vi.fn().mockResolvedValue(undefined), assertWorktreeCleanForRemoval: assertWorktreeCleanForRemovalMock, addWorktree: addWorktreeMock, diff --git a/src/main/ipc/worktrees/listing/detected-provider-listing.ts b/src/main/ipc/worktrees/listing/detected-provider-listing.ts index 388c825530f..ec5e7606e45 100644 --- a/src/main/ipc/worktrees/listing/detected-provider-listing.ts +++ b/src/main/ipc/worktrees/listing/detected-provider-listing.ts @@ -25,7 +25,11 @@ import { type DetectedWorktreeMetadataPrune, type DetectedWorktreeSideEffectToken } from './detected-worktree-scan-cache' -import { loggedWorktreeListFailures, warnOnce } from './worktree-listing-diagnostics' +import { + describeWorktreeScanFailure, + loggedWorktreeListFailures, + warnOnce +} from './worktree-listing-diagnostics' import { readAllWorktreeMetaForRepo } from '../../../persistence/host-qualified-worktree-meta' export async function listDetectedWorktreesForCapturedRepo( @@ -157,16 +161,25 @@ export async function listDetectedWorktreesForCapturedRepo( `[worktrees] failed to list detected worktrees for repo "${repo.displayName}" (${repo.id}) at ${repo.path}`, err ) + // Why: retention alone leaves inert rows with no explanation; the cause rides with the listing. + const unavailableReason = describeWorktreeScanFailure(err) if (repo.connectionId) { const worktrees = listDisconnectedSshWorktrees(store, repo, sshWorktreeMetaIndex()) return { repoId: repo.id, authoritative: false, source: 'metadata-fallback', - worktrees: buildDisconnectedDetectedWorktrees(store, repo, worktrees) + worktrees: buildDisconnectedDetectedWorktrees(store, repo, worktrees), + unavailableReason } } - return { repoId: repo.id, authoritative: false, source: 'metadata-fallback', worktrees: [] } + return { + repoId: repo.id, + authoritative: false, + source: 'metadata-fallback', + worktrees: [], + unavailableReason + } } } diff --git a/src/main/ipc/worktrees/listing/detected-scan-failure-authority.test.ts b/src/main/ipc/worktrees/listing/detected-scan-failure-authority.test.ts new file mode 100644 index 00000000000..4e436b8dcf1 --- /dev/null +++ b/src/main/ipc/worktrees/listing/detected-scan-failure-authority.test.ts @@ -0,0 +1,166 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { Repo } from '../../../../shared/repo-types' +import type { DetectedWorktreeListResult } from '../../../../shared/worktree/types' + +const gitExecFileAsyncMock = vi.hoisted(() => vi.fn()) + +vi.mock('electron', () => ({ + ipcMain: { handle: vi.fn(), removeHandler: vi.fn() }, + app: { getPath: () => '/tmp/orca-test' } +})) +vi.mock('../../../git/runner', async (importOriginal) => ({ + ...(await importOriginal>()), + gitExecFileAsync: gitExecFileAsyncMock +})) + +const { listDetectedWorktreesForCapturedRepo } = await import('./detected-provider-listing') +const { __resetDetectedWorktreeScanCacheForTests } = await import('./detected-worktree-scan-cache') +const { _resetWorktreeScanCacheForTests } = await import('../../../git/worktree-scan-cache') +const { isRegisteredWorktreePath, invalidateAuthorizedRootsCache } = + await import('../../registered-worktree-roots-cache') + +const REPO_PATH = '/workspace/repo' +const repo = { + id: 'repo-1', + path: REPO_PATH, + displayName: 'repo', + badgeColor: '#000', + addedAt: 0 +} as Repo + +const removeWorktreeLineage = vi.fn() + +function createStore() { + return { + getRepo: () => repo, + getRepos: () => [repo], + getProjects: () => [], + getSettings: () => ({}), + getAllWorktreeMeta: () => ({}), + getProjectHostSetups: () => [], + getWorktreeMeta: () => undefined, + setWorktreeMeta: vi.fn(), + getAllWorktreeLineage: () => ({}), + getAllWorkspaceLineage: () => ({}), + removeWorktreeLineage, + captureNativeLocalWorktreeMetadataScanExpectation: () => undefined + } as never +} + +/** The field failure: wsl.exe exits 0xFFFFFFFF, says nothing on stderr, and git never ran. */ +function wslHostFailure(): Error { + return Object.assign(new Error('Command failed: wsl.exe -d kali-linux --exec sh -lc ...'), { + code: 4294967295, + stdout: 'Error code: Wsl/Service/WSL_E_DISTRO_NOT_FOUND\r\n', + stderr: '' + }) +} + +async function listDetected(): Promise { + const result = await listDetectedWorktreesForCapturedRepo(createStore(), repo, () => true) + return result as DetectedWorktreeListResult +} + +describe('detected worktree listing authority', () => { + beforeEach(() => { + gitExecFileAsyncMock.mockReset() + removeWorktreeLineage.mockReset() + __resetDetectedWorktreeScanCacheForTests() + _resetWorktreeScanCacheForTests() + invalidateAuthorizedRootsCache() + }) + + it('reports a failed scan as non-authoritative and prunes nothing', async () => { + gitExecFileAsyncMock.mockRejectedValue(wslHostFailure()) + + const result = await listDetected() + + expect(result.authoritative).toBe(false) + expect(result.source).toBe('metadata-fallback') + expect(result.worktrees).toEqual([]) + // Why: the retained rows must carry the cause, or the user sees inert worktrees with no explanation. + expect(result.unavailableReason).toContain('Command failed: wsl.exe') + // The destructive halves of a fresh scan must not run against a listing that failed. + expect(isRegisteredWorktreePath(REPO_PATH)).toBe(false) + expect(removeWorktreeLineage).not.toHaveBeenCalled() + }) + + it('surfaces the annotated wsl.exe diagnostic as the unavailable reason', async () => { + gitExecFileAsyncMock.mockRejectedValue( + Object.assign( + new Error( + 'wsl.exe host failure (distro "kali-linux"): There is no distribution with the supplied name.\r\nError code: Wsl/Service/WSL_E_DISTRO_NOT_FOUND\nCommand failed: wsl.exe -d kali-linux --exec sh -lc ...' + ), + { code: 4294967295, stdout: '', stderr: '' } + ) + ) + + const result = await listDetected() + + expect(result.authoritative).toBe(false) + expect(result.unavailableReason).toBe( + 'wsl.exe host failure (distro "kali-linux"): There is no distribution with the supplied name. Error code: Wsl/Service/WSL_E_DISTRO_NOT_FOUND' + ) + }) + + // Why (measured on a real Windows host): under WSL the spawn cwd is the interop directory, so a + // deleted guest repo fails as `bash: cd` exit 1 — not ENOENT — and must stay retained, not pruned. + it('retains a WSL repo whose guest directory is gone, and says why', async () => { + gitExecFileAsyncMock.mockRejectedValue( + Object.assign(new Error('bash: line 1: cd: /home/neil/repo: No such file or directory'), { + code: 1, + stdout: '', + stderr: 'bash: line 1: cd: /home/neil/repo: No such file or directory\n' + }) + ) + + const result = await listDetected() + + expect(result.authoritative).toBe(false) + expect(result.unavailableReason).toContain('No such file or directory') + expect(isRegisteredWorktreePath(REPO_PATH)).toBe(false) + expect(removeWorktreeLineage).not.toHaveBeenCalled() + }) + + it('keeps an empty listing authoritative when the path is not a Git repo', async () => { + gitExecFileAsyncMock.mockRejectedValue( + Object.assign(new Error('Command failed: git worktree list'), { + code: 128, + stderr: 'fatal: not a git repository (or any of the parent directories): .git\n' + }) + ) + + const result = await listDetected() + + expect(result.authoritative).toBe(true) + expect(result.source).toBe('git') + expect(result.worktrees).toEqual([]) + expect(result.unavailableReason).toBeUndefined() + expect(isRegisteredWorktreePath(REPO_PATH)).toBe(true) + }) + + it('keeps an empty listing authoritative when the repo path is gone', async () => { + gitExecFileAsyncMock.mockRejectedValue( + Object.assign(new Error('spawn git ENOENT'), { code: 'ENOENT', stderr: '' }) + ) + + const result = await listDetected() + + expect(result.authoritative).toBe(true) + expect(result.source).toBe('git') + expect(result.worktrees).toEqual([]) + }) + + it('stays authoritative for a healthy scan', async () => { + gitExecFileAsyncMock.mockResolvedValue({ + stdout: `worktree ${REPO_PATH}\u0000HEAD abc\u0000branch refs/heads/main\u0000\u0000`, + stderr: '' + }) + + const result = await listDetected() + + expect(result.authoritative).toBe(true) + expect(result.worktrees.map((worktree) => worktree.path)).toEqual([REPO_PATH]) + expect(isRegisteredWorktreePath(REPO_PATH)).toBe(true) + }) +}) diff --git a/src/main/ipc/worktrees/listing/detected-worktree-scan-cache.ts b/src/main/ipc/worktrees/listing/detected-worktree-scan-cache.ts index b694bfbf11d..09b0156cb54 100644 --- a/src/main/ipc/worktrees/listing/detected-worktree-scan-cache.ts +++ b/src/main/ipc/worktrees/listing/detected-worktree-scan-cache.ts @@ -3,7 +3,7 @@ import type { Store } from '../../../persistence/loading-store/store' import type { Repo } from '../../../../shared/repo-types' import { getLocalProjectWorktreeGitOptions } from '../../../project-runtime-git-options' import { isFolderRepo } from '../../../../shared/repo-kind' -import { listRepoWorktrees } from '../../../repo-worktrees' +import { listRepoWorktreesForDetectedScan } from '../../../repo-worktrees' import { getRegisteredWorktreeRootsRevision, registerWorktreeRootsForRepo @@ -111,7 +111,7 @@ export async function listDetectedGitWorktrees( const localWorktreeGitOptions = getLocalProjectWorktreeGitOptions(store, repo) if (repo.connectionId || isFolderRepo(repo)) { return { - gitWorktrees: await listRepoWorktrees(repo, localWorktreeGitOptions), + gitWorktrees: await listRepoWorktreesForDetectedScan(repo, localWorktreeGitOptions), fresh: true } } @@ -144,7 +144,7 @@ export async function listDetectedGitWorktrees( : undefined const scan: DetectedWorktreeScan = { invalidated: false, - promise: listRepoWorktrees(repo, localWorktreeGitOptions), + promise: listRepoWorktreesForDetectedScan(repo, localWorktreeGitOptions), sideEffectToken: { generation, authorizedRootsRevision }, hygieneDue, ...(metadataPruneExpectation diff --git a/src/main/ipc/worktrees/listing/detected-worktree-scan-hygiene-gate.test.ts b/src/main/ipc/worktrees/listing/detected-worktree-scan-hygiene-gate.test.ts index cb8c48833ef..655cca4fa77 100644 --- a/src/main/ipc/worktrees/listing/detected-worktree-scan-hygiene-gate.test.ts +++ b/src/main/ipc/worktrees/listing/detected-worktree-scan-hygiene-gate.test.ts @@ -10,7 +10,9 @@ const { listRepoWorktreesMock, pruneLineageMock, pruneMetadataMock, registerWork registerWorktreeRootsMock: vi.fn() })) -vi.mock('../../../repo-worktrees', () => ({ listRepoWorktrees: listRepoWorktreesMock })) +vi.mock('../../../repo-worktrees', () => ({ + listRepoWorktreesForDetectedScan: listRepoWorktreesMock +})) vi.mock('../../../project-runtime-git-options', () => ({ getLocalProjectWorktreeGitOptions: () => ({}) })) diff --git a/src/main/ipc/worktrees/listing/worktree-listing-diagnostics.ts b/src/main/ipc/worktrees/listing/worktree-listing-diagnostics.ts index 128bb46dd28..56bbe5d4ad2 100644 --- a/src/main/ipc/worktrees/listing/worktree-listing-diagnostics.ts +++ b/src/main/ipc/worktrees/listing/worktree-listing-diagnostics.ts @@ -14,3 +14,23 @@ export function warnOnce(keySet: Set, key: string, message: string, erro console.warn(message) } } + +const SCAN_FAILURE_REASON_MAX_CHARS = 240 + +/** + * The cause a retained-but-unscannable repo shows the user. The first two lines carry the + * classifier's summary plus its `Wsl/Service/WSL_E_*` code; everything after is the raw command. + */ +export function describeWorktreeScanFailure(error: unknown): string { + const message = error instanceof Error ? error.message : String(error) + const summary = message + .split(/\r?\n/) + .map((line) => line.trim()) + .filter((line) => line.length > 0) + .slice(0, 2) + .join(' ') + const reason = summary.length > 0 ? summary : 'Worktree scan failed with no diagnostic.' + return reason.length > SCAN_FAILURE_REASON_MAX_CHARS + ? `${reason.slice(0, SCAN_FAILURE_REASON_MAX_CHARS - 1)}…` + : reason +} diff --git a/src/main/main-process-tree-kill-gate.test.ts b/src/main/main-process-tree-kill-gate.test.ts new file mode 100644 index 00000000000..40b1421c32f --- /dev/null +++ b/src/main/main-process-tree-kill-gate.test.ts @@ -0,0 +1,184 @@ +import { readFileSync, readdirSync, statSync } from 'node:fs' +import { join, relative, resolve } from 'node:path' +import { describe, expect, it } from 'vitest' + +/** + * The ratchet behind the guard's claim to be a choke point. + * + * `admitSelfInitiatedTreeKill` is only "one decision" for as long as every + * pid-addressed `taskkill /pid /t /f` in Electron main asks it. Each such + * kill can land on a recycled pid that is now one of Orca's own Chromium + * processes (#10680), and an ungated one is also invisible to + * `selfInitiatedTreeKillCount`, which makes a zero read as exculpatory when it + * is not. A new family fails here rather than in the field. + * + * Exactly what is enforced, so no comment elsewhere claims more: per file, the + * number of gate admissions must be at least the number of `/pid` call sites. + * Counting sites rather than files is the point — a file-granular scan would let + * a second, ungated taskkill land inside a family that already mentions the gate, + * which is the shape the six highest-risk files now have. What it still cannot + * see: a site that pairs an ungated kill with a second admission of an already + * gated one in the same file, and a kill whose `/pid` argument is itself built + * from a variable. + */ +const REPOSITORY_ROOT = resolve(__dirname, '..', '..') +const MAIN_DIRECTORY = 'src/main/' +const SCANNED_EXTENSIONS = ['.ts', '.tsx'] +const IGNORED_DIRECTORIES = new Set([ + 'node_modules', + 'dist', + 'out', + 'build', + '.git', + '__fixtures__' +]) + +/** + * One match per call site. Keyed on the `/pid` argument rather than the program + * name because `/pid ` is what makes the kill pid-addressed — it walks + * whatever tree owns that pid *now* — and because the literal survives a + * `taskkill` spawned through a constant or a variable, which a quoted-program + * pattern misses entirely. + */ +const PID_ADDRESSED_KILL_SITE = /['"]\/pid['"]/gi + +/** + * A call, not an import or a comment: `admitSelfInitiatedTreeKill` in main, and + * `admitProcessTreeKill` for the `src/shared` seam main installs the same gate + * into, which shared code cannot import directly. + */ +const GATE_ADMISSION = /\badmit(?:SelfInitiatedTreeKill|ProcessTreeKill)\s*\(/g + +function countMatches(source: string, pattern: RegExp): number { + return source.match(pattern)?.length ?? 0 +} + +/** Sites left over once each admission in the file has claimed one. */ +function ungatedKillSiteCount(source: string): number { + return Math.max( + countMatches(source, PID_ADDRESSED_KILL_SITE) - countMatches(source, GATE_ADMISSION), + 0 + ) +} + +/** + * Only ever shrinks. Each entry states why the gate cannot reach it — never + * "not got to yet", which is what a new ungated family would also look like. + */ +const UNGATED_TASKKILL_ALLOWLIST = new Map([ + [ + 'src/main/browser/browser-route-egress-electron-launch.ts', + 'Electron probe reached only from *.electron.test.ts; kills the probe Electron it spawned' + ], + [ + 'src/main/browser/browser-route-persisted-worker-electron-process.ts', + 'Electron probe reached only from *.electron.test.ts; kills the probe Electron it spawned' + ], + [ + 'src/cli/handlers/interactive-login-interruption.ts', + 'CLI host: no Chromium pid on the machine to reach, and no reader for the ring' + ], + [ + 'src/relay/subprocess-tree-termination.ts', + 'Relay host: same, and the relay cannot import the main-process gate' + ] +]) + +function isTestFile(path: string): boolean { + return /\.(?:test|spec)\.tsx?$/.test(path) || /(?:test-harness|test-fixture|fixture)/.test(path) +} + +function scanSourceFiles(directory: string, found: string[] = []): string[] { + for (const entry of readdirSync(directory)) { + if (IGNORED_DIRECTORIES.has(entry)) { + continue + } + const path = join(directory, entry) + if (statSync(path).isDirectory()) { + scanSourceFiles(path, found) + continue + } + if (SCANNED_EXTENSIONS.some((extension) => entry.endsWith(extension)) && !isTestFile(path)) { + found.push(path) + } + } + return found +} + +// Only the Node-side hosts: a renderer or preload cannot spawn a process at all. +const SCANNED_HOSTS = ['src/main', 'src/shared', 'src/cli', 'src/relay'] + +/** Scanned once at import: 10k files is seconds, and every case below reuses it. */ +const PID_ADDRESSED_KILL_FILES = SCANNED_HOSTS.flatMap((host) => + scanSourceFiles(join(REPOSITORY_ROOT, host)) + .map((path) => ({ + path: relative(REPOSITORY_ROOT, path).split('\\').join('/'), + source: readFileSync(path, 'utf8') + })) + .filter((file) => countMatches(file.source, PID_ADDRESSED_KILL_SITE) > 0) +) + +function pidAddressedKillFiles(): { path: string; source: string }[] { + return PID_ADDRESSED_KILL_FILES +} + +describe('main-process tree-kill gate', () => { + it('finds the taskkill families it is meant to police', () => { + // Falsifiable: a scanner that matched nothing would pass every case below. + expect(pidAddressedKillFiles().map((file) => file.path)).toContain( + 'src/main/windows-process-tree-kill.ts' + ) + }) + + it('routes every pid-addressed taskkill in Electron main through the gate', () => { + const ungated = pidAddressedKillFiles() + .filter((file) => file.path.startsWith(MAIN_DIRECTORY)) + .filter((file) => ungatedKillSiteCount(file.source) > 0) + .map((file) => file.path) + .filter((path) => !UNGATED_TASKKILL_ALLOWLIST.has(path)) + + expect(ungated).toEqual([]) + }) + + it('leaves no pid-addressed taskkill outside main unaccounted for', () => { + const unaccounted = pidAddressedKillFiles() + .filter((file) => !file.path.startsWith(MAIN_DIRECTORY)) + .filter((file) => ungatedKillSiteCount(file.source) > 0) + .map((file) => file.path) + .filter((path) => !UNGATED_TASKKILL_ALLOWLIST.has(path)) + + expect(unaccounted).toEqual([]) + }) + + it('counts call sites, not files: a second ungated kill in a gated file is caught', () => { + // The failure a file-granular scan let through: one gate mention exempting + // every taskkill in the file. + const gated = ` + import { admitSelfInitiatedTreeKill } from './own-chromium-tree-kill-guard' + if (admitSelfInitiatedTreeKill({ pid, site: 's', scope: 'win-taskkill-tree' })) { + spawn('taskkill', ['/pid', String(pid), '/t', '/f']) + } + ` + + expect(ungatedKillSiteCount(gated)).toBe(0) + expect( + ungatedKillSiteCount(`${gated}\nspawn('taskkill', ['/pid', String(other), '/t', '/f'])`) + ).toBe(1) + }) + + it('sees a kill whose program name comes from a constant', () => { + // A quoted-program pattern misses this shape; the `/pid` argument does not. + expect( + ungatedKillSiteCount(` + const KILLER = 'taskkill' + spawn(KILLER, ['/pid', String(pid), '/t', '/f']) + `) + ).toBe(1) + }) + + it('keeps the allowlist honest: every entry still spawns a taskkill', () => { + const spawning = new Set(pidAddressedKillFiles().map((file) => file.path)) + + expect([...UNGATED_TASKKILL_ALLOWLIST.keys()].filter((path) => !spawning.has(path))).toEqual([]) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts b/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts index 4f3ef118af5..4f21285e417 100644 --- a/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts +++ b/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts @@ -1,4 +1,4 @@ -// SDKMessage discriminators from Claude Agent SDK 0.3.231 / Claude Code 2.1.231. +// SDKMessage discriminators from Claude Agent SDK 0.3.251 / Claude Code 2.1.258. export const CLAUDE_STREAM_JSON_FRAME_KINDS = [ 'message:assistant', 'message:user', @@ -42,7 +42,15 @@ export const CLAUDE_STREAM_JSON_FRAME_KINDS = [ 'message:prompt_suggestion', 'message:system:mirror_error', 'message:system:informational', - 'message:conversation_reset' + 'message:conversation_reset', + // Queue bookkeeping the CLI emits per client-supplied command uuid. Absent + // from the SDK's SDKMessage union, which is why it reached users as raw JSON. + 'message:command_lifecycle', + 'message:result:success', + 'message:result:error_during_execution', + 'message:result:error_max_turns', + 'message:result:error_max_budget_usd', + 'message:result:error_max_structured_output_retries' ] as const export type ClaudeStreamJsonFrameKind = (typeof CLAUDE_STREAM_JSON_FRAME_KINDS)[number] diff --git a/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts b/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts index ad9ca66c52a..22bd645d8a6 100644 --- a/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts +++ b/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts @@ -65,6 +65,29 @@ describe('provider frame classification catalog', () => { ).toBe('error-surface') }) + it('keeps command queue bookkeeping off the transcript without hiding a failed one', () => { + expect( + classifyProviderFrame('claude', 'message:command_lifecycle', { + command_uuid: 'command-1', + state: 'started' + }) + ).toBe('status-chrome') + expect( + classifyProviderFrame('claude', 'message:command_lifecycle', { + command_uuid: 'command-1', + state: 'cancelled' + }) + ).toBe('status-chrome') + // Payload inspection outranks the catalogue, so suppressing the kind cannot + // swallow a state the provider reports as a failure. + expect( + classifyProviderFrame('claude', 'message:command_lifecycle', { + command_uuid: 'command-1', + state: 'failed' + }) + ).toBe('error-surface') + }) + it('keeps unknown future frames on the substantive bounded fallback path', () => { expect(classifyProviderFrame('codex', 'notification:future/event', {})).toBe( 'timeline-substantive' diff --git a/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts b/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts index 8d11df995a6..474b1385a4f 100644 --- a/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts +++ b/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts @@ -131,7 +131,18 @@ export const PROVIDER_FRAME_CLASSIFICATIONS = { 'message:prompt_suggestion': 'status-chrome', 'message:system:mirror_error': 'error-surface', 'message:system:informational': 'timeline-substantive', - 'message:conversation_reset': 'status-chrome' + 'message:conversation_reset': 'status-chrome', + // A `started`/`completed`/`cancelled` state for one queued command uuid and + // nothing else; the CLI keeps it out of its own transcript too. A state that + // reads as a failure still surfaces, via the payload check in classify. + 'message:command_lifecycle': 'status-chrome', + // The turn-complete signal: lifecycle, never a transcript row. Error subtypes + // included — the turn's assistant frames already carry any user-facing text. + 'message:result:success': 'status-chrome', + 'message:result:error_during_execution': 'status-chrome', + 'message:result:error_max_turns': 'status-chrome', + 'message:result:error_max_budget_usd': 'status-chrome', + 'message:result:error_max_structured_output_retries': 'status-chrome' } } as const satisfies ProviderFrameClassificationTable diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.test.ts new file mode 100644 index 00000000000..c6566083eac --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.test.ts @@ -0,0 +1,114 @@ +import { describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' +import { StructuredAgentSessionAdapterRouter } from './structured-agent-session-adapter-router' + +function adapterOf( + releaseAcquisition: StructuredAgentSessionAdapter['releaseAcquisition'] +): StructuredAgentSessionAdapter { + return { + acquire: vi.fn(async () => ({ process: { pid: 1 } }) as never), + releaseAcquisition, + dispatch: vi.fn(), + cancelTurn: vi.fn(), + answerPrompt: vi.fn(), + setOption: vi.fn() + } as unknown as StructuredAgentSessionAdapter +} + +describe('StructuredAgentSessionAdapterRouter.releaseAcquisition', () => { + it('drops the owner even when its release reports a typed failure', async () => { + const failure = new Error('root exited') + const claude = adapterOf(vi.fn().mockRejectedValueOnce(failure).mockResolvedValue(false)) + const codex = adapterOf(vi.fn(async () => false)) + const router = new StructuredAgentSessionAdapterRouter({ claude, codex }, async () => {}) + const identity = { sessionId: 'session-1', agent: 'claude' } as never + await router.acquire({ identity, fence: 1, spawnToken: 'spawn-1' }) + + await expect(router.releaseAcquisition({ sessionId: 'session-1' })).rejects.toBe(failure) + // With no owner left, a later release asks every adapter instead of the stale one. + await expect(router.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(false) + expect(claude.releaseAcquisition).toHaveBeenCalledTimes(2) + expect(codex.releaseAcquisition).toHaveBeenCalledTimes(1) + }) +}) + +describe('StructuredAgentSessionAdapterRouter.closeSession', () => { + it('retains the owner after an unproven close so a later retry reaches the same adapter', async () => { + const claude = adapterOf(vi.fn(async () => true)) + const closeSession = vi.fn().mockResolvedValueOnce(false).mockResolvedValueOnce(true) + const dispatch = vi.fn().mockResolvedValue({ state: 'unknown', reason: 'test' }) + claude.closeSession = closeSession + claude.dispatch = dispatch + const codex = adapterOf(vi.fn(async () => false)) + const router = new StructuredAgentSessionAdapterRouter({ claude, codex }, async () => {}) + const identity = { sessionId: 'session-1', agent: 'claude' } as never + await router.acquire({ identity, fence: 1, spawnToken: 'spawn-1' }) + + await expect(router.closeSession('session-1')).resolves.toBe(false) + await expect( + router.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: {} as never, + fence: 1 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + await expect(router.closeSession('session-1')).resolves.toBe(true) + expect(closeSession).toHaveBeenCalledTimes(2) + expect(dispatch).toHaveBeenCalledTimes(1) + }) +}) + +describe('StructuredAgentSessionAdapterRouter optional lifecycle methods', () => { + it.each([ + ['forceCloseSession', 'forceCloseSession'], + ['disposeSession', 'disposeSession'] + ] as const)( + '%s forwards to the owner and retains it until proven stopped', + async (_label, method) => { + const claude = adapterOf(vi.fn(async () => true)) + const stop = vi.fn().mockResolvedValueOnce(false).mockResolvedValueOnce(true) + claude[method] = stop + const dispatch = vi.fn().mockResolvedValue({ state: 'unknown', reason: 'test' }) + claude.dispatch = dispatch + const codex = adapterOf(vi.fn(async () => false)) + const router = new StructuredAgentSessionAdapterRouter({ claude, codex }, async () => {}) + const identity = { sessionId: 'session-1', agent: 'claude' } as never + await router.acquire({ identity, fence: 1, spawnToken: 'spawn-1' }) + const stopSession = router[method] + + await expect(stopSession('session-1')).resolves.toBe(false) + await expect( + router.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: {} as never, + fence: 1 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + await expect(stopSession('session-1')).resolves.toBe(true) + expect(stop).toHaveBeenCalledTimes(2) + expect(dispatch).toHaveBeenCalledOnce() + } + ) + + it.each(['forceCloseSession', 'disposeSession'] as const)( + 'falls back to closeSession when an owner lacks %s', + async (method) => { + const closeSession = vi.fn().mockResolvedValue(true) + const claude = adapterOf(vi.fn(async () => true)) + claude.closeSession = closeSession + const codex = adapterOf(vi.fn(async () => false)) + const router = new StructuredAgentSessionAdapterRouter({ claude, codex }, async () => {}) + await router.acquire({ + identity: { sessionId: 'session-1', agent: 'claude' } as never, + fence: 1, + spawnToken: 'spawn-1' + }) + const stopSession = router[method] + + await expect(stopSession('session-1')).resolves.toBe(true) + expect(closeSession).toHaveBeenCalledWith('session-1') + } + ) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.ts new file mode 100644 index 00000000000..226b9c1aab5 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.ts @@ -0,0 +1,135 @@ +import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' +import type { AgentSessionExecutionLocation } from '../../../shared/agent-session-record' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' + +type RoutedAgent = 'claude' | 'codex' + +export class StructuredAgentSessionAdapterRouter implements StructuredAgentSessionAdapter { + private readonly owners = new Map() + + constructor( + private readonly adapters: Record, + private readonly closeAdapters: () => Promise + ) {} + + supportsCreate = (location: AgentSessionExecutionLocation, agent: string): boolean => { + const adapter = this.adapterForAgent(agent) + return adapter ? (adapter.supportsLocation?.(location) ?? false) : false + } + + supportsLocation = (location: AgentSessionExecutionLocation): boolean => + Object.values(this.adapters).some((adapter) => adapter.supportsLocation?.(location) ?? false) + + async acquire(input: Parameters[0]) { + const adapter = this.requireAgent(input.identity) + const acquired = await adapter.acquire(input) + this.owners.set(input.identity.sessionId, adapter) + return acquired + } + + async releaseAcquisition(input: { sessionId: string }): Promise { + const adapter = this.owners.get(input.sessionId) + if (adapter) { + try { + return (await adapter.releaseAcquisition?.(input)) === true + } finally { + this.owners.delete(input.sessionId) + } + } + let released = false + for (const candidate of Object.values(this.adapters)) { + released = (await candidate.releaseAcquisition?.(input)) === true || released + } + return released + } + + dispatch: StructuredAgentSessionAdapter['dispatch'] = (input) => + this.owner(input.sessionId).dispatch(input) + + cancelTurn: StructuredAgentSessionAdapter['cancelTurn'] = (input) => + this.owner(input.sessionId).cancelTurn(input) + + stopBackgroundTasks: NonNullable = ( + input + ) => { + const stop = this.owner(input.sessionId).stopBackgroundTasks + return stop ? stop(input) : Promise.resolve({ cancelled: false }) + } + + backgroundTaskState: NonNullable = ( + sessionId + ) => this.owners.get(sessionId)?.backgroundTaskState?.(sessionId) + + answerPrompt: StructuredAgentSessionAdapter['answerPrompt'] = (input) => + this.owner(input.sessionId).answerPrompt(input) + + setOption: StructuredAgentSessionAdapter['setOption'] = (input) => + this.owner(input.sessionId).setOption(input) + + readOptions = (input: { sessionId: string; fence: number }) => { + const reader = this.owner(input.sessionId).readOptions + if (!reader) { + throw new Error(`structured session ${input.sessionId} does not report options`) + } + return reader(input) + } + + readOptionRestoreFailures = (sessionId: string): readonly string[] => + this.owner(sessionId).readOptionRestoreFailures?.(sessionId) ?? [] + + historyFilePath = (input: { identity: AgentSessionJournalIdentity }) => + this.requireAgent(input.identity).historyFilePath?.(input) ?? Promise.resolve(null) + + closeSession = (sessionId: string): Promise => + this.stopSession(sessionId, (adapter) => adapter.closeSession) + + forceCloseSession = (sessionId: string): Promise => + this.stopSession(sessionId, (adapter) => adapter.forceCloseSession ?? adapter.closeSession) + + disposeSession = (sessionId: string): Promise => + this.stopSession(sessionId, (adapter) => adapter.disposeSession ?? adapter.closeSession) + + private async stopSession( + sessionId: string, + selectStop: ( + adapter: StructuredAgentSessionAdapter + ) => NonNullable | undefined + ): Promise { + const adapter = this.owners.get(sessionId) + if (!adapter) { + return false + } + const stop = selectStop(adapter) + const stopped = await stop?.call(adapter, sessionId) + if (stopped === true) { + this.owners.delete(sessionId) + return true + } + return false + } + + async closeAll(): Promise { + this.owners.clear() + await this.closeAdapters() + } + + private owner(sessionId: string): StructuredAgentSessionAdapter { + const adapter = this.owners.get(sessionId) + if (!adapter) { + throw new Error(`no live structured adapter owns ${sessionId}`) + } + return adapter + } + + private requireAgent(identity: AgentSessionJournalIdentity): StructuredAgentSessionAdapter { + const adapter = this.adapterForAgent(identity.agent) + if (!adapter) { + throw new Error(`structured sessions do not support ${identity.agent}`) + } + return adapter + } + + private adapterForAgent(agent: string): StructuredAgentSessionAdapter | null { + return agent === 'claude' || agent === 'codex' ? this.adapters[agent] : null + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.test.ts index 5f67240c1c6..77cce9153f5 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it, vi } from 'vitest' import { AgentSessionAcquisitionExitUnprovenError, + AgentSessionAcquisitionRootExitObservedError, rethrowAfterAgentSessionAcquisitionCleanup } from './structured-agent-session-adapter' @@ -28,6 +29,27 @@ describe('failed agent-session acquisition cleanup', () => { ).rejects.toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) }) + it('keeps a first-hand root exit that cleanup observed, with the provider diagnostic', async () => { + const cause = new Error('proof failed') + const exit = new AgentSessionAcquisitionRootExitObservedError( + new Error('claude stream-json exited (code 1): crashed') + ) + const error = await rethrowAfterAgentSessionAcquisitionCleanup( + { + releaseAcquisition: vi.fn(async () => { + throw exit + }) + }, + 'session-1', + cause + ).catch((thrown: unknown) => thrown) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + expect(error).not.toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) + expect((error as Error).message).toBe('claude stream-json exited (code 1): crashed') + expect((error as Error).cause).toMatchObject({ errors: [cause, exit] }) + }) + it('reports unproven exit when cleanup throws', async () => { const error = await rethrowAfterAgentSessionAcquisitionCleanup( { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts index cccc8ce6f13..e44e8c39152 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts @@ -17,6 +17,7 @@ import type { AgentSessionProcessIdentity } from '../../../shared/agent-session-record' import type { + AgentSessionBackgroundTaskState, AgentSessionOptionsResult, AgentSessionWireRefusalCode } from '../../../shared/agent-session-wire' @@ -32,6 +33,20 @@ export class AgentSessionAcquisitionRefusal extends Error { } } +/** + * The provider's own root process was observed to exit, but its descendant tree + * could not be verified. The lease keys on the root's pid and start time, so its + * observed death releases the reservation; nothing is claimed about descendants. + * Never thrown when a descendant was observed still alive — that stays unproven. + */ +export class AgentSessionAcquisitionRootExitObservedError extends Error { + constructor(cause: unknown) { + // The provider's own diagnostic is the only thing the user can act on. + super(cause instanceof Error ? cause.message : String(cause), { cause }) + this.name = 'AgentSessionAcquisitionRootExitObservedError' + } +} + export class AgentSessionAcquisitionExitUnprovenError extends Error { constructor(cause: unknown) { super('agent_session_acquisition_exit_unproven', { cause }) @@ -49,7 +64,7 @@ export type AgentSessionAcquisition = { acquisitionGeneration?: string } -/** Acquisition validation failed before the adapter attempted to spawn. */ +/** Acquisition failed with first-hand proof that no provider process existed. */ export class AgentSessionPreSpawnError extends Error { constructor(cause: unknown) { super(cause instanceof Error ? cause.message : String(cause), { cause }) @@ -105,7 +120,9 @@ export type StructuredAgentSessionAdapter = { * at — the store rejects a link minted at any other fence. */ acquire(input: StructuredAgentSessionAcquireInput): Promise /** Reaps an acquired provider when the host cannot commit or prove its lease. - * Returns true only after provider child exit is proven. */ + * Returns true only after provider child exit is proven. Throws + * `AgentSessionAcquisitionRootExitObservedError` when the provider root's own + * exit was observed first-hand but its descendants could not be verified. */ releaseAcquisition?(input: { sessionId: string }): Promise dispatch(input: { sessionId: string @@ -120,6 +137,8 @@ export type StructuredAgentSessionAdapter = { turnId: string fence: number }): Promise<{ cancelled: boolean }> + stopBackgroundTasks?(input: { sessionId: string; fence: number }): Promise<{ cancelled: boolean }> + backgroundTaskState?(sessionId: string): AgentSessionBackgroundTaskState | null | undefined /** Fires the provider callback for an approval or a question. The wire calls * this only after the durable compare-and-set won, so it runs exactly once. */ answerPrompt(input: { @@ -133,6 +152,8 @@ export type StructuredAgentSessionAdapter = { input: StructuredAgentSessionSetOptionInput ): Promise>> readOptions?(input: { sessionId: string; fence: number }): Promise + /** Option keys skipped after a provider rejected their persisted restore value. */ + readOptionRestoreFailures?(sessionId: string): readonly string[] /** Transcript path for journal recovery. Omit to let the existing session-file * resolver discover it from the provider session id. */ historyFilePath?(input: { identity: AgentSessionJournalIdentity }): Promise @@ -154,9 +175,15 @@ export async function rethrowAfterAgentSessionAcquisitionCleanup( try { released = (await adapter.releaseAcquisition?.({ sessionId })) === true } catch (cleanupError) { - throw new AgentSessionAcquisitionExitUnprovenError( - new AggregateError([cause, cleanupError], 'agent session acquisition cleanup failed') - ) + // A root exit the cleanup observed first-hand keeps its classification and its + // provider diagnostic; the failure that triggered cleanup rides along as cause. + throw cleanupError instanceof AgentSessionAcquisitionRootExitObservedError + ? new AgentSessionAcquisitionRootExitObservedError( + new AggregateError([cause, cleanupError], cleanupError.message) + ) + : new AgentSessionAcquisitionExitUnprovenError( + new AggregateError([cause, cleanupError], 'agent session acquisition cleanup failed') + ) } if (released) { throw cause diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts index 05c3d8c9e5e..7113be8d54b 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts @@ -5,7 +5,6 @@ import type { AgentSessionWireRefusal } from '../../../shared/agent-session-wire' import type { AgentJournalResetReason } from '../../../shared/agent-session-journal-types' -import type { AgentSessionAttachParams } from './structured-agent-session-attach' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' import type { StructuredAgentSessionHostDeps, @@ -30,8 +29,6 @@ export type StructuredAgentSessionAttachContext = { } tasks: StructuredAgentSessionTaskQueue reconcileLeases: (sessionId: string) => Promise - /** Retries a durable provider-exit journal settlement before a new owner is reserved. */ - retryPendingSettlement?: (sessionId: string, params: AgentSessionAttachParams) => Promise serialize: (sessionId: string, task: () => Promise) => Promise now: () => number } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts index a21edcee35c..b08a56ea4d9 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts @@ -26,6 +26,7 @@ import type { AgentSessionRecordStore } from '../../runtime/agent-session-record import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' import { AgentSessionAcquisitionExitUnprovenError, + AgentSessionAcquisitionRootExitObservedError, AgentSessionAcquisitionRefusal, AgentSessionPreSpawnError, isAgentSessionPreSpawnError, @@ -119,7 +120,9 @@ export async function performAttach( ? 'processless' : error instanceof AgentSessionAcquisitionExitUnprovenError ? 'unproven' - : 'exit-proven' + : error instanceof AgentSessionAcquisitionRootExitObservedError + ? 'root-exit-observed' + : 'exit-proven' const outcome = error instanceof AgentSessionAcquisitionExitUnprovenError ? { @@ -209,13 +212,17 @@ async function settlePostAcquisitionAttachFailure( cause: unknown ): Promise { let cleanupError: unknown = cause - let exitProof: 'exit-proven' | 'unproven' = 'unproven' + let exitProof: 'exit-proven' | 'root-exit-observed' | 'unproven' = 'unproven' try { await rethrowAfterAgentSessionAcquisitionCleanup(input.adapter, record.sessionId, cause) } catch (error) { cleanupError = error exitProof = - error instanceof AgentSessionAcquisitionExitUnprovenError ? 'unproven' : 'exit-proven' + error instanceof AgentSessionAcquisitionExitUnprovenError + ? 'unproven' + : error instanceof AgentSessionAcquisitionRootExitObservedError + ? 'root-exit-observed' + : 'exit-proven' } // Why: the close is awaited so the map entry is gone only once its handle is // released, but a failed close must not also cost the store settlement below. diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts index f4551ef9313..a22bbdcbb3e 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts @@ -17,6 +17,7 @@ import { pinnedAgentSessionLaunchEnv } from './structured-agent-session-launch-env' import { refuseAgentSessionMutation } from './structured-agent-session-mutation-admission' +import { retryPendingStructuredAgentSessionSettlement } from './structured-agent-session-settlement-retry' import type { StructuredAgentSessionAttachContext } from './structured-agent-session-attach-context' import type { DeferredStructuredAgentSessionEventSink } from './structured-agent-session-event-sink' import { agentSessionJournalCloseRetries } from '../agent-session-journal/journal-close-retry' @@ -41,14 +42,20 @@ export function attachStructuredAgentSession( return refuseAgentSessionMutation(unreconciled) } await context.runtimeState.resolveRecovery(sessionId) - if (context.retryPendingSettlement) { - const settled = await context.retryPendingSettlement(sessionId, params) - if (!settled) { - return refuseAgentSessionMutation({ - code: 'agent_session_ownership_unknown', - message: 'The provider-exit terminal journal settlement is still pending; retry attach.' - }) - } + // Retries a durable provider-exit journal settlement before a new owner is reserved. Answers + // settled when the record has none pending, so every attach can ask unconditionally. + const settled = await retryPendingStructuredAgentSessionSettlement({ + deps: context.deps, + sessions: context.sessions, + sessionId, + params, + now: () => context.now() + }) + if (!settled) { + return refuseAgentSessionMutation({ + code: 'agent_session_ownership_unknown', + message: 'The provider-exit terminal journal settlement is still pending; retry attach.' + }) } const eventSink = context.runtimeState.eventSinkFor(sessionId) const attached = await performAttach({ diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts index 83766f25fc5..ce58e31b4ee 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts @@ -5,7 +5,6 @@ // the record store's compare-and-swap, which also owns the idempotency row, so // a retried attach replays instead of reserving a second owner. -import type { AgentType } from '../../../shared/agent-status-types' import type { AgentSessionJournalIdentity, AgentSessionProviderHandle @@ -52,7 +51,7 @@ export type AgentSessionAttachParams = { envelope: AgentSessionMutationEnvelope location: AgentSessionExecutionLocation provider: AgentSessionHandleProvider - agent: AgentType + agent: AgentSessionHandleProvider accountHome: AgentSessionAccountHome runtimeKind: AgentSessionOwnerRuntimeKind /** Omitted only for create-by-intent; the adapter proves the durable handle. */ diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-background-task-channel.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-background-task-channel.ts new file mode 100644 index 00000000000..4592dea26e9 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-background-task-channel.ts @@ -0,0 +1,62 @@ +import type { + AgentSessionBackgroundTaskState, + AgentSessionHistoryRequest, + AgentSessionHistoryResult +} from '../../../shared/agent-session-wire' +import { readStructuredAgentSessionHistoryResult } from './structured-agent-session-history-result' +import type { + AgentSessionSubscribers, + AgentSessionSubscribeInput +} from './structured-agent-session-subscribers' +import type { + StructuredAgentSessionHostDeps, + StructuredAgentSessionHostSession +} from './structured-agent-session-host-types' + +export class StructuredAgentSessionBackgroundTaskChannel { + constructor( + private readonly deps: StructuredAgentSessionHostDeps, + private readonly sessions: Map, + private readonly subscribers: AgentSessionSubscribers, + private readonly requireSession: (sessionId: string) => StructuredAgentSessionHostSession, + private readonly handoffStatus: ( + sessionId: string + ) => Parameters[0]['handoff'] + ) {} + + history(request: AgentSessionHistoryRequest): AgentSessionHistoryResult { + const result = readStructuredAgentSessionHistoryResult({ + journal: this.requireSession(request.sessionId).journal, + record: this.deps.store.getRecord(request.sessionId), + request + }) + const backgroundTasks = this.state(request.sessionId) + return backgroundTasks === undefined + ? result + : { ...result, page: { ...result.page, backgroundTasks } } + } + + subscribe(input: AgentSessionSubscribeInput): () => void { + const session = this.requireSession(input.sessionId) + const backgroundTasks = this.state(input.sessionId) + return this.subscribers.open({ + ...input, + journal: session.journal, + fence: this.deps.store.getRecord(input.sessionId)?.lease.runtimeFence ?? 0, + handoff: this.handoffStatus(input.sessionId), + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) + }) + } + + publish(sessionId: string, publishedState?: AgentSessionBackgroundTaskState | null): void { + const session = this.sessions.get(sessionId) + const state = publishedState !== undefined ? publishedState : this.state(sessionId) + if (session && state !== undefined) { + this.subscribers.backgroundTasks(sessionId, state, session.fence) + } + } + + private state(sessionId: string): AgentSessionBackgroundTaskState | null | undefined { + return this.deps.adapter.backgroundTaskState?.(sessionId) + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-claude-options-round-trip.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-claude-options-round-trip.test.ts new file mode 100644 index 00000000000..87bc33bc4b9 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-claude-options-round-trip.test.ts @@ -0,0 +1,182 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { + AgentSessionHandoffDirection, + AgentSessionHandoffRequest, + AgentSessionMutationEnvelope +} from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' +import { StructuredAgentSessionHost } from './structured-agent-session-host' +import { + HOST_TEST_NOW as NOW, + HOST_TEST_SESSION as SESSION, + hostTestAttachParams, + hostTestOperationId, + resetHostTestOperationIds +} from './structured-agent-session-host-test-data' +import type { + StructuredAgentSessionHandoffTransport, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +const CALLER = { callerKey: 'client-claude' } +const CLAUDE_SESSION = '019fd532-7c11-7a90-b6de-4e1a2c3d5f61' +const DEFAULT_MODEL = 'sonnet' +const PICKED_MODEL = 'opus' + +let root: string +let store: AgentSessionRecordStore +let host: StructuredAgentSessionHost +let acquire: Mock +let activeModel: string +let transcriptPath: string + +function envelope(method: string, fields: Record): AgentSessionMutationEnvelope { + return { + sessionId: SESSION, + clientOperationId: hostTestOperationId(), + expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method, + sessionId: SESSION, + fields + }) + } +} + +function handoff(direction: AgentSessionHandoffDirection): AgentSessionHandoffRequest { + const fields = { direction, mode: 'now' as const, action: 'start' as const } + return { envelope: envelope('agentSession.requestHandoff', fields), ...fields } +} + +function owner(fence: number, spawnToken: string): StructuredTuiOwner { + return { + terminal: { + handle: 'term-claude', + tabId: 'tab-claude', + paneKey: 'pane-claude', + ptyId: 'pty-claude' + }, + process: { hostId: 'local', pid: 5200, processStartTimeMs: NOW, spawnToken }, + link: { + linkId: `claude-tui-${fence}`, + handle: { provider: 'claude', sessionId: CLAUDE_SESSION, leafUuid: 'tui-leaf' }, + origin: 'resumed', + mintedAtFence: fence, + observedAt: NOW + }, + transcriptPath + } +} + +function transport(): StructuredAgentSessionHandoffTransport { + return { + hostLabel: 'Test host', + launchTui: async ({ fence, spawnToken }) => owner(fence, spawnToken), + reproveTuiOwner: async ({ owner: current }) => current, + recoverTuiOwner: async (record) => + owner( + record.lease.runtimeFence, + record.lease.ownerProcess?.spawnToken ?? record.lease.reservedSpawnToken ?? 'recovered' + ), + stopRecoveredOwner: async () => undefined, + waitForTuiExit: async (current) => ({ transcriptPath: current.transcriptPath }), + waitForTuiIdleOrExit: async () => 'idle', + tuiStatus: () => 'idle' + } +} + +function adapter(): StructuredAgentSessionAdapter { + acquire = vi.fn(async ({ fence, spawnToken, options }) => { + activeModel = options?.model ?? DEFAULT_MODEL + return { + process: { hostId: 'local', pid: 4200, processStartTimeMs: NOW, spawnToken }, + link: { + linkId: `claude-native-${fence}`, + handle: { provider: 'claude', sessionId: CLAUDE_SESSION, leafUuid: 'native-leaf' }, + origin: acquire.mock.calls.length === 1 ? 'created' : 'resumed', + mintedAtFence: fence, + observedAt: NOW + } + } + }) + return { + acquire, + dispatch: vi.fn(), + cancelTurn: vi.fn(async () => ({ cancelled: true })), + answerPrompt: vi.fn(async () => undefined), + setOption: vi.fn(async ({ value }) => { + activeModel = value + return { model: value } + }), + readOptions: vi.fn(async () => ({ current: { model: activeModel }, models: [] })), + closeSession: vi.fn(async () => { + activeModel = DEFAULT_MODEL + return true + }) + } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-claude-handoff-options-')) + resetHostTestOperationIds() + activeModel = DEFAULT_MODEL + transcriptPath = join(root, 'claude.jsonl') + await writeFile(transcriptPath, '', 'utf8') + store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) + host = new StructuredAgentSessionHost({ + store, + adapter: adapter(), + journalRoot: root, + claimKeyId: 'key-1', + mintSpawnToken: () => 'spawn-claude', + handoffTransport: transport(), + now: () => NOW + }) + expect( + await host.attach( + CALLER, + hostTestAttachParams(null, { + provider: 'claude', + agent: 'claude', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: join(root, 'claude-home') }, + providerHandle: { kind: 'claude', sessionId: CLAUDE_SESSION, leafUuid: 'native-leaf' } + }) + ) + ).toMatchObject({ ok: true }) +}) + +afterEach(async () => { + await new Promise((resolve) => setTimeout(resolve, 100)) + await host.flushAllStreamedEvents() + await rm(root, { recursive: true, force: true, maxRetries: 3, retryDelay: 50 }) +}) + +describe('Claude structured session handoff options', () => { + it('keeps a directly selected model through chat to TUI to chat', async () => { + const fields = { key: 'model', value: PICKED_MODEL } + expect( + await host.setOption(CALLER, { + envelope: envelope('agentSession.setOption', fields), + ...fields + }) + ).toMatchObject({ ok: true, value: { options: { model: PICKED_MODEL } } }) + + expect(await host.requestHandoff(CALLER, handoff('to-tui'))).toMatchObject({ ok: true }) + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'tui' }) + ) + expect(await host.requestHandoff(CALLER, handoff('to-native'))).toMatchObject({ ok: true }) + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'native' }) + ) + + expect(acquire.mock.calls[1]?.[0].options).toEqual({ model: PICKED_MODEL }) + expect(store.getRecord(SESSION)?.options).toEqual({ model: PICKED_MODEL }) + expect(activeModel).toBe(PICKED_MODEL) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-grouped-prompt.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-grouped-prompt.test.ts new file mode 100644 index 00000000000..6082ab074f5 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-grouped-prompt.test.ts @@ -0,0 +1,166 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { AgentSessionMutationEnvelope } from '../../../shared/agent-session-wire' +import { encodeAgentSessionQuestionAnswers } from '../../../shared/agent-session-question-answer' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { journalDirectoryFor } from '../agent-session-journal/journal-paths' +import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import type { + AgentSessionDispatchOutcome, + StructuredAgentSessionAdapter +} from './structured-agent-session-adapter' +import type { AgentSessionAttachParams } from './structured-agent-session-attach' +import { StructuredAgentSessionHost } from './structured-agent-session-host' +import { + HOST_TEST_NOW as NOW, + HOST_TEST_SESSION as SESSION, + HOST_TEST_THREAD as THREAD, + hostTestAttachParams, + hostTestOperationId, + resetHostTestOperationIds +} from './structured-agent-session-host-test-data' + +const CALLER = { callerKey: 'client-1' } + +function envelope(method: string, fields: Record): AgentSessionMutationEnvelope { + return { + sessionId: SESSION, + clientOperationId: hostTestOperationId(), + expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? 1, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method, + sessionId: SESSION, + fields + }) + } +} + +const attachParams = (): AgentSessionAttachParams => hostTestAttachParams(null) + +let root: string +let store: AgentSessionRecordStore +let host: StructuredAgentSessionHost +let acquire: Mock +let answerPrompt: Mock +let ordinal = 0 + +function adapter(): StructuredAgentSessionAdapter { + const dispatch = vi.fn(async (): Promise => { + ordinal += 1 + return { + state: 'accepted', + providerIdentity: { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal } + } + }) + return { + acquire, + releaseAcquisition: vi.fn(async () => true), + dispatch, + cancelTurn: vi.fn(async () => ({ cancelled: true })), + answerPrompt, + setOption: vi.fn(async () => undefined) + } +} + +async function seedGroupedQuestion(): Promise<{ itemId: string; revision: number }> { + const journal = await openAgentSessionJournal({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: THREAD } + }, + journalDir: journalDirectoryFor(root, { workspaceId: 'workspace-1', sessionId: SESSION }) + }) + const appended = await journal.appendItem( + { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal: 100 }, + { + kind: 'question', + question: '2 grouped questions from Claude', + options: [], + questions: [ + { + id: 'q1', + question: 'Targets', + multiSelect: true, + options: [ + { id: 'target-web', label: 'Web' }, + { id: 'target-mobile', label: 'Mobile' } + ] + }, + { + id: 'q2', + question: 'Host', + multiSelect: false, + options: [], + freeTextQuestionId: 'q2' + } + ], + resolution: { state: 'pending', selectedOptionId: null, resolvedBy: null, resolvedAt: null } + }, + { fence: 1 } + ) + return { itemId: appended.itemId, revision: appended.revision } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-wire-grouped-')) + resetHostTestOperationIds() + ordinal = 0 + acquire = vi.fn(async ({ fence }) => ({ + process: { + hostId: 'local', + pid: 4242, + processStartTimeMs: 1_700_000_000_000, + spawnToken: store.getRecord(SESSION)?.lease.reservedSpawnToken ?? 'spawn-a' + }, + link: { + linkId: `link-${fence}`, + handle: { provider: 'codex', threadId: THREAD }, + origin: store.getRecord(SESSION)?.providerHandleChain.length ? 'resumed' : 'created', + mintedAtFence: fence, + observedAt: NOW + } + })) + answerPrompt = vi.fn(async () => undefined) + store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) + host = new StructuredAgentSessionHost({ + store, + adapter: adapter(), + journalRoot: root, + claimKeyId: 'key-1', + mintSpawnToken: () => 'spawn-a', + now: () => NOW + }) +}) + +afterEach(async () => { + await host.flushAllStreamedEvents() + await rm(root, { recursive: true, force: true }) +}) + +describe('grouped question admission', () => { + it('admits renderer question-group payloads with child ids and multi-select answers', async () => { + const prompt = await seedGroupedQuestion() + const attached = await host.attach(CALLER, attachParams()) + expect(attached.ok).toBe(true) + const optionId = encodeAgentSessionQuestionAnswers([ + { questionId: 'q1', optionIds: ['target-web', 'target-mobile'] }, + { questionId: 'q2', optionIds: [], other: 'SSH host' } + ]) + const fields = { itemId: prompt.itemId, expectedRevision: prompt.revision, optionId } + const result = await host.respondToPrompt(CALLER, { + envelope: envelope('agentSession.respondTo:question', fields), + kind: 'question', + ...fields + }) + expect(result).toMatchObject({ ok: true, value: { resolution: { state: 'resolved' } } }) + expect(answerPrompt).toHaveBeenCalledWith( + expect.objectContaining({ itemId: prompt.itemId, optionId }) + ) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-admission.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-admission.ts new file mode 100644 index 00000000000..b881d55e771 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-admission.ts @@ -0,0 +1,138 @@ +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { AgentSessionOperationOutcome } from '../../../shared/agent-session-operation-ledger' +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffResult, + AgentSessionHandoffStatus, + AgentSessionMutationResult, + AgentSessionWireRefusal +} from '../../../shared/agent-session-wire' +import { AGENT_SESSION_WIRE_REFUSAL_CODES } from '../../../shared/agent-session-wire' +import { + agentSessionFingerprintConflict, + computeAgentSessionPayloadFingerprint +} from '../../../shared/agent-session-mutation-envelope' +import type { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import type { StructuredAgentSessionHandoffDeps } from './structured-agent-session-handoff-types' + +export type StructuredHandoffAdmission = + | { decision: 'continue'; record: AgentSessionRecord; fingerprint: string } + | { decision: 'replay'; outcome: AgentSessionOperationOutcome } + | { decision: 'refused'; refusal: AgentSessionWireRefusal } + +export async function admitStructuredHandoffRequest(input: { + deps: StructuredAgentSessionHandoffDeps + operationGuard: StructuredAgentSessionHandoffOperationGuard + callerKey: string + params: AgentSessionHandoffRequest + record: AgentSessionRecord + status?: AgentSessionHandoffStatus +}): Promise { + const action = input.params.action ?? 'start' + const requestFingerprint = computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff', + sessionId: input.record.sessionId, + fields: { direction: input.params.direction, mode: input.params.mode, action } + }) + const conflict = agentSessionFingerprintConflict(input.params.envelope, requestFingerprint) + if (conflict) { + return { decision: 'refused', refusal: conflict } + } + const fingerprint = computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff.operation', + sessionId: input.record.sessionId, + fields: { direction: input.params.direction } + }) + const operation = await input.operationGuard.check({ + callerKey: input.callerKey, + sessionId: input.record.sessionId, + operationId: input.params.envelope.clientOperationId, + fingerprint, + action, + ...(input.status ? { status: input.status } : {}), + now: input.deps.now() + }) + if (operation.decision === 'replay') { + return { decision: 'replay', outcome: operation.outcome } + } + if (operation.decision === 'refused') { + return { + decision: 'refused', + refusal: { + code: operation.code as 'agent_session_operation_conflict', + message: 'This handoff operation could not be admitted.' + } + } + } + if (input.params.envelope.expectedRuntimeFence !== input.record.lease.runtimeFence) { + await input.deps.store.recordOperationOutcome({ + callerKey: input.callerKey, + operationId: input.params.envelope.clientOperationId, + outcome: { status: 'failed', code: 'agent_session_checkpoint_stale' } + }) + input.operationGuard.finish(input.record.sessionId, input.params.envelope.clientOperationId) + return { + decision: 'refused', + refusal: { + code: 'agent_session_checkpoint_stale', + message: 'The session owner changed before the handoff request arrived.', + currentFence: input.record.lease.runtimeFence + } + } + } + return { decision: 'continue', record: input.record, fingerprint } +} + +export function replayedStructuredHandoffRefusal( + outcome: AgentSessionOperationOutcome +): AgentSessionWireRefusal | null { + if ( + outcome.status !== 'failed' || + !AGENT_SESSION_WIRE_REFUSAL_CODES.includes( + outcome.code as (typeof AGENT_SESSION_WIRE_REFUSAL_CODES)[number] + ) + ) { + return null + } + return { + code: outcome.code as (typeof AGENT_SESSION_WIRE_REFUSAL_CODES)[number], + message: 'This handoff request was previously refused.' + } +} + +export async function refuseAdmittedStructuredHandoff(input: { + deps: StructuredAgentSessionHandoffDeps + callerKey: string + params: AgentSessionHandoffRequest + refusal: AgentSessionWireRefusal +}): Promise> { + await input.deps.store.recordOperationOutcome({ + callerKey: input.callerKey, + operationId: input.params.envelope.clientOperationId, + outcome: { status: 'failed', code: input.refusal.code } + }) + return { ok: false, refusal: input.refusal } +} + +export function structuredHandoffRetryIsAdmissible( + status: AgentSessionHandoffStatus, + params: AgentSessionHandoffRequest +): boolean { + return ( + status.phase === 'failed' && + status.direction === params.direction && + status.operationId === params.envelope.clientOperationId && + status.error?.recoverableOwner !== 'none' + ) +} + +export function structuredHandoffRetryResumesStoppedOwner( + record: AgentSessionRecord, + params: AgentSessionHandoffRequest +): boolean { + return ( + record.lease.claimStatus === 'released' && + record.lease.handoffStage === 'old-owner-stopped' && + record.lease.handoffOperationId === params.envelope.clientOperationId + ) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.test.ts new file mode 100644 index 00000000000..20402c8c05e --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.test.ts @@ -0,0 +1,130 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { StructuredAgentSessionHandoffFlowRunner } from './structured-agent-session-handoff-flow-runner' +import { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import type { StructuredAgentSessionHandoffFlowContext } from './structured-agent-session-handoff-types' + +const NOW = 1_800_000_000_000 +const SESSION = 'session-flow-runner-outcome-write-failure' +const THREAD = '019fd532-7c11-7a90-b6de-4e1a2c3d5f61' +const OPERATION = `${NOW}-00000000000000000000000000000002` +const roots: string[] = [] + +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +const fields = { direction: 'to-native' as const, mode: 'now' as const, action: 'retry' as const } + +const params: AgentSessionHandoffRequest = { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION, + expectedRuntimeFence: null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff', + sessionId: SESSION, + fields + }) + }, + ...fields +} + +/** A runner whose scheduled flow always rejects, so every case here exercises the failure path. */ +async function failingFlowRunner( + fail: (params: AgentSessionHandoffRequest, error: unknown) => void +): Promise<{ + runner: StructuredAgentSessionHandoffFlowRunner + store: AgentSessionRecordStore + root: string +}> { + const root = await mkdtemp(join(tmpdir(), 'orca-handoff-flow-runner-')) + roots.push(root) + const store = await AgentSessionRecordStore.open({ + directory: join(root, 'store'), + hostId: 'local' + }) + const journal = await openAgentSessionJournal({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: THREAD } + }, + journalDir: join(root, 'journal') + }) + const runner = new StructuredAgentSessionHandoffFlowRunner({ + deps: { + store, + claimKeyId: 'key-1', + session: () => ({ journal, fence: 1 }), + suspendNative: async () => ({ state: 'stopped' as const }), + acquireNative: async () => { + throw new Error('unused') + }, + importTuiHistory: async () => {}, + publish: () => {}, + schedule: async () => { + throw new Error('scheduling failed') + }, + now: () => NOW + }, + operationGuard: new StructuredAgentSessionHandoffOperationGuard(store), + flowContext: (): StructuredAgentSessionHandoffFlowContext => { + throw new Error('unreachable: scheduling rejects before the flow needs context') + }, + fail + }) + return { runner, store, root } +} + +describe('structured handoff flow runner outcome-write failure', () => { + it('still reports the flow failure when the failed-outcome ledger write throws', async () => { + const failures: unknown[] = [] + const { runner, store, root } = await failingFlowRunner( + (_params, error) => void failures.push(error) + ) + // Materialize the store file so its later disappearance reads as corruption, making every + // subsequent ledger write reject. + await store.admitOperation({ + callerKey: 'seed', + operationId: `${NOW}-00000000000000000000000000000009`, + fingerprint: 'seed', + now: NOW + }) + await rm(join(root, 'store'), { recursive: true, force: true }) + + runner.begin({ callerKey: 'client-1', params, turnId: null, fingerprint: 'fp' }) + await runner.drain() + + expect(failures).toHaveLength(1) + expect((failures[0] as Error).message).toBe('scheduling failed') + }) + + it('does not leak an unhandled rejection when the failure notification itself throws', async () => { + // The host's status publish threw exactly here once eviction had dropped the session. + const { runner } = await failingFlowRunner(() => { + throw new Error('agent_session_ownership_unknown') + }) + const leaked: unknown[] = [] + const observe = (reason: unknown): void => void leaked.push(reason) + process.on('unhandledRejection', observe) + try { + runner.begin({ callerKey: 'client-1', params, turnId: null, fingerprint: 'fp' }) + await runner.drain() + // Node reports an orphaned rejection on the tick after it settles. + await new Promise((resolve) => setImmediate(resolve)) + } finally { + process.off('unhandledRejection', observe) + } + + expect(leaked).toEqual([]) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.ts new file mode 100644 index 00000000000..4259fc61725 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.ts @@ -0,0 +1,114 @@ +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { stopStructuredNativeTurn } from './structured-agent-session-handoff-flow-context' +import { handoffStructuredSessionToTui } from './structured-agent-session-handoff-forward' +import type { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import { assertScheduledStructuredHandoffIsAdmissible } from './structured-agent-session-handoff-revalidation' +import { handoffStructuredSessionToNative } from './structured-agent-session-handoff-reverse' +import { structuredTuiStatus } from './structured-agent-session-handoff-status' +import type { + StructuredAgentSessionHandoffDeps, + StructuredAgentSessionHandoffFlowContext +} from './structured-agent-session-handoff-types' + +export class StructuredAgentSessionHandoffFlowRunner { + private readonly active = new Set>() + + constructor( + private readonly input: { + deps: StructuredAgentSessionHandoffDeps + operationGuard: StructuredAgentSessionHandoffOperationGuard + flowContext: () => StructuredAgentSessionHandoffFlowContext + fail: (params: AgentSessionHandoffRequest, error: unknown) => void + } + ) {} + + async drain(): Promise { + await Promise.allSettled(this.active) + } + + track(task: Promise): void { + this.active.add(task) + // Settle-only bookkeeping. `.finally` forwards a rejection onto a promise nobody awaits, so a + // failure notification that threw escaped as an unhandled rejection even though `drain` — the + // one consumer — settles the flow through `allSettled`. + const forget = (): void => void this.active.delete(task) + void task.then(forget, forget) + } + + begin(input: { + callerKey: string + params: AgentSessionHandoffRequest + turnId: string | null + fingerprint: string + tuiAlreadyExited?: boolean + }): void { + const { callerKey, params, turnId, fingerprint, tuiAlreadyExited = false } = input + const sessionId = params.envelope.sessionId + const journalSequence = this.input.deps.session(sessionId).journal.cursor().sequence + this.input.operationGuard.start(sessionId, { + callerKey, + operationId: params.envelope.clientOperationId, + fingerprint + }) + const flow = this.run(params, turnId, tuiAlreadyExited, journalSequence) + .then(() => { + this.input.operationGuard.finish(sessionId, params.envelope.clientOperationId) + return this.input.deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'succeeded', sessionId } + }) + }) + .catch(async (error) => { + try { + await this.input.deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'failed', code: 'agent_session_handoff_failed' } + }) + } catch { + // Best-effort: a store write failure must not suppress the client's failure + // notification or leak the flow as an unhandled rejection. + } + this.input.operationGuard.finish(sessionId, params.envelope.clientOperationId) + this.input.fail(params, error) + }) + .finally(() => this.input.operationGuard.finish(sessionId, params.envelope.clientOperationId)) + this.track(flow) + } + + private run( + params: AgentSessionHandoffRequest, + turnId: string | null, + tuiAlreadyExited: boolean, + journalSequence: number + ): Promise { + const sessionId = params.envelope.sessionId + return this.input.deps.schedule(sessionId, async () => { + const context = this.input.flowContext() + assertScheduledStructuredHandoffIsAdmissible({ + record: context.requireRecord(sessionId), + journal: this.input.deps.session(sessionId).journal, + params, + turnId, + journalSequence, + tuiAlreadyExited, + tuiStatus: structuredTuiStatus(context.owner(sessionId), this.input.deps.transport) + }) + if (turnId && params.mode === 'stop-turn') { + const stopped = await stopStructuredNativeTurn(this.input.deps, sessionId, turnId) + if (!stopped) { + throw new Error('The current turn did not acknowledge cancellation.') + } + } + await (params.direction === 'to-tui' + ? handoffStructuredSessionToTui(context, params, params.action === 'retry') + : handoffStructuredSessionToNative( + context, + params, + params.action === 'retry', + tuiAlreadyExited + )) + }) + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.test.ts new file mode 100644 index 00000000000..7c77806ad91 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.test.ts @@ -0,0 +1,221 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + agentSessionLeaseFixture, + agentSessionRecordFixture +} from '../../../shared/agent-session-record.test-fixture' +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffStatus +} from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { + queuedStructuredHandoffCanBegin, + StructuredAgentSessionHandoffQueue +} from './structured-agent-session-handoff-queue' +import { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import { assertScheduledStructuredHandoffIsAdmissible } from './structured-agent-session-handoff-revalidation' + +const NOW = 1_800_000_000_000 +const SESSION = 'session-alpha-1' +const OPERATION_A = `${NOW}-00000000000000000000000000000001` +const OPERATION_B = `${NOW}-00000000000000000000000000000002` + +let root: string | null = null + +afterEach(async () => { + if (root) { + await rm(root, { recursive: true, force: true }) + root = null + } +}) + +async function createGuard() { + root = await mkdtemp(join(tmpdir(), 'orca-handoff-operation-guard-')) + const store = await AgentSessionRecordStore.open({ directory: root, hostId: 'local' }) + return { guard: new StructuredAgentSessionHandoffOperationGuard(store), store } +} + +function status(phase: 'switching' | 'queued' | 'idle'): AgentSessionHandoffStatus { + return { + owner: phase === 'idle' ? 'native' : 'none', + direction: phase === 'idle' ? null : 'to-tui', + phase, + stage: phase === 'switching' ? 'preparing' : null, + operationId: phase === 'idle' ? null : OPERATION_A + } +} + +describe('structured handoff operation ownership', () => { + it('reserves one winner across concurrent admissions', async () => { + const { guard } = await createGuard() + const check = (operationId: string) => + guard.check({ + callerKey: operationId, + sessionId: SESSION, + operationId, + fingerprint: operationId, + action: 'start', + now: NOW + }) + + const decisions = await Promise.all([check(OPERATION_A), check(OPERATION_B)]) + + expect(decisions.map(({ decision }) => decision).sort()).toEqual(['new', 'refused']) + }) + + it.each(['switching', 'queued'] as const)( + 'durably refuses a distinct operation while the %s operation owns the session', + async (phase) => { + const { guard } = await createGuard() + guard.start(SESSION, { + callerKey: 'client-a', + operationId: OPERATION_A, + fingerprint: 'fingerprint-a' + }) + + expect( + await guard.check({ + callerKey: 'client-b', + sessionId: SESSION, + operationId: OPERATION_B, + fingerprint: 'fingerprint-b', + action: 'start', + status: status(phase), + now: NOW + }) + ).toEqual({ decision: 'refused', code: 'agent_session_operation_conflict' }) + + guard.finish(SESSION, OPERATION_A) + expect( + await guard.check({ + callerKey: 'client-b', + sessionId: SESSION, + operationId: OPERATION_B, + fingerprint: 'fingerprint-b', + action: 'start', + status: status('idle'), + now: NOW + }) + ).toMatchObject({ + decision: 'replay', + outcome: { status: 'failed', code: 'agent_session_operation_conflict' } + }) + } + ) + + it('admits only cancellation beside a queued operation', async () => { + const { guard } = await createGuard() + guard.start(SESSION, { + callerKey: 'client-a', + operationId: OPERATION_A, + fingerprint: 'fingerprint-a' + }) + + await expect( + guard.check({ + callerKey: 'client-b', + sessionId: SESSION, + operationId: OPERATION_B, + fingerprint: 'fingerprint-b', + action: 'cancel-queued', + status: status('queued'), + now: NOW + }) + ).resolves.toEqual({ decision: 'new' }) + }) +}) + +describe('queued handoff fence revalidation', () => { + const params: AgentSessionHandoffRequest = { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION_A, + expectedRuntimeFence: 7, + payloadFingerprint: 'fingerprint' + }, + direction: 'to-tui', + mode: 'after-turn', + action: 'start' + } + const queued = status('queued') + + it('accepts the same live owner and fence', () => { + const record = agentSessionRecordFixture( + agentSessionLeaseFixture({ runtimeKind: 'native', ownerProcess: null }) + ) + expect(queuedStructuredHandoffCanBegin(record, queued, params)).toBe(true) + }) + + it.each([ + agentSessionLeaseFixture({ runtimeKind: 'native', runtimeFence: 8, ownerProcess: null }), + agentSessionLeaseFixture({ runtimeKind: 'tui' }), + agentSessionLeaseFixture({ + runtimeKind: 'native', + ownerProcess: null, + handoffStage: 'preparing' + }) + ])('refuses a changed durable owner or fence', (lease) => { + expect(queuedStructuredHandoffCanBegin(agentSessionRecordFixture(lease), queued, params)).toBe( + false + ) + }) + + it('cannot cancel after the idle waiter claims the queued operation', async () => { + const queue = new StructuredAgentSessionHandoffQueue() + const ready = vi.fn() + queue.enqueue(SESSION, () => true, ready) + await vi.waitFor(() => expect(ready).toHaveBeenCalledOnce()) + expect(queue.cancel(SESSION)).toBe(false) + }) +}) + +describe('scheduled handoff revalidation', () => { + it('refuses a native turn accepted ahead of the scheduled handoff', async () => { + root = await mkdtemp(join(tmpdir(), 'orca-handoff-revalidation-')) + const journal = await openAgentSessionJournal({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: SESSION, leafUuid: null } + }, + journalDir: join(root, 'journal') + }) + const journalSequence = journal.cursor().sequence + await journal.appendItem( + { provider: 'orca', clientMessageId: 'turn-running' }, + { kind: 'status', text: 'running', turnLifecycle: { turnId: 'turn-1', state: 'running' } }, + { fence: 7 } + ) + const params: AgentSessionHandoffRequest = { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION_A, + expectedRuntimeFence: 7, + payloadFingerprint: 'fingerprint' + }, + direction: 'to-tui', + mode: 'now', + action: 'start' + } + + expect(() => + assertScheduledStructuredHandoffIsAdmissible({ + record: agentSessionRecordFixture( + agentSessionLeaseFixture({ runtimeKind: 'native', ownerProcess: null }) + ), + journal, + params, + turnId: null, + journalSequence, + tuiAlreadyExited: false, + tuiStatus: 'busy' + }) + ).toThrow('session changed') + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.ts new file mode 100644 index 00000000000..da8d2eaad3d --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.ts @@ -0,0 +1,125 @@ +import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' +import type { + AgentSessionOperationOutcome, + AgentSessionOperationRefusalCode +} from '../../../shared/agent-session-operation-ledger' +import type { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' + +type ActiveOperation = { callerKey: string; operationId: string; fingerprint: string } + +export type HandoffOperationDecision = + | { decision: 'new' } + | { decision: 'replay'; outcome: AgentSessionOperationOutcome } + | { decision: 'retry' } + | { decision: 'refused'; code: AgentSessionOperationRefusalCode } + +export class StructuredAgentSessionHandoffOperationGuard { + private readonly activeBySession = new Map() + + constructor(private readonly store: AgentSessionRecordStore) {} + + async check(input: { + callerKey: string + sessionId: string + operationId: string + fingerprint: string + action: 'start' | 'cancel-queued' | 'retry' | 'recover' + status?: AgentSessionHandoffStatus + now: number + }): Promise { + const ledger = await this.store.admitOperation({ + callerKey: input.callerKey, + operationId: input.operationId, + fingerprint: input.fingerprint, + now: input.now + }) + if (ledger.decision === 'refused') { + return { decision: 'refused', code: ledger.code } + } + const active = this.activeBySession.get(input.sessionId) + const queuedCancellation = + input.action === 'cancel-queued' && + input.status?.phase === 'queued' && + input.status.operationId === active?.operationId + const activeConflict = Boolean( + active && + ((active.operationId === input.operationId && + (active.fingerprint !== input.fingerprint || active.callerKey !== input.callerKey)) || + (active.operationId !== input.operationId && !queuedCancellation)) + ) + const queuedConflict = Boolean( + !active && + input.status?.phase === 'queued' && + input.status.operationId !== input.operationId && + input.action !== 'cancel-queued' + ) + if (activeConflict || queuedConflict) { + if (ledger.decision === 'admit') { + await this.store.recordOperationOutcome({ + callerKey: input.callerKey, + operationId: input.operationId, + outcome: { status: 'failed', code: 'agent_session_operation_conflict' } + }) + } + return { decision: 'refused', code: 'agent_session_operation_conflict' } + } + if (ledger.decision === 'admit') { + this.reserve(input) + return { decision: 'new' } + } + if (input.action === 'retry' && ledger.row.outcome.status === 'failed') { + await this.store.recordOperationOutcome({ + callerKey: input.callerKey, + operationId: input.operationId, + outcome: { status: 'pending' } + }) + this.reserve(input) + return { decision: 'retry' } + } + if ( + ledger.row.outcome.status === 'pending' && + !active && + input.status?.operationId !== input.operationId + ) { + this.reserve(input) + return { decision: 'new' } + } + return { decision: 'replay', outcome: ledger.row.outcome } + } + + start(sessionId: string, operation: ActiveOperation): void { + this.activeBySession.set(sessionId, operation) + } + + private reserve(input: { + action: 'start' | 'cancel-queued' | 'retry' | 'recover' + callerKey: string + sessionId: string + operationId: string + fingerprint: string + }): void { + if (input.action !== 'cancel-queued') { + this.start(input.sessionId, input) + } + } + + finish(sessionId: string, operationId: string): void { + if (this.activeBySession.get(sessionId)?.operationId === operationId) { + this.activeBySession.delete(sessionId) + } + } + + async settle( + sessionId: string, + operationId: string, + outcome: AgentSessionOperationOutcome + ): Promise { + const active = this.activeBySession.get(sessionId) + await this.store.recordOperationOutcome({ + ...(active?.operationId === operationId ? { callerKey: active.callerKey } : {}), + operationId, + outcome + }) + this.finish(sessionId, operationId) + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.test.ts new file mode 100644 index 00000000000..e5ee4f7ca9b --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.test.ts @@ -0,0 +1,286 @@ +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { + AgentSessionHandoffDirection, + AgentSessionHandoffRequest, + AgentSessionMutationEnvelope +} from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' +import { AgentSessionOptionRejectedError } from './structured-agent-session-option-error' +import { StructuredAgentSessionHost } from './structured-agent-session-host' +import { + HOST_TEST_NOW as NOW, + HOST_TEST_SESSION as SESSION, + HOST_TEST_THREAD as THREAD, + hostTestAttachParams, + hostTestMessage, + hostTestOperationId, + resetHostTestOperationIds +} from './structured-agent-session-host-test-data' +import type { + StructuredAgentSessionHandoffTransport, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +const CALLER = { callerKey: 'client-1' } +const DEFAULT_MODEL = 'gpt-default' +const PICKED_MODEL = 'gpt-picked' +const PICKED_EFFORT = 'medium' + +let root: string +let store: AgentSessionRecordStore +let host: StructuredAgentSessionHost +let acquire: Mock +let activeModel: string +let activeEffort: string | null +let transcriptPath: string +let optionFailure: Error | null +const dispatchedModels: string[] = [] +const launchedOptions: (Readonly> | undefined)[] = [] +const closedTuiOwners: StructuredTuiOwner[] = [] + +function envelope(method: string, fields: Record): AgentSessionMutationEnvelope { + return { + sessionId: SESSION, + clientOperationId: hostTestOperationId(), + expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method, + sessionId: SESSION, + fields + }) + } +} + +function handoff(direction: AgentSessionHandoffDirection): AgentSessionHandoffRequest { + const fields = { direction, mode: 'now' as const, action: 'start' as const } + return { envelope: envelope('agentSession.requestHandoff', fields), ...fields } +} + +function tuiOwner(fence: number, spawnToken: string): StructuredTuiOwner { + return { + terminal: { handle: 'term-tui', tabId: 'tab-tui', paneKey: 'pane-tui', ptyId: 'pty-tui' }, + process: { + hostId: 'local', + pid: 5200, + processStartTimeMs: NOW, + spawnToken + }, + link: { + linkId: `tui-link-${fence}`, + handle: { provider: 'codex', threadId: THREAD }, + origin: 'resumed', + mintedAtFence: fence, + observedAt: NOW + }, + transcriptPath + } +} + +function handoffTransport(): StructuredAgentSessionHandoffTransport { + return { + hostLabel: 'Test host', + launchTui: async ({ record, fence, spawnToken }) => { + launchedOptions.push(record.options) + return tuiOwner(fence, spawnToken) + }, + reproveTuiOwner: async ({ owner }) => owner, + recoverTuiOwner: async (record) => + tuiOwner( + record.lease.runtimeFence, + record.lease.ownerProcess?.spawnToken ?? record.lease.reservedSpawnToken ?? 'recovered' + ), + stopRecoveredOwner: async () => undefined, + closeTuiOwner: async (owner) => { + closedTuiOwners.push(owner) + return { transcriptPath: owner.transcriptPath } + }, + waitForTuiExit: async (owner) => ({ transcriptPath: owner.transcriptPath }), + waitForTuiIdleOrExit: async () => 'idle', + tuiStatus: () => 'idle' + } +} + +function adapter(): StructuredAgentSessionAdapter { + acquire = vi.fn(async ({ fence, spawnToken, options }) => { + activeModel = options?.model ?? DEFAULT_MODEL + activeEffort = options?.effort ?? null + return { + process: { + hostId: 'local', + pid: 4200 + acquire.mock.calls.length, + processStartTimeMs: NOW, + spawnToken + }, + link: { + linkId: `native-link-${fence}`, + handle: { provider: 'codex', threadId: THREAD }, + origin: acquire.mock.calls.length === 1 ? 'created' : 'resumed', + mintedAtFence: fence, + observedAt: NOW + } + } + }) + return { + acquire, + dispatch: vi.fn(async () => { + dispatchedModels.push(activeModel) + return { + state: 'accepted', + providerIdentity: { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal: 1 } + } + }), + cancelTurn: vi.fn(async () => ({ cancelled: true })), + answerPrompt: vi.fn(async () => undefined), + setOption: vi.fn(async ({ key, value }) => { + if (optionFailure) { + const error = optionFailure + optionFailure = null + throw error + } + if (key === 'model') { + activeModel = value + } else if (key === 'effort') { + activeEffort = value + } + return { + model: activeModel, + ...(activeEffort ? { effort: activeEffort } : {}) + } + }), + readOptions: vi.fn(async () => ({ + current: { model: activeModel, ...(activeEffort ? { effort: activeEffort } : {}) }, + models: [] + })), + closeSession: vi.fn(async () => { + activeModel = DEFAULT_MODEL + return true + }) + } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-handoff-options-')) + resetHostTestOperationIds() + activeModel = DEFAULT_MODEL + activeEffort = null + optionFailure = null + dispatchedModels.length = 0 + launchedOptions.length = 0 + closedTuiOwners.length = 0 + const accountHome = join(root, 'codex-home') + const sessionsDir = join(accountHome, 'sessions', '2026', '08', '12') + transcriptPath = join(sessionsDir, `rollout-2026-08-12T10-00-00-${THREAD}.jsonl`) + await mkdir(sessionsDir, { recursive: true }) + await writeFile( + transcriptPath, + `${JSON.stringify({ + type: 'session_meta', + timestamp: '2026-08-12T10:00:00.000Z', + payload: { id: THREAD, session_id: THREAD } + })}\n`, + 'utf8' + ) + store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) + host = new StructuredAgentSessionHost({ + store, + adapter: adapter(), + journalRoot: root, + claimKeyId: 'key-1', + mintSpawnToken: () => 'spawn-native', + handoffTransport: handoffTransport(), + now: () => NOW + }) + const attached = await host.attach( + CALLER, + hostTestAttachParams(null, { accountHome: { variable: 'CODEX_HOME', path: accountHome } }) + ) + expect(attached).toMatchObject({ ok: true }) +}) + +afterEach(async () => { + await host.flushAllStreamedEvents() + await rm(root, { recursive: true, force: true }) +}) + +describe('structured session handoff options', () => { + it('settles a pre-mutation rejection so a fresh retry can succeed', async () => { + optionFailure = new AgentSessionOptionRejectedError('model list unavailable') + const fields = { key: 'model', value: PICKED_MODEL } + const rejected = { + envelope: envelope('agentSession.setOption', fields), + ...fields + } + + expect(await host.setOption(CALLER, rejected)).toMatchObject({ + ok: false, + refusal: { code: 'agent_session_operation_invalid', message: 'model list unavailable' } + }) + expect(await host.setOption(CALLER, rejected)).toMatchObject({ + ok: false, + refusal: { code: 'agent_session_operation_invalid' } + }) + expect( + await host.setOption(CALLER, { + envelope: envelope('agentSession.setOption', fields), + ...fields + }) + ).toMatchObject({ ok: true, value: { options: { model: PICKED_MODEL } } }) + expect(store.getRecord(SESSION)?.options).toEqual({ model: PICKED_MODEL }) + }) + + it('keeps a picked model through a native to TUI to native round trip', async () => { + const optionFields = { key: 'model', value: PICKED_MODEL } + expect( + await host.setOption(CALLER, { + envelope: envelope('agentSession.setOption', optionFields), + ...optionFields + }) + ).toMatchObject({ ok: true }) + expect(store.getRecord(SESSION)?.options).toEqual({ model: PICKED_MODEL }) + + const effortFields = { key: 'effort', value: PICKED_EFFORT } + expect( + await host.setOption(CALLER, { + envelope: envelope('agentSession.setOption', effortFields), + ...effortFields + }) + ).toMatchObject({ ok: true }) + expect(store.getRecord(SESSION)?.options).toEqual({ + model: PICKED_MODEL, + effort: PICKED_EFFORT + }) + + expect(await host.requestHandoff(CALLER, handoff('to-tui'))).toMatchObject({ ok: true }) + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'tui' }) + ) + expect(await host.requestHandoff(CALLER, handoff('to-native'))).toMatchObject({ ok: true }) + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'native' }) + ) + + expect(launchedOptions).toEqual([{ model: PICKED_MODEL, effort: PICKED_EFFORT }]) + expect(closedTuiOwners).toHaveLength(1) + expect(acquire.mock.calls[1]?.[0].options).toEqual({ + model: PICKED_MODEL, + effort: PICKED_EFFORT + }) + expect(store.getRecord(SESSION)?.options).toEqual({ + model: PICKED_MODEL, + effort: PICKED_EFFORT + }) + const body = hostTestMessage('use the selected model') + expect( + await host.send(CALLER, { + envelope: envelope('agentSession.send', { body }), + body + }) + ).toMatchObject({ ok: true }) + expect(dispatchedModels).toEqual([PICKED_MODEL]) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.ts new file mode 100644 index 00000000000..a5afd9891a1 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.ts @@ -0,0 +1,23 @@ +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' + +export async function readNativeHandoffSessionOptions(input: { + adapter: Pick + sessionId: string + fence: number + priorOptions?: Readonly> +}): Promise> | undefined> { + const { adapter, sessionId, fence, priorOptions } = input + const reported = await adapter.readOptions?.({ + sessionId, + fence + }) + if (!reported) { + return undefined + } + const { model: _model, effort: _effort, ...restored } = priorOptions ?? {} + return { + ...restored, + model: reported.current.model, + ...(reported.current.effort ? { effort: reported.current.effort } : {}) + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue-start.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue-start.ts new file mode 100644 index 00000000000..f8d6db2bace --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue-start.ts @@ -0,0 +1,46 @@ +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { activeStructuredAgentSessionTurnId } from '../../../shared/structured-agent-session-projection' +import type { StructuredAgentSessionHandoffQueue } from './structured-agent-session-handoff-queue' +import type { + StructuredAgentSessionHandoffDeps, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +export function queueStructuredHandoffAfterTurn(input: { + callerKey: string + params: AgentSessionHandoffRequest + deps: StructuredAgentSessionHandoffDeps + queue: StructuredAgentSessionHandoffQueue + owner: (sessionId: string) => StructuredTuiOwner | undefined + setStatus: ( + sessionId: string, + status: Parameters[1] + ) => void + begin: (callerKey: string, params: AgentSessionHandoffRequest, tuiAlreadyExited?: boolean) => void +}): void { + const { callerKey, params, deps, queue, owner, setStatus, begin } = input + const sessionId = params.envelope.sessionId + let tuiReadiness: 'idle' | 'exited' | null = null + setStatus(sessionId, { + owner: params.direction === 'to-tui' ? 'native' : 'tui', + direction: params.direction, + phase: 'queued', + stage: null, + operationId: params.envelope.clientOperationId, + hostLabel: deps.transport?.hostLabel + }) + const tuiOwner = owner(sessionId) + queue.enqueue( + sessionId, + async (signal) => { + if (params.direction === 'to-tui') { + return !activeStructuredAgentSessionTurnId(deps.session(sessionId).journal.snapshot().items) + } + tuiReadiness = tuiOwner + ? ((await deps.transport?.waitForTuiIdleOrExit(tuiOwner, signal)) ?? null) + : null + return tuiReadiness !== null + }, + () => begin(callerKey, { ...params, mode: 'now' }, tuiReadiness === 'exited') + ) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue.ts new file mode 100644 index 00000000000..9ea3d7ff08f --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue.ts @@ -0,0 +1,133 @@ +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffStatus +} from '../../../shared/agent-session-wire' +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import { activeStructuredAgentSessionTurnId } from '../../../shared/structured-agent-session-projection' +import type { + StructuredAgentSessionHandoffDeps, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +export class StructuredAgentSessionHandoffQueue { + private readonly controllers = new Map() + + cancel(sessionId: string): boolean { + const controller = this.controllers.get(sessionId) + controller?.abort() + this.controllers.delete(sessionId) + return controller !== undefined + } + + enqueue( + sessionId: string, + isIdle: (signal: AbortSignal) => boolean | Promise, + onReady: () => void + ): void { + this.cancel(sessionId) + const controller = new AbortController() + this.controllers.set(sessionId, controller) + void this.waitUntilIdle(sessionId, controller, isIdle).then((ready) => { + if (ready) { + onReady() + } + }) + } + + private async waitUntilIdle( + sessionId: string, + controller: AbortController, + isIdle: (signal: AbortSignal) => boolean | Promise + ): Promise { + while (this.controllers.get(sessionId) === controller && !controller.signal.aborted) { + try { + if (await isIdle(controller.signal)) { + this.controllers.delete(sessionId) + return true + } + } catch { + if (controller.signal.aborted) { + return false + } + } + await new Promise((resolve) => setTimeout(resolve, 150)) + } + return false + } +} + +export function queuedStructuredHandoffCanBegin( + record: AgentSessionRecord, + status: AgentSessionHandoffStatus, + params: AgentSessionHandoffRequest +): boolean { + const expectedOwner = params.direction === 'to-tui' ? 'native' : 'tui' + return ( + record.sessionId === params.envelope.sessionId && + status.phase === 'queued' && + status.direction === params.direction && + status.operationId === params.envelope.clientOperationId && + record.lease.runtimeFence === params.envelope.expectedRuntimeFence && + record.lease.runtimeKind === expectedOwner && + record.lease.claimStatus === 'live' && + record.lease.handoffStage === null && + !record.lease.unreconciled + ) +} + +export function enqueueStructuredHandoffAfterTurn(input: { + deps: StructuredAgentSessionHandoffDeps + queue: StructuredAgentSessionHandoffQueue + params: AgentSessionHandoffRequest + tuiOwner: StructuredTuiOwner | undefined + status: () => AgentSessionHandoffStatus + requireRecord: () => AgentSessionRecord + setStatus: (status: AgentSessionHandoffStatus) => void + begin: (params: AgentSessionHandoffRequest, tuiAlreadyExited: boolean) => void + refuse: (record: AgentSessionRecord) => void +}): void { + const { deps, params, queue, tuiOwner } = input + const sessionId = params.envelope.sessionId + let tuiReadiness: 'idle' | 'exited' | null = null + let observedTuiQueue = false + input.setStatus({ + owner: params.direction === 'to-tui' ? 'native' : 'tui', + direction: params.direction, + phase: 'queued', + stage: null, + operationId: params.envelope.clientOperationId, + hostLabel: deps.transport?.hostLabel + }) + queue.enqueue( + sessionId, + async (signal) => { + if (params.direction === 'to-tui') { + return !activeStructuredAgentSessionTurnId(deps.session(sessionId).journal.snapshot().items) + } + if (!observedTuiQueue) { + observedTuiQueue = true + return false + } + tuiReadiness = tuiOwner + ? ((await deps.transport?.waitForTuiIdleOrExit(tuiOwner, signal)) ?? null) + : null + if (tuiReadiness === 'exited') { + return true + } + if (!activeStructuredAgentSessionTurnId(deps.session(sessionId).journal.snapshot().items)) { + tuiReadiness = 'idle' + return true + } + return false + }, + () => { + const record = input.requireRecord() + const status = input.status() + if (!queuedStructuredHandoffCanBegin(record, status, params)) { + input.refuse(record) + return + } + input.begin(params, tuiReadiness === 'exited') + } + ) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-recover.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-recover.ts new file mode 100644 index 00000000000..0aab4f0335b --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-recover.ts @@ -0,0 +1,30 @@ +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffStatus +} from '../../../shared/agent-session-wire' +import { + beginStructuredManualRecovery, + structuredManualRecoveryIsAdmissible +} from './structured-agent-session-manual-recovery' +import type { StructuredAgentSessionHandoffDeps } from './structured-agent-session-handoff-types' +import type { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import type { AgentSessionRecord } from '../../../shared/agent-session-record' + +export async function requestStructuredManualRecovery(input: { + deps: StructuredAgentSessionHandoffDeps + operationGuard: StructuredAgentSessionHandoffOperationGuard + callerKey: string + params: AgentSessionHandoffRequest + fingerprint: string + record: AgentSessionRecord + status: AgentSessionHandoffStatus + requireRecord: (sessionId: string) => AgentSessionRecord + restore: (sessionId: string) => Promise + setStatus: (sessionId: string, status: AgentSessionHandoffStatus) => void +}): Promise { + if (!structuredManualRecoveryIsAdmissible(input.record, input.status)) { + return false + } + beginStructuredManualRecovery(input) + return true +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-result.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-result.ts new file mode 100644 index 00000000000..fe480d6359c --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-result.ts @@ -0,0 +1,32 @@ +import type { + AgentSessionHandoffResult, + AgentSessionMutationResult, + AgentSessionWireRefusal +} from '../../../shared/agent-session-wire' +import type { StructuredAgentSessionHandoffDeps } from './structured-agent-session-handoff-types' + +export function structuredHandoffRefusal( + code: AgentSessionWireRefusal['code'], + message: string +): AgentSessionWireRefusal { + return { code, message } +} + +export function structuredHandoffSuccess( + deps: StructuredAgentSessionHandoffDeps, + sessionId: string, + replayed: boolean, + status: AgentSessionHandoffResult['status'] +): AgentSessionMutationResult { + const record = deps.store.getRecord(sessionId) + if (!record) { + throw new Error('agent_session_identity_required') + } + return { + ok: true, + replayed, + fence: record.lease.runtimeFence, + cursor: deps.session(sessionId).journal.cursor(), + value: { status } + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-revalidation.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-revalidation.ts new file mode 100644 index 00000000000..66005959378 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-revalidation.ts @@ -0,0 +1,52 @@ +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { activeStructuredAgentSessionTurnId } from '../../../shared/structured-agent-session-projection' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { structuredHandoffRetryResumesStoppedOwner } from './structured-agent-session-handoff-admission' +import { structuredSessionHasPendingPrompt } from './structured-agent-session-handoff-status' + +export function assertScheduledStructuredHandoffIsAdmissible(input: { + record: AgentSessionRecord + journal: AgentSessionJournal + params: AgentSessionHandoffRequest + turnId: string | null + journalSequence: number + tuiAlreadyExited: boolean + tuiStatus: 'idle' | 'busy' +}): void { + const { params, record } = input + if (params.action === 'retry' && structuredHandoffRetryResumesStoppedOwner(record, params)) { + return + } + const expectedOwner = params.direction === 'to-tui' ? 'native' : 'tui' + if ( + record.lease.runtimeFence !== params.envelope.expectedRuntimeFence || + record.lease.runtimeKind !== expectedOwner || + record.lease.claimStatus !== 'live' || + record.lease.handoffStage !== null || + record.lease.unreconciled + ) { + throw new Error('agent_session_checkpoint_stale') + } + if (structuredSessionHasPendingPrompt(input.journal)) { + throw new Error('Resolve the pending question or approval before switching.') + } + if (params.mode !== 'stop-turn' && input.journal.cursor().sequence !== input.journalSequence) { + throw new Error('The session changed before the handoff started.') + } + const activeTurn = activeStructuredAgentSessionTurnId(input.journal.snapshot().items) + if (params.direction === 'to-tui') { + const expectedTurn = params.mode === 'stop-turn' ? input.turnId : null + if (activeTurn !== expectedTurn) { + throw new Error('The native turn changed before the handoff started.') + } + return + } + if ( + !input.tuiAlreadyExited && + input.tuiStatus !== 'idle' && + (params.mode !== 'after-turn' || activeTurn !== null) + ) { + throw new Error('The agent terminal became busy before the handoff started.') + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.test.ts new file mode 100644 index 00000000000..91c6163bd17 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.test.ts @@ -0,0 +1,100 @@ +import { describe, expect, it, vi } from 'vitest' +import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import { handoffStructuredSessionToNative } from './structured-agent-session-handoff-reverse' +import type { StructuredAgentSessionHandoffFlowContext } from './structured-agent-session-handoff-types' + +const OPERATION_ID = 'operation-1' +const SESSION_ID = 'session-1' + +vi.mock('../../runtime/agent-session-handoff-record-transitions', () => ({ + abandonStoredAgentSessionHandoffAttempt: vi.fn(async () => undefined), + reserveStoredAgentSessionHandoffOwner: vi.fn(async () => record()), + rollbackStoredAgentSessionHandoffPreparation: vi.fn(async () => undefined), + stopStoredAgentSessionOwnerForHandoff: vi.fn(async () => record()) +})) + +function record(): AgentSessionRecord { + return { + sessionId: SESSION_ID, + provider: 'claude', + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }, + lease: { + runtimeFence: 3, + handoffStage: 'old-owner-stopped', + handoffOperationId: OPERATION_ID + } + } as unknown as AgentSessionRecord +} + +function contextWith( + revealNativeSession: () => Promise, + statuses: AgentSessionHandoffStatus[] +): StructuredAgentSessionHandoffFlowContext { + return { + deps: { + store: {} as never, + claimKeyId: 'key-1', + now: () => 1_800_000_000_000, + importTuiHistory: vi.fn(async () => undefined), + acquireNative: vi.fn(async () => record()), + transport: { revealNativeSession } + } as never, + owner: () => undefined, + retainOwner: vi.fn(), + releaseOwner: vi.fn(), + setStatus: (_sessionId, status) => statuses.push(status), + enterPreparing: vi.fn(async () => undefined), + publishStage: vi.fn(), + requireRecord: () => record() + } +} + +// Why this ordering matters: releaseOwner has already run by the time the reveal fires, +// so a reveal that rejects before the status flip leaves the session released but never +// marked native — a stuck chat with no owner on either side. +describe('handoffStructuredSessionToNative', () => { + it('marks the session native before revealing it', async () => { + const statuses: AgentSessionHandoffStatus[] = [] + const order: string[] = [] + const context = contextWith(async () => { + order.push('reveal') + }, statuses) + const setStatus = context.setStatus + context.setStatus = (sessionId, status) => { + order.push('status') + setStatus(sessionId, status) + } + + await handoffStructuredSessionToNative( + context, + { envelope: { sessionId: SESSION_ID, clientOperationId: OPERATION_ID } } as never, + true + ) + + expect(order).toEqual(['status', 'reveal']) + expect(statuses.at(-1)).toMatchObject({ owner: 'native', direction: null, phase: 'idle' }) + }) + + it('still leaves the session marked native when the reveal rejects', async () => { + const statuses: AgentSessionHandoffStatus[] = [] + const context = contextWith(async () => { + throw new Error('publish failed') + }, statuses) + + await expect( + handoffStructuredSessionToNative( + context, + { envelope: { sessionId: SESSION_ID, clientOperationId: OPERATION_ID } } as never, + true + ) + ).rejects.toThrow('publish failed') + + expect(statuses.at(-1)).toMatchObject({ owner: 'native' }) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.ts index f59f7735245..ebfca81525c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.ts @@ -130,12 +130,8 @@ export async function handoffStructuredSessionToNative( throw error } context.releaseOwner(sessionId) - await deps.transport?.revealNativeSession?.({ - workspaceId: record.location.workspaceId, - sessionId, - agent: record.provider, - ...(owner?.adoptedTerminal ? { adoptedTerminal: true } : {}) - }) + // Why status lands before the reveal: the native owner is already proven here, and a + // reveal that rejects must not leave the session released but never marked native. context.setStatus(sessionId, { owner: 'native', direction: null, @@ -143,4 +139,10 @@ export async function handoffStructuredSessionToNative( stage: record.lease.handoffStage, operationId: record.lease.handoffOperationId }) + await deps.transport?.revealNativeSession?.({ + workspaceId: record.location.workspaceId, + sessionId, + agent: record.provider, + ...(owner?.adoptedTerminal ? { adoptedTerminal: true } : {}) + }) } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-coordinator.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-coordinator.ts new file mode 100644 index 00000000000..e5dd478f719 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-coordinator.ts @@ -0,0 +1,80 @@ +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' +import type { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { StructuredAgentSessionHandoffCoordinator } from './structured-agent-session-handoff' +import type { + StructuredAgentSessionHandoffTransport, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +type TestCoordinatorInput = { + store: AgentSessionRecordStore + journal: AgentSessionJournal + sessionId: string + provider: 'claude' | 'codex' + claudeSessionId: string + codexThreadId: string + now: number + launchTui: StructuredAgentSessionHandoffTransport['launchTui'] + reproveTuiOwner: StructuredAgentSessionHandoffTransport['reproveTuiOwner'] + stopRecoveredOwner: StructuredAgentSessionHandoffTransport['stopRecoveredOwner'] + closeTuiOwner: NonNullable + waitForTuiExit: StructuredAgentSessionHandoffTransport['waitForTuiExit'] + waitForTuiIdleOrExit: StructuredAgentSessionHandoffTransport['waitForTuiIdleOrExit'] + stopFailedTuiLaunch: NonNullable + recoverTuiOwner: (record: AgentSessionRecord) => Promise + tuiStatus: () => 'idle' | 'busy' + acquireNative: (input: { + sessionId: string + fence: number + spawnToken: string + }) => Promise + acquireNativeStop: (turnId: string) => Promise + takeImportFailure: () => Error | null + statuses: AgentSessionHandoffStatus[] +} + +export function createStructuredAgentSessionHandoffTestCoordinator( + input: TestCoordinatorInput +): StructuredAgentSessionHandoffCoordinator { + return new StructuredAgentSessionHandoffCoordinator({ + store: input.store, + claimKeyId: 'key-1', + transport: { + hostLabel: 'Test host', + launchTui: input.launchTui, + reproveTuiOwner: input.reproveTuiOwner, + recoverTuiOwner: input.recoverTuiOwner, + stopRecoveredOwner: input.stopRecoveredOwner, + closeTuiOwner: input.closeTuiOwner, + waitForTuiExit: input.waitForTuiExit, + waitForTuiIdleOrExit: input.waitForTuiIdleOrExit, + tuiStatus: input.tuiStatus, + stopFailedTuiLaunch: input.stopFailedTuiLaunch + }, + session: () => ({ + journal: input.journal, + fence: input.store.getRecord(input.sessionId)?.lease.runtimeFence ?? 1 + }), + suspendNative: async () => ({ state: 'stopped' as const }), + acquireNative: input.acquireNative, + acquireNativeStop: (_sessionId, turnId) => input.acquireNativeStop(turnId), + importTuiHistory: async ({ fence }) => { + const importFailure = input.takeImportFailure() + if (importFailure) { + throw importFailure + } + await input.journal.appendItem( + input.provider === 'claude' + ? { provider: 'claude', sessionId: input.claudeSessionId, uuid: 'tui-turn' } + : { provider: 'codex', threadId: input.codexThreadId, turnId: 'tui-turn', ordinal: 0 }, + { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'from tui' }] }, + { fence, recovered: true } + ) + }, + publish: (_sessionId, status) => input.statuses.push(status), + schedule: async (_sessionId, task) => task(), + now: () => input.now + }) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-identities.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-identities.ts new file mode 100644 index 00000000000..ca33e486999 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-identities.ts @@ -0,0 +1,40 @@ +export type StructuredHandoffProviderCase = { + provider: 'claude' | 'codex' + accountHome: { variable: 'CLAUDE_CONFIG_DIR' | 'CODEX_HOME'; pathName: string } +} + +export const STRUCTURED_HANDOFF_PROVIDER_CASES: StructuredHandoffProviderCase[] = [ + { provider: 'codex', accountHome: { variable: 'CODEX_HOME', pathName: 'codex-home' } }, + { + provider: 'claude', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', pathName: 'claude-home' } + } +] + +export function structuredHandoffTestProcess(now: number, spawnToken: string, pid: number) { + return { hostId: 'local', pid, processStartTimeMs: now - 1_000, spawnToken } +} + +export function structuredHandoffTestLink(input: { + provider: 'claude' | 'codex' + fence: number + id: string + now: number + claudeSessionId: string + codexThreadId: string +}) { + return { + linkId: input.id, + handle: + input.provider === 'claude' + ? ({ + provider: 'claude' as const, + sessionId: input.claudeSessionId, + leafUuid: input.id.startsWith('native-link') ? 'tui-exit-leaf' : 'current-leaf' + } as const) + : ({ provider: 'codex' as const, threadId: input.codexThreadId } as const), + origin: 'resumed' as const, + mintedAtFence: input.fence, + observedAt: input.now + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-requests.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-requests.ts new file mode 100644 index 00000000000..550ebded0da --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-requests.ts @@ -0,0 +1,53 @@ +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { + AgentSessionHandoffDirection, + AgentSessionHandoffAction, + AgentSessionHandoffMode, + AgentSessionHandoffRequest +} from '../../../shared/agent-session-wire' + +export type StructuredHandoffTestRequestOptions = { + action?: AgentSessionHandoffAction + operationId?: string +} + +export class StructuredHandoffTestRequests { + private operations = 0 + + constructor( + private readonly now: number, + private readonly sessionId: string, + private readonly readFence: () => number + ) {} + + reset(): void { + this.operations = 0 + } + + operationId(): string { + this.operations += 1 + return `${this.now}-${this.operations.toString(16).padStart(32, '0')}` + } + + request( + direction: AgentSessionHandoffDirection, + mode: AgentSessionHandoffMode, + options: StructuredHandoffTestRequestOptions = {} + ): AgentSessionHandoffRequest { + const action = options.action ?? 'start' + const fields = { direction, mode, action } + return { + envelope: { + sessionId: this.sessionId, + clientOperationId: options.operationId ?? this.operationId(), + expectedRuntimeFence: this.readFence(), + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff', + sessionId: this.sessionId, + fields + }) + }, + ...fields + } + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.ts index 0e213921609..57333d1c90c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.ts @@ -1,63 +1,286 @@ import type { AgentSessionRecord } from '../../../shared/agent-session-record' -import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffResult, + AgentSessionHandoffStatus, + AgentSessionMutationResult, + AgentSessionWireRefusal +} from '../../../shared/agent-session-wire' +import { activeStructuredAgentSessionTurnId } from '../../../shared/structured-agent-session-projection' +import { + admitStructuredHandoffRequest, + refuseAdmittedStructuredHandoff, + replayedStructuredHandoffRefusal, + structuredHandoffRetryIsAdmissible +} from './structured-agent-session-handoff-admission' import { createStructuredHandoffFlowContext, requireStructuredHandoffRecord } from './structured-agent-session-handoff-flow-context' -import { restoreStructuredAgentSessionHandoff } from './structured-agent-session-handoff-restart' +import { StructuredAgentSessionHandoffFlowRunner } from './structured-agent-session-handoff-flow-runner' +import { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import { StructuredAgentSessionHandoffQueue } from './structured-agent-session-handoff-queue' +import { queueStructuredHandoffAfterTurn } from './structured-agent-session-handoff-queue-start' import { closeRetainedTuiOwner } from './structured-agent-session-handoff-owner-close' +import { requestStructuredManualRecovery } from './structured-agent-session-handoff-recover' +import { restoreStructuredAgentSessionHandoff } from './structured-agent-session-handoff-restart' +import { + structuredHandoffRefusal as refusal, + structuredHandoffSuccess +} from './structured-agent-session-handoff-result' +import { + failedStructuredHandoffStatus, + idleStructuredHandoffStatus, + structuredSessionHasPendingPrompt, + structuredTuiStatus +} from './structured-agent-session-handoff-status' import type { StructuredAgentSessionHandoffDeps, StructuredAgentSessionHandoffFlowContext } from './structured-agent-session-handoff-types' import { StructuredAgentSessionHandoffState } from './structured-agent-session-handoff-state' - export class StructuredAgentSessionHandoffCoordinator { private readonly state: StructuredAgentSessionHandoffState - + private readonly queue = new StructuredAgentSessionHandoffQueue() + private readonly operationGuard: StructuredAgentSessionHandoffOperationGuard + private readonly flowRunner: StructuredAgentSessionHandoffFlowRunner constructor(private readonly deps: StructuredAgentSessionHandoffDeps) { - // oxfmt-ignore - this.state = new StructuredAgentSessionHandoffState({ requireRecord: (sessionId) => this.requireRecord(sessionId), publish: deps.publish, hostLabel: deps.transport?.hostLabel }) - } - - status = (sessionId: string) => this.state.status(sessionId) - - closeRetainedTuiOwner = (sessionId: string): Promise => - closeRetainedTuiOwner({ - sessionId, - deps: this.deps, - owner: this.state.owner, - requireRecord: this.requireRecord, - releaseOwner: this.state.releaseOwner + this.state = new StructuredAgentSessionHandoffState({ + requireRecord: (sessionId) => this.requireRecord(sessionId), + publish: deps.publish, + hostLabel: deps.transport?.hostLabel }) - + this.operationGuard = new StructuredAgentSessionHandoffOperationGuard(deps.store) + this.flowRunner = new StructuredAgentSessionHandoffFlowRunner({ + deps, + operationGuard: this.operationGuard, + flowContext: () => this.flowContext(), + fail: (params, error) => this.fail(params, error) + }) + } + status = (sessionId: string): AgentSessionHandoffStatus => this.state.status(sessionId) + drain = (): Promise => this.flowRunner.drain() + closeRetainedTuiOwner = (sessionId: string): Promise => + this.closeRetainedOwner(sessionId) setStatus = (sessionId: string, status: AgentSessionHandoffStatus): void => this.state.setStatus(sessionId, status) - + async request( + callerKey: string, + params: AgentSessionHandoffRequest + ): Promise> { + const record = this.requireRecord(params.envelope.sessionId) + const currentStatus = this.state.cachedStatus(record.sessionId) + const admission = await admitStructuredHandoffRequest({ + deps: this.deps, + operationGuard: this.operationGuard, + callerKey, + params, + record, + ...(currentStatus ? { status: currentStatus } : {}) + }) + if (admission.decision === 'replay') { + const replayedRefusal = replayedStructuredHandoffRefusal(admission.outcome) + if (replayedRefusal) { + return { ok: false, refusal: replayedRefusal } + } + return this.success(record.sessionId, true) + } + if (admission.decision === 'refused') { + return { ok: false, refusal: admission.refusal } + } + const { fingerprint } = admission + const action = params.action ?? 'start' + if (action === 'cancel-queued') { + if (currentStatus?.phase !== 'queued' || currentStatus?.direction !== params.direction) { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_operation_conflict', + 'No matching queued handoff exists.' + ) + } + this.queue.cancel(record.sessionId) + this.setStatus(record.sessionId, idleStructuredHandoffStatus(record)) + await this.deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'succeeded', sessionId: record.sessionId } + }) + return this.success(record.sessionId, false) + } + if (!this.deps.transport) { + return this.refuseAdmitted( + callerKey, + params, + 'structured_agent_session_unsupported', + 'Agent TUI handoff is unavailable on this host.' + ) + } + if (action === 'recover') { + const status = this.status(record.sessionId) + const started = await requestStructuredManualRecovery({ + deps: this.deps, + operationGuard: this.operationGuard, + callerKey, + params, + fingerprint, + record, + status, + requireRecord: this.requireRecord, + restore: this.restore, + setStatus: this.setStatus + }) + if (!started) { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_operation_conflict', + 'This handoff is no longer eligible for proof recovery.' + ) + } + return this.success(record.sessionId, false) + } + if (action === 'retry') { + if (!structuredHandoffRetryIsAdmissible(this.status(record.sessionId), params)) { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_operation_conflict', + 'This handoff is no longer retryable.' + ) + } + this.begin(callerKey, params, null, fingerprint) + return this.success(record.sessionId, false) + } + const expectedOwner = params.direction === 'to-tui' ? 'native' : 'tui' + if (record.lease.runtimeKind !== expectedOwner || record.lease.claimStatus !== 'live') { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_conflict', + `The ${expectedOwner} runtime does not own this session.` + ) + } + if (structuredSessionHasPendingPrompt(this.deps.session(record.sessionId).journal)) { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_conflict', + 'Resolve the pending question or approval before switching.' + ) + } + const turnId = activeStructuredAgentSessionTurnId( + this.deps.session(record.sessionId).journal.snapshot().items + ) + const tuiOwner = this.state.owner(record.sessionId) + const busy = + expectedOwner === 'native' + ? turnId !== null + : structuredTuiStatus(tuiOwner, this.deps.transport) !== 'idle' + if (busy && params.mode === 'now') { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_conflict', + 'The current turn must finish before switching.' + ) + } + if (busy && params.mode === 'after-turn') { + queueStructuredHandoffAfterTurn({ + callerKey, + params, + deps: this.deps, + queue: this.queue, + owner: (sessionId) => this.state.owner(sessionId), + setStatus: this.setStatus, + begin: (key, next, tuiAlreadyExited) => + this.begin(key, next, null, fingerprint, tuiAlreadyExited) + }) + return this.success(record.sessionId, false) + } + if (busy && expectedOwner === 'tui' && params.mode === 'stop-turn') { + return this.refuseAdmitted( + callerKey, + params, + 'structured_agent_session_unsupported', + 'Exit the agent terminal after this turn to continue in chat.' + ) + } + this.begin(callerKey, params, turnId, fingerprint) + return this.success(record.sessionId, false) + } async restore(sessionId: string): Promise { await restoreStructuredAgentSessionHandoff( { deps: this.deps, requireRecord: (id) => this.requireRecord(id), flowContext: () => this.flowContext(), - retainOwner: this.state.retainOwner, - setStatus: this.state.setStatus + retainOwner: (id, owner) => this.state.retainOwner(id, owner), + setStatus: (id, status) => this.state.setStatus(id, status) }, sessionId ) } - + private refuseAdmitted( + callerKey: string, + params: AgentSessionHandoffRequest, + code: AgentSessionWireRefusal['code'], + message: string + ): Promise> { + return refuseAdmittedStructuredHandoff({ + deps: this.deps, + callerKey, + params, + refusal: refusal(code, message) + }) + } + private success( + sessionId: string, + replayed: boolean + ): AgentSessionMutationResult { + return structuredHandoffSuccess(this.deps, sessionId, replayed, this.status(sessionId)) + } + private begin( + callerKey: string, + params: AgentSessionHandoffRequest, + turnId: string | null, + fingerprint: string, + tuiAlreadyExited = false + ): void { + this.flowRunner.begin({ + callerKey, + params, + turnId, + fingerprint, + tuiAlreadyExited + }) + } private flowContext(): StructuredAgentSessionHandoffFlowContext { return createStructuredHandoffFlowContext({ deps: this.deps, - owner: this.state.owner, - retainOwner: this.state.retainOwner, - releaseOwner: this.state.releaseOwner, - setStatus: this.state.setStatus, + owner: (sessionId) => this.state.owner(sessionId), + retainOwner: (sessionId, owner) => this.state.retainOwner(sessionId, owner), + releaseOwner: (sessionId) => this.state.releaseOwner(sessionId), + setStatus: (sessionId, status) => this.state.setStatus(sessionId, status), requireRecord: (sessionId) => this.requireRecord(sessionId) }) } - + private fail(params: AgentSessionHandoffRequest, error: unknown): void { + const record = this.requireRecord(params.envelope.sessionId) + this.setStatus( + record.sessionId, + failedStructuredHandoffStatus(record, params, error, this.deps.transport?.hostLabel) + ) + } + private closeRetainedOwner(sessionId: string): Promise { + return closeRetainedTuiOwner({ + sessionId, + deps: this.deps, + owner: this.state.owner, + requireRecord: this.requireRecord, + releaseOwner: this.state.releaseOwner + }) + } private requireRecord = (sessionId: string): AgentSessionRecord => requireStructuredHandoffRecord(this.deps, sessionId) } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.test.ts index 4c807d879b6..9bf27a11106 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.test.ts @@ -6,12 +6,14 @@ import type { AgentSessionExecutionLocation, AgentSessionRecord } from '../../../shared/agent-session-record' +import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' import { LOCAL_EXECUTION_HOST_ID } from '../../../shared/execution-host' import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import { createDeferredStructuredAgentSessionEventSink } from './structured-agent-session-event-sink' import { acquireNativeHandoffOwner, + createStructuredAgentSessionHostHandoff, structuredTuiTranscriptImportOptions } from './structured-agent-session-host-handoff' @@ -173,6 +175,7 @@ describe('native handoff acquisition', () => { }, { session: () => session, + findSession: () => session, eventSink: () => eventSink, flush: async () => undefined, serialize: async (_session, task) => task(), @@ -200,3 +203,91 @@ describe('native handoff acquisition', () => { expect(order).toEqual(['append-entered', 'append-complete', 'unbind', 'acquire']) }) }) + +describe('handoff status published for a session the host no longer holds', () => { + const sessionId = 'session-handoff-publish-detached' + const now = 1_800_000_000_000 + const failed: AgentSessionHandoffStatus = { + owner: 'native', + direction: 'to-tui', + phase: 'failed', + stage: null, + operationId: null + } + let root: string + let store: AgentSessionRecordStore + + beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-handoff-publish-')) + store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) + }) + + afterEach(async () => { + await rm(root, { recursive: true, force: true }) + }) + + function detachedHandoff(frames: { fence: number; status: AgentSessionHandoffStatus }[]) { + return createStructuredAgentSessionHostHandoff( + { store, adapter: {} as never, journalRoot: root, claimKeyId: 'key-1' }, + { + // Eviction and host teardown both drop the map entry while a flow is still settling. + session: () => { + throw new Error('agent_session_ownership_unknown') + }, + findSession: () => undefined, + eventSink: () => { + throw new Error('unreachable: publishing reads no sink') + }, + flush: async () => undefined, + serialize: async (_sessionId, task) => task(), + subscribers: { + publish: vi.fn(), + reset: vi.fn(), + snapshot: vi.fn(), + handoff: (_id: string, fence: number, status: AgentSessionHandoffStatus) => + void frames.push({ fence, status }) + } as never, + now: () => now + } + ) + } + + it('still reaches subscribers at the record fence instead of throwing', async () => { + const reserved = await store.reserveOwner({ + sessionId, + location: { + executionHostId: LOCAL_EXECUTION_HOST_ID, + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }, + provider: 'codex', + accountHome: { variable: 'CODEX_HOME', path: join(root, 'codex-home') }, + runtimeKind: 'native', + expectedFence: null, + spawnToken: 'detached-publish', + claimKeyId: 'key-1', + handoffOperationId: `${now}-00000000000000000000000000000001`, + probe: { outcome: 'reservation-unused' }, + operation: { + callerKey: 'test', + operationId: `${now}-00000000000000000000000000000002`, + fingerprint: 'handoff' + }, + now + }) + const frames: { fence: number; status: AgentSessionHandoffStatus }[] = [] + + expect(() => detachedHandoff(frames).setStatus(sessionId, failed)).not.toThrow() + + expect(frames).toEqual([{ fence: reserved.record.lease.runtimeFence, status: failed }]) + }) + + it('drops the publish when neither a session nor a record remains', () => { + const frames: { fence: number; status: AgentSessionHandoffStatus }[] = [] + + expect(() => detachedHandoff(frames).setStatus(sessionId, failed)).not.toThrow() + + expect(frames).toEqual([]) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.ts index 6fb12f9bd13..586df1476cf 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.ts @@ -17,6 +17,8 @@ import { StructuredTuiTranscriptCatchup } from './structured-tui-transcript-catc type HostHandoffAccess = { session: (sessionId: string) => StructuredAgentSessionHostSession + /** Non-throwing lookup, for the paths that only observe a detached session. */ + findSession: (sessionId: string) => StructuredAgentSessionHostSession | undefined eventSink: (sessionId: string) => DeferredStructuredAgentSessionEventSink flush: (sessionId: string) => Promise serialize: (sessionId: string, task: () => Promise) => Promise @@ -95,8 +97,14 @@ export function createStructuredAgentSessionHostHandoff( activateTuiHistoryCatchup: (sessionId) => tuiHistoryCatchup.activate(sessionId), stopTuiHistoryCatchup: (sessionId) => tuiHistoryCatchup.stop(sessionId), publish: (sessionId, status) => { - const session = host.session(sessionId) - const fence = deps.store.getRecord(sessionId)?.lease.runtimeFence ?? session.fence + // A status publish is a notification, not a mutation. Eviction and host teardown both drop + // the session while a handoff flow is still settling, and `requireSession` would turn that + // last publish — usually the FAILED one — into an unhandled rejection nothing can catch. + const fence = + deps.store.getRecord(sessionId)?.lease.runtimeFence ?? host.findSession(sessionId)?.fence + if (fence === undefined) { + return + } host.subscribers.handoff(sessionId, fence, status) }, schedule: host.serialize, diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts index c9afa06e525..7d2648930c4 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts @@ -74,7 +74,11 @@ export function sendStructuredAgentSessionTurn( export function cancelStructuredAgentSessionTurn( context: StructuredAgentSessionMutationContext, caller: StructuredAgentSessionCaller, - params: { envelope: AgentSessionMutationEnvelope; turnId: string } + params: { + envelope: AgentSessionMutationEnvelope + turnId: string + scope?: 'background-tasks' + } ): Promise> { return mutate(context, caller, params.envelope, cancelPlan(params)) } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts index 99d6c212c10..8989f5e4d72 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts @@ -6,6 +6,8 @@ import type { AgentSessionAttachResult, AgentSessionHistoryRequest, AgentSessionHistoryResult, + AgentSessionHandoffRequest, + AgentSessionHandoffResult, AgentSessionHandoffStatus, AgentSessionMutationResult, AgentSessionOptionsResult, @@ -55,10 +57,13 @@ import type { StructuredAgentSessionHostDeps, StructuredAgentSessionHostSession } from './structured-agent-session-host-types' -import { readStructuredAgentSessionHistoryResult } from './structured-agent-session-history-result' -import { retryPendingStructuredAgentSessionSettlement } from './structured-agent-session-settlement-retry' import { StructuredAgentSessionEventRecovery } from './structured-agent-session-event-recovery' +import { StructuredAgentSessionBackgroundTaskChannel } from './structured-agent-session-background-task-channel' +import { withTimeout } from '../../../shared/promise-timeout-fallback' export type { StructuredAgentSessionHostDeps } from './structured-agent-session-host-types' +/** Quit must not wait indefinitely on an in-flight handoff; see the drain phase below. */ +const HANDOFF_DRAIN_TIMEOUT_MS = 5_000 + export class StructuredAgentSessionHost { private readonly sessions = new Map() private readonly subscribers = new AgentSessionSubscribers() @@ -70,8 +75,16 @@ export class StructuredAgentSessionHost { private readonly restartRestore = new StructuredAgentSessionRestartRestoreGate() private readonly holds: StructuredAgentSessionHolds private readonly eventRecovery: StructuredAgentSessionEventRecovery + private readonly backgroundTasks: StructuredAgentSessionBackgroundTaskChannel constructor(readonly deps: StructuredAgentSessionHostDeps) { + this.backgroundTasks = new StructuredAgentSessionBackgroundTaskChannel( + deps, + this.sessions, + this.subscribers, + (sessionId) => this.requireSession(sessionId), + (sessionId) => this.handoffs.status(sessionId) + ) this.runtimeState = new StructuredAgentSessionHostRuntimeState( deps, (record) => this.restoreRenewedHandoff(record.sessionId), @@ -91,6 +104,7 @@ export class StructuredAgentSessionHost { }) this.handoffs = createStructuredAgentSessionHostHandoff(deps, { session: (sessionId) => this.requireSession(sessionId), + findSession: (sessionId) => this.sessions.get(sessionId), eventSink: (sessionId) => this.runtimeState.eventSinkFor(sessionId), flush: (sessionId) => this.flushStreamedEvents(sessionId), serialize: (sessionId, task) => this.serialize(sessionId, task), @@ -181,14 +195,6 @@ export class StructuredAgentSessionHost { subscribers: this.subscribers, tasks: this.tasks, reconcileLeases: (sessionId) => this.reconcileLeases(sessionId), - retryPendingSettlement: (sessionId, params) => - retryPendingStructuredAgentSessionSettlement({ - deps: this.deps, - sessions: this.sessions, - sessionId, - params, - now: () => this.now() - }), serialize: (sessionId, task) => this.serialize(sessionId, task), now: () => this.now() } @@ -257,6 +263,15 @@ export class StructuredAgentSessionHost { { name: 'dispose-holds', run: () => this.holds.dispose() }, { name: 'stop-lease-renewal', run: () => this.runtimeState.stopLeaseRenewal() }, { name: 'stop-tui-catchup', run: () => this.handoffs.stopTuiHistoryCatchup() }, + // Before the session map is dropped: a handoff flow left running writes rows into a + // journal this teardown is about to close, and publishes against a session it removed. + // Why bounded: this phase is on the app-quit path, and a flow wedged in `launchTui` would + // otherwise hold the quit open forever. Giving up merely restores the old orphaning, which + // the publish guard above already makes survivable. + { + name: 'drain-handoffs', + run: () => withTimeout(this.handoffs.drain(), HANDOFF_DRAIN_TIMEOUT_MS, undefined) + }, { name: 'drain-attaches', run: () => this.tasks.drainAttaches() }, { name: 'flush-event-sinks', run: () => this.runtimeState.flushAllEventSinks() } ], @@ -299,6 +314,12 @@ export class StructuredAgentSessionHost { ): ReturnType => setStructuredAgentSessionOption(this.mutationContext(), caller, params) + requestHandoff = ( + caller: StructuredAgentSessionCaller, + params: AgentSessionHandoffRequest + ): Promise> => + this.handoffs.request(caller.callerKey, params) + readOptions = (sessionId: string): Promise => readStructuredAgentSessionOptions(this.mutationContext(), sessionId) @@ -309,24 +330,16 @@ export class StructuredAgentSessionHost { ) } - history(request: AgentSessionHistoryRequest): AgentSessionHistoryResult { - return readStructuredAgentSessionHistoryResult({ - journal: this.requireSession(request.sessionId).journal, - record: this.deps.store.getRecord(request.sessionId), - request - }) - } + history = (request: AgentSessionHistoryRequest): AgentSessionHistoryResult => + this.backgroundTasks.history(request) - subscribe(input: AgentSessionSubscribeInput): () => void { - const session = this.requireSession(input.sessionId) - const fence = this.deps.store.getRecord(input.sessionId)?.lease.runtimeFence ?? 0 - return this.subscribers.open({ - ...input, - journal: session.journal, - fence, - handoff: this.handoffs.status(input.sessionId) - }) - } + subscribe = (input: AgentSessionSubscribeInput): (() => void) => + this.backgroundTasks.subscribe(input) + + publishBackgroundTaskState: StructuredAgentSessionBackgroundTaskChannel['publish'] = ( + sessionId, + state + ) => this.backgroundTasks.publish(sessionId, state) unsubscribe = (sessionId: string, id: string): void => this.subscribers.close(sessionId, id) private requireSession(sessionId: string): StructuredAgentSessionHostSession { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-manual-recovery.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-manual-recovery.ts new file mode 100644 index 00000000000..91be1a15aa5 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-manual-recovery.ts @@ -0,0 +1,103 @@ +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffStatus +} from '../../../shared/agent-session-wire' +import { setStoredAgentSessionHandoffStage } from '../../runtime/agent-session-handoff-record-transitions' +import type { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import { idleStructuredHandoffStatus } from './structured-agent-session-handoff-status' +import type { StructuredAgentSessionHandoffDeps } from './structured-agent-session-handoff-types' + +export function structuredManualRecoveryIsAdmissible( + record: AgentSessionRecord, + status: AgentSessionHandoffStatus | undefined +): boolean { + return ( + record.lease.handoffStage === 'manual-recovery' && + record.lease.runtimeKind === 'tui' && + record.lease.ownerProcess !== null && + status?.error?.canRetryProof === true + ) +} + +export function beginStructuredManualRecovery(input: { + deps: StructuredAgentSessionHandoffDeps + operationGuard: StructuredAgentSessionHandoffOperationGuard + callerKey: string + params: AgentSessionHandoffRequest + fingerprint: string + requireRecord: (sessionId: string) => AgentSessionRecord + restore: (sessionId: string) => Promise + setStatus: (sessionId: string, status: AgentSessionHandoffStatus) => void +}): Promise { + const { + callerKey, + deps, + fingerprint, + operationGuard, + params, + requireRecord, + restore, + setStatus + } = input + const sessionId = params.envelope.sessionId + operationGuard.start(sessionId, { + callerKey, + operationId: params.envelope.clientOperationId, + fingerprint + }) + setStatus(sessionId, { + owner: 'none', + direction: params.direction, + phase: 'switching', + stage: 'recovering', + operationId: params.envelope.clientOperationId, + hostLabel: deps.transport?.hostLabel + }) + return deps + .schedule(sessionId, async () => { + let record = requireRecord(sessionId) + if (record.lease.claimStatus === 'reserved' && record.lease.handoffOperationId !== null) { + record = await setStoredAgentSessionHandoffStage(deps.store, { + sessionId, + fence: record.lease.runtimeFence, + stage: 'new-owner-proving', + handoffOperationId: record.lease.handoffOperationId, + now: deps.now() + }) + } + await restore(record.sessionId) + if (requireRecord(sessionId).lease.handoffStage === 'manual-recovery') { + throw new Error('The TUI owner proof is still unavailable.') + } + }) + .then(() => { + operationGuard.finish(sessionId, params.envelope.clientOperationId) + return deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'succeeded', sessionId } + }) + }) + .catch(async (error) => { + await deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'failed', code: 'agent_session_handoff_failed' } + }) + operationGuard.finish(sessionId, params.envelope.clientOperationId) + const status = idleStructuredHandoffStatus(requireRecord(sessionId)) + setStatus(sessionId, { + ...status, + ...(status.error + ? { + error: { + ...status.error, + details: error instanceof Error ? error.message : String(error) + } + } + : {}) + }) + }) + .finally(() => operationGuard.finish(sessionId, params.envelope.clientOperationId)) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-plans.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-plans.ts index 99835da7cea..d0eb905443c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-plans.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-plans.ts @@ -75,14 +75,16 @@ export function sendPlan(params: { export function cancelPlan(params: { envelope: AgentSessionMutationEnvelope turnId: string + scope?: 'background-tasks' }): MutationPlan { return { method: 'agentSession.cancel', - fields: { turnId: params.turnId }, + fields: { turnId: params.turnId, ...(params.scope ? { scope: params.scope } : {}) }, run: (ctx) => performCancel(ctx, { clientOperationId: params.envelope.clientOperationId, - turnId: params.turnId + turnId: params.turnId, + ...(params.scope ? { scope: params.scope } : {}) }), // Interrupting twice would kill a turn the client never asked to stop, so a // replay reports the turn as already handled instead. diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-option-restoration.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-option-restoration.ts index b9ba03ff327..6e71f822170 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-option-restoration.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-option-restoration.ts @@ -1,7 +1,7 @@ import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' export async function readNativeSessionOptions(input: { - adapter: Pick + adapter: Pick sessionId: string fence: number priorOptions?: Readonly> @@ -11,7 +11,13 @@ export async function readNativeSessionOptions(input: { if (!reported) { return undefined } - const { model: _model, effort: _effort, ...restored } = priorOptions ?? {} + const skipped = new Set(input.adapter.readOptionRestoreFailures?.(sessionId) ?? []) + const restored = priorOptions ? { ...priorOptions } : {} + delete restored.model + delete restored.effort + for (const key of skipped) { + delete restored[key] + } return { ...restored, model: reported.current.model, diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-proven-dead-retry.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-proven-dead-retry.test.ts new file mode 100644 index 00000000000..3caba894cb9 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-proven-dead-retry.test.ts @@ -0,0 +1,178 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { recoverStoredDeadTuiOwnerForHandoff } from '../../runtime/agent-session-handoff-record-transitions' +import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { StructuredAgentSessionHandoffCoordinator } from './structured-agent-session-handoff' +import type { StructuredAgentSessionHandoffTransport } from './structured-agent-session-handoff-types' + +const NOW = 1_800_000_000_000 +const SESSION = 'session-proven-dead-retry' +const THREAD = '019fd532-7c11-7a90-b6de-4e1a2c3d5f60' +const CREATE_OPERATION = `${NOW}-00000000000000000000000000000000` +const OPERATION = `${NOW}-00000000000000000000000000000001` +const roots: string[] = [] + +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe('structured session proven-dead TUI retry', () => { + it('acquires native ownership without trying to close the dead TUI again', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-handoff-dead-retry-')) + roots.push(root) + const store = await AgentSessionRecordStore.open({ + directory: join(root, 'store'), + hostId: 'local' + }) + const reserved = await store.reserveOwner({ + sessionId: SESSION, + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'folder' + }, + provider: 'codex', + accountHome: { variable: 'CODEX_HOME', path: join(root, 'codex-home') }, + runtimeKind: 'tui', + expectedFence: null, + spawnToken: 'tui-spawn', + claimKeyId: 'key-1', + handoffOperationId: null, + probe: { outcome: 'reservation-unused' }, + operation: { callerKey: 'test', operationId: CREATE_OPERATION, fingerprint: 'create' }, + now: NOW + }) + const tuiFence = reserved.record.lease.runtimeFence + await store.commitProcessIdentity({ + sessionId: SESSION, + fence: tuiFence, + process: { + hostId: 'local', + pid: 4200, + processStartTimeMs: NOW - 1_000, + spawnToken: 'tui-spawn' + }, + now: NOW + }) + await store.proveOwner({ + sessionId: SESSION, + fence: tuiFence, + link: { + linkId: 'tui-link', + handle: { provider: 'codex', threadId: THREAD }, + origin: 'created', + mintedAtFence: tuiFence, + observedAt: NOW + }, + now: NOW + }) + await recoverStoredDeadTuiOwnerForHandoff(store, { + sessionId: SESSION, + expectedFence: tuiFence, + operationId: OPERATION, + probe: { outcome: 'pid-absent' }, + now: NOW + }) + const journal = await openAgentSessionJournal({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: THREAD } + }, + journalDir: join(root, 'journal') + }) + const closeTuiOwner = + vi.fn>() + const coordinator = new StructuredAgentSessionHandoffCoordinator({ + store, + claimKeyId: 'key-1', + transport: { + hostLabel: 'Test host', + launchTui: vi.fn(), + reproveTuiOwner: vi.fn(), + recoverTuiOwner: vi.fn(), + stopRecoveredOwner: vi.fn(), + closeTuiOwner, + waitForTuiExit: vi.fn(), + waitForTuiIdleOrExit: vi.fn(), + tuiStatus: () => 'busy' + }, + session: () => ({ journal, fence: store.getRecord(SESSION)?.lease.runtimeFence ?? 1 }), + suspendNative: vi.fn(), + acquireNative: async ({ fence, spawnToken }) => { + await store.commitProcessIdentity({ + sessionId: SESSION, + fence, + process: { + hostId: 'local', + pid: 4300, + processStartTimeMs: NOW, + spawnToken + }, + now: NOW + }) + return store.proveOwner({ + sessionId: SESSION, + fence, + link: { + linkId: 'native-link', + handle: { provider: 'codex', threadId: THREAD }, + origin: 'resumed', + mintedAtFence: fence, + observedAt: NOW + }, + now: NOW + }) + }, + acquireNativeStop: vi.fn(async () => true), + importTuiHistory: vi.fn(), + publish: vi.fn(), + schedule: async (_sessionId, task) => task(), + now: () => NOW + }) + const fields = { + direction: 'to-native' as const, + mode: 'now' as const, + action: 'retry' as const + } + const request: AgentSessionHandoffRequest = { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION, + expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff', + sessionId: SESSION, + fields + }) + }, + ...fields + } + + expect(coordinator.status(SESSION)).toMatchObject({ phase: 'failed', owner: 'tui' }) + expect( + await ( + coordinator as { + request: (callerKey: string, params: AgentSessionHandoffRequest) => Promise + } + ).request('client-1', request) + ).toMatchObject({ ok: true }) + await vi.waitFor(() => expect(coordinator.status(SESSION).owner).toBe('native')) + // Settle the flow's trailing outcome write before afterEach removes the store root. + await coordinator.drain() + expect(closeTuiOwner).not.toHaveBeenCalled() + expect(store.getRecord(SESSION)?.lease).toMatchObject({ + runtimeKind: 'native', + claimStatus: 'live', + handoffStage: null + }) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-recovery-exits.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-recovery-exits.test.ts index a5927dc0c14..5cd888cc5cc 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-recovery-exits.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-recovery-exits.test.ts @@ -8,7 +8,7 @@ import { spawnProcess } from '../../../shared/child-process/run-process' import { CODEX_SPAWN_TOKEN_ENV } from '../../codex/codex-structured-owner-identity' import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' import { readProcessStartTimeMs } from '../../runtime/agent-session-process-identity-probe' -import { createStructuredAgentSessionOwnerProbe } from '../../runtime/structured-agent-session-runtime' +import { createStructuredAgentSessionOwnerProbe } from '../../runtime/structured-agent-session-owner-probe' import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' import { StructuredAgentSessionHost } from './structured-agent-session-host' import type { StructuredAgentSessionHostDeps } from './structured-agent-session-host-types' diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts index 6b627726f98..6e156d91532 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts @@ -114,6 +114,54 @@ describe('AgentSessionSubscribers', () => { }) }) + it('publishes background lifecycle without advancing the journal and carries its fence forward', async () => { + const journal = await journals.open({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: 'provider-1', leafUuid: null } + }, + journalDir: join(root, 'background-journal') + }) + const subscribers = new AgentSessionSubscribers() + const events: AgentSessionSubscribeEvent[] = [] + subscribers.open({ + id: 'subscriber-1', + sessionId: SESSION, + journal, + fence: 1, + backgroundTasks: null, + emit: (event) => events.push(event) + }) + const cursor = journal.cursor() + + const backgroundTasks = { + state: 'monitoring' as const, + tasks: [{ id: 'task-1', kind: 'command' as const, description: 'run the build' }] + } + subscribers.backgroundTasks(SESSION, backgroundTasks, 2) + + expect(journal.cursor()).toEqual(cursor) + expect(events.at(-1)).toEqual({ + type: 'batch', + sessionId: SESSION, + batch: { cursor, items: [], removedItemIds: [], submissions: [] }, + fence: 2, + backgroundTasks + }) + + await journal.appendItem( + { provider: 'orca', clientMessageId: 'after-background-fence' }, + { kind: 'status', text: 'After background state' }, + { fence: 2 } + ) + subscribers.publish(SESSION, journal) + + expect(events.at(-1)).toMatchObject({ type: 'batch', fence: 2 }) + }) + it('catches a subscriber up past a pre-existing unsendable removal with a bounded reset', async () => { const journalDir = join(root, 'oversized-removal-journal') const seeded = await journals.open({ diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts index 3fa80b28d85..d3131505384 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts @@ -10,6 +10,7 @@ import type { } from '../../../shared/agent-session-journal-types' import { AGENT_SESSION_HISTORY_MAX_LIMIT, + type AgentSessionBackgroundTaskState, type AgentSessionHandoffStatus, type AgentSessionSubscribeEvent } from '../../../shared/agent-session-wire' @@ -48,6 +49,7 @@ export class AgentSessionSubscribers { emit: AgentSessionSubscriberEmit cursor?: AgentJournalCursor handoff?: AgentSessionHandoffStatus + backgroundTasks?: AgentSessionBackgroundTaskState | null }): () => void { const liveCursor = input.journal.cursor() const subscriber: Subscriber = { @@ -62,7 +64,7 @@ export class AgentSessionSubscribers { this.bySession.set(input.sessionId, session) if (input.cursor) { - this.deliver(subscriber, input.journal, input.handoff, true) + this.deliver(subscriber, input.journal, input.handoff, true, input.backgroundTasks) } else { const page = readAgentSessionHydrationPage(input.journal, input.fence) this.emit(subscriber, { @@ -70,7 +72,8 @@ export class AgentSessionSubscribers { sessionId: input.sessionId, page, fence: input.fence, - ...(input.handoff ? { handoff: input.handoff } : {}) + ...(input.handoff ? { handoff: input.handoff } : {}), + ...(input.backgroundTasks !== undefined ? { backgroundTasks: input.backgroundTasks } : {}) }) subscriber.cursor = page.liveCursor ?? page.window.nextCursor } @@ -104,20 +107,39 @@ export class AgentSessionSubscribers { sessionId: string, journal: AgentSessionJournal, reason: AgentJournalResetReason, - fence: number + fence: number, + backgroundTasks?: AgentSessionBackgroundTaskState | null ): void { const page = readAgentSessionHydrationPage(journal, fence) for (const subscriber of this.subscribers(sessionId)) { - this.emit(subscriber, { type: 'reset', sessionId, reset: reason, page, fence }) + this.emit(subscriber, { + type: 'reset', + sessionId, + reset: reason, + page, + fence, + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) + }) subscriber.cursor = page.liveCursor ?? page.window.nextCursor subscriber.fence = fence } } - snapshot(sessionId: string, journal: AgentSessionJournal, fence: number): void { + snapshot( + sessionId: string, + journal: AgentSessionJournal, + fence: number, + backgroundTasks?: AgentSessionBackgroundTaskState | null + ): void { const page = readAgentSessionHydrationPage(journal, fence) for (const subscriber of this.subscribers(sessionId)) { - this.emit(subscriber, { type: 'snapshot', sessionId, page, fence }) + this.emit(subscriber, { + type: 'snapshot', + sessionId, + page, + fence, + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) + }) subscriber.cursor = page.liveCursor ?? page.window.nextCursor subscriber.fence = fence } @@ -141,6 +163,28 @@ export class AgentSessionSubscribers { } } + backgroundTasks( + sessionId: string, + state: AgentSessionBackgroundTaskState | null, + fence: number + ): void { + for (const subscriber of this.subscribers(sessionId)) { + this.emit(subscriber, { + type: 'batch', + sessionId, + batch: { + cursor: subscriber.cursor, + items: [], + removedItemIds: [], + submissions: [] + }, + fence, + backgroundTasks: state + }) + subscriber.fence = fence + } + } + private subscribers(sessionId: string): Subscriber[] { return [...(this.bySession.get(sessionId)?.values() ?? [])] } @@ -149,7 +193,8 @@ export class AgentSessionSubscribers { subscriber: Subscriber, journal: AgentSessionJournal, handoff?: AgentSessionHandoffStatus, - emitCheckpoint = false + emitCheckpoint = false, + backgroundTasks?: AgentSessionBackgroundTaskState | null ): void { while (true) { const result = readAgentSessionHistory(journal, { @@ -166,7 +211,8 @@ export class AgentSessionSubscribers { reset: result.reset, page, fence: subscriber.fence, - ...(handoff ? { handoff } : {}) + ...(handoff ? { handoff } : {}), + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) }) subscriber.cursor = page.liveCursor ?? page.window.nextCursor return @@ -185,7 +231,8 @@ export class AgentSessionSubscribers { submissions: [] }, fence: subscriber.fence, - ...(handoff ? { handoff } : {}) + ...(handoff ? { handoff } : {}), + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) }) } return @@ -200,7 +247,8 @@ export class AgentSessionSubscribers { submissions: page.submissions }, fence: subscriber.fence, - ...(handoff ? { handoff } : {}) + ...(handoff ? { handoff } : {}), + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) }) subscriber.cursor = page.window.nextCursor if (!page.hasNewer || !this.isActive(subscriber)) { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-teardown-handoff-drain.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-teardown-handoff-drain.test.ts new file mode 100644 index 00000000000..340d45c05af --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-teardown-handoff-drain.test.ts @@ -0,0 +1,165 @@ +// Host teardown against a handoff that has not finished switching owners. +// +// The flow runs on the session's serialized chain and nothing else awaits it, so a teardown that +// only flushed sinks left it writing into a journal it had just closed — and publishing a status +// against a session it had just dropped, which surfaced as an unhandled rejection. + +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' +import { StructuredAgentSessionHost } from './structured-agent-session-host' +import { + HOST_TEST_NOW as NOW, + HOST_TEST_SESSION as SESSION, + HOST_TEST_THREAD as THREAD, + hostTestAttachParams, + hostTestOperationId, + resetHostTestOperationIds +} from './structured-agent-session-host-test-data' +import { StructuredHandoffTestRequests } from './structured-agent-session-handoff-test-requests' +import type { + StructuredAgentSessionHandoffTransport, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +const CALLER = { callerKey: 'client-1' } + +let root: string +let store: AgentSessionRecordStore +let host: StructuredAgentSessionHost +let launchEntered: PromiseWithResolvers +let launchGate: PromiseWithResolvers + +const requests = new StructuredHandoffTestRequests( + NOW, + SESSION, + () => store.getRecord(SESSION)?.lease.runtimeFence ?? 0 +) + +function tuiOwner(fence: number, spawnToken: string): StructuredTuiOwner { + return { + terminal: { handle: 'term-tui', tabId: 'tab-tui', paneKey: 'pane-tui', ptyId: 'pty-tui' }, + process: { hostId: 'local', pid: 5200, processStartTimeMs: NOW, spawnToken }, + link: { + linkId: `tui-link-${fence}`, + handle: { provider: 'codex', threadId: THREAD }, + origin: 'resumed', + mintedAtFence: fence, + observedAt: NOW + } + } +} + +function gatedTransport(): StructuredAgentSessionHandoffTransport { + return { + hostLabel: 'Test host', + launchTui: async ({ fence, spawnToken }) => { + launchEntered.resolve() + await launchGate.promise + return tuiOwner(fence, spawnToken) + }, + reproveTuiOwner: async ({ owner }) => owner, + recoverTuiOwner: async (record) => + tuiOwner(record.lease.runtimeFence, record.lease.reservedSpawnToken ?? 'recovered'), + stopRecoveredOwner: async () => undefined, + closeTuiOwner: async (owner) => ({ transcriptPath: owner.transcriptPath }), + waitForTuiExit: async (owner) => ({ transcriptPath: owner.transcriptPath }), + waitForTuiIdleOrExit: async () => 'idle', + tuiStatus: () => 'idle' + } +} + +function adapter(): StructuredAgentSessionAdapter { + return { + acquire: vi.fn(async ({ fence, spawnToken }) => ({ + process: { hostId: 'local', pid: 4242, processStartTimeMs: NOW, spawnToken }, + link: { + linkId: `native-link-${fence}`, + handle: { provider: 'codex' as const, threadId: THREAD }, + origin: 'created' as const, + mintedAtFence: fence, + observedAt: NOW + } + })), + dispatchTurn: vi.fn(async () => ({ state: 'accepted' as const })), + cancelTurn: vi.fn(async () => ({ cancelled: true })), + answerPrompt: vi.fn(async () => undefined), + setOption: vi.fn(async () => undefined), + closeSession: vi.fn(async () => true), + supportsCreate: () => true, + supportsRecord: () => true + } as unknown as StructuredAgentSessionAdapter +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-teardown-handoff-drain-')) + resetHostTestOperationIds() + launchEntered = Promise.withResolvers() + launchGate = Promise.withResolvers() + store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) + host = new StructuredAgentSessionHost({ + store, + adapter: adapter(), + journalRoot: root, + claimKeyId: 'key-1', + mintSpawnToken: () => 'spawn-native', + handoffTransport: gatedTransport(), + now: () => NOW + }) + expect(await host.attach(CALLER, hostTestAttachParams(null))).toMatchObject({ ok: true }) +}) + +afterEach(async () => { + launchGate.resolve() + await host.flushAllStreamedEvents() + await rm(root, { recursive: true, force: true }) +}) + +describe('structured agent-session host teardown', () => { + it('waits for an in-flight handoff before dropping the session it is switching', async () => { + // One operation-id source with attach, so the durable ledger sees no duplicate. + const request = requests.request('to-tui', 'now', { operationId: hostTestOperationId() }) + expect(await host.requestHandoff(CALLER, request)).toMatchObject({ ok: true }) + await launchEntered.promise + + let settled = false + const teardown = host.flushAllStreamedEvents().then(() => { + settled = true + }) + // Quiescence probe, not a wait for the flow: teardown must still be blocked on it. + for (let tick = 0; tick < 20; tick += 1) { + await new Promise((resolve) => setTimeout(resolve, 0)) + } + expect(settled).toBe(false) + + launchGate.resolve() + await teardown + + // The new owner was proven while the session was still indexed, not after it vanished. + expect(store.getRecord(SESSION)?.lease).toMatchObject({ + runtimeKind: 'tui', + claimStatus: 'live', + handoffStage: null + }) + expect(host.hasSession(SESSION)).toBe(false) + }) + + it('gives up on a wedged handoff instead of holding the quit open', async () => { + const request = requests.request('to-tui', 'now', { operationId: hostTestOperationId() }) + expect(await host.requestHandoff(CALLER, request)).toMatchObject({ ok: true }) + await launchEntered.promise + + // The gate is never opened: this is the flow that never comes back. + vi.useFakeTimers() + try { + const teardown = host.flushAllStreamedEvents() + await vi.advanceTimersByTimeAsync(5_000) + await expect(teardown).resolves.toBeUndefined() + } finally { + vi.useRealTimers() + } + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-turns-prompt.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-turns-prompt.ts index 9d16e19a7f7..26ad85b5cfb 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-turns-prompt.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-turns-prompt.ts @@ -1,6 +1,11 @@ import { parseAgentJournalItemKey } from '../../../shared/agent-session-journal-item-key' +import { + decodeAgentSessionQuestionAnswers, + isValidAgentSessionQuestionAnswers +} from '../../../shared/agent-session-question-answer' import type { AgentJournalItemBody, + AgentJournalQuestion, AgentJournalResolution } from '../../../shared/agent-session-journal-types' import type { AgentSessionPromptResult } from '../../../shared/agent-session-wire' @@ -14,6 +19,7 @@ function invalid(message: string): TurnOutcome { function promptBodyOf(body: AgentJournalItemBody): { options: readonly { id: string }[] freeTextQuestionId?: string + questions?: AgentJournalQuestion[] resolution: AgentJournalResolution } | null { return body.kind === 'approval' || body.kind === 'question' ? body : null @@ -64,7 +70,19 @@ export async function performPrompt( prompt.freeTextQuestionId !== undefined && freeText?.questionId === prompt.freeTextQuestionId && freeText.answer.trim().length > 0 - if (!acceptsFreeText && !prompt.options.some((option) => option.id === input.optionId)) { + const grouped = + item.body.kind === 'question' && prompt.questions + ? decodeAgentSessionQuestionAnswers(input.optionId) + : null + const acceptsGrouped = + grouped !== null && + prompt.questions !== undefined && + isValidAgentSessionQuestionAnswers(prompt.questions, grouped) + if ( + !acceptsFreeText && + !acceptsGrouped && + !prompt.options.some((option) => option.id === input.optionId) + ) { return invalid(`Option ${input.optionId} is not offered by item ${input.itemId}.`) } const identity = parseAgentJournalItemKey(input.itemId) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts index 5222f9557f9..e8e6f998bdd 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts @@ -73,4 +73,35 @@ describe('performCancel', () => { { kind: 'status', text: 'Cancellation requested.' } ]) }) + + it('stops background tasks without interrupting the foreground turn or writing a row', async () => { + root = await mkdtemp(join(tmpdir(), 'orca-background-task-cancel-')) + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) + const cancelTurn = vi.fn(async () => ({ cancelled: true })) + const stopBackgroundTasks = vi.fn(async () => ({ cancelled: true })) + const ctx: AgentSessionTurnContext = { + sessionId: 'session-1', + journal, + fence: 1, + adapter: { cancelTurn, stopBackgroundTasks } as unknown as StructuredAgentSessionAdapter, + persistOptions: async () => undefined, + resolvedBy: 'client-1', + publish: vi.fn(), + now: () => 1 + } + + const result = await performCancel(ctx, { + clientOperationId: 'cancel-background-tasks', + turnId: 'background-tasks', + scope: 'background-tasks' + }) + + expect(result).toEqual({ + ok: true, + value: { turnId: 'background-tasks', cancelled: true } + }) + expect(stopBackgroundTasks).toHaveBeenCalledWith({ sessionId: 'session-1', fence: 1 }) + expect(cancelTurn).not.toHaveBeenCalled() + expect(journal.snapshot().items).toEqual([]) + }) }) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts index f1717027b8b..76c4e8e89d6 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts @@ -146,18 +146,29 @@ export async function performSend( export async function performCancel( ctx: AgentSessionTurnContext, - input: { clientOperationId: string; turnId: string } + input: { + clientOperationId: string + turnId: string + scope?: 'background-tasks' + } ): Promise> { let cancelled = false let note = 'Cancellation requested.' try { - cancelled = ( - await ctx.adapter.cancelTurn({ - sessionId: ctx.sessionId, - turnId: input.turnId, - fence: ctx.fence - }) - ).cancelled + cancelled = input.scope + ? ( + await ctx.adapter.stopBackgroundTasks?.({ + sessionId: ctx.sessionId, + fence: ctx.fence + }) + )?.cancelled === true + : ( + await ctx.adapter.cancelTurn({ + sessionId: ctx.sessionId, + turnId: input.turnId, + fence: ctx.fence + }) + ).cancelled if (!cancelled) { note = 'The provider had already finished this turn.' } @@ -166,6 +177,9 @@ export async function performCancel( error instanceof Error ? error.message : String(error) }` } + if (input.scope) { + return { ok: true, value: { turnId: input.turnId, cancelled } } + } // Keyed by the operation id so a replayed cancel upserts one item, not two. await appendStatus(ctx, input.clientOperationId, note) return { ok: true, value: { turnId: input.turnId, cancelled } } diff --git a/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts b/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts index 804d2900a22..60f34707ac9 100644 --- a/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts +++ b/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts @@ -54,7 +54,8 @@ function directReadableMessage(payload: unknown): string | null { return null } -function readableMessage(payload: unknown): string | null { +/** The provider's own sentence for a frame, when it carries one. */ +export function readableProviderFrameText(payload: unknown): string | null { const direct = directReadableMessage(payload) if (direct || typeof payload !== 'object' || payload === null || Array.isArray(payload)) { return direct @@ -89,7 +90,7 @@ export function unhandledProviderFrameJournalItem( // Why: the opcode alone ("codex · notification:warning") tells the user nothing // and reads as protocol noise. Lead with the provider's own sentence when it has // one; the raw frame stays behind the row's disclosure either way. - const message = readableMessage(payload) + const message = readableProviderFrameText(payload) const display = message ? boundInlineText(message, limits) : null return { body: { diff --git a/src/main/native-chat/claude-structured-managed-account-support.test.ts b/src/main/native-chat/claude-structured-managed-account-support.test.ts new file mode 100644 index 00000000000..f647579d03e --- /dev/null +++ b/src/main/native-chat/claude-structured-managed-account-support.test.ts @@ -0,0 +1,154 @@ +import { describe, expect, it } from 'vitest' +import { getSelectedClaudeAccountIdForTarget } from '../claude-accounts/runtime-selection' +import { + structuredClaudeMatchesActiveManagedAccount, + type ClaudeManagedAccountGateSettings +} from './claude-structured-managed-account-support' + +function account(id: string, managedAuthRuntime: 'host' | 'wsl') { + return { + id, + email: `${id}@example.com`, + managedAuthPath: `/managed/${id}`, + managedAuthRuntime, + authMethod: 'subscription-oauth' as const, + createdAt: 0, + updatedAt: 0, + lastAuthenticatedAt: 0 + } +} + +function settings( + overrides: Partial +): ClaudeManagedAccountGateSettings { + return { claudeManagedAccounts: [], activeClaudeManagedAccountId: null, ...overrides } +} + +describe('structuredClaudeMatchesActiveManagedAccount', () => { + it('allows an unmanaged install, where nothing claims an identity', () => { + expect(structuredClaudeMatchesActiveManagedAccount(settings({}))).toBe(true) + }) + + it('allows a selected host account, which the runtime syncs into the ambient config', () => { + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [account('host-1', 'host')], + activeClaudeManagedAccountIdsByRuntime: { host: 'host-1', wsl: {} } + }) + ) + ).toBe(true) + }) + + it('refuses a WSL-only managed account, which never reaches the ambient config', () => { + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [account('wsl-1', 'wsl')], + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-1' } } + }) + ) + ).toBe(false) + }) + + it('refuses when a host selection names an account that is WSL-bound or missing', () => { + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [account('wsl-1', 'wsl')], + activeClaudeManagedAccountIdsByRuntime: { host: 'wsl-1', wsl: {} } + }) + ) + ).toBe(false) + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [account('host-1', 'host')], + activeClaudeManagedAccountIdsByRuntime: { host: 'gone', wsl: {} } + }) + ) + ).toBe(false) + }) + + /** Absent and empty are the same answer: this user has no managed Claude accounts, so nothing + * claims an identity and the ambient path is legitimate. Only settings that cannot be READ are + * unknown. Treating a missing key as unknown strands profiles that simply never wrote it — the + * auth policy's own predicate takes `(accounts ?? [])` for exactly this reason. */ + it('treats an absent account list the same as an empty one', () => { + expect( + structuredClaudeMatchesActiveManagedAccount(settings({ claudeManagedAccounts: [] })) + ).toBe(true) + expect( + structuredClaudeMatchesActiveManagedAccount({ + activeClaudeManagedAccountId: null + } as unknown as ClaudeManagedAccountGateSettings) + ).toBe(true) + }) + + it('fails closed when the settings cannot be read at all', () => { + expect(structuredClaudeMatchesActiveManagedAccount(null)).toBe(false) + expect(structuredClaudeMatchesActiveManagedAccount(undefined)).toBe(false) + }) + + /** The four states this gate exists to tell apart, pinned together so a change to one is visible + * against the others. */ + it.each([ + ['no managed accounts', [], null, true], + ['accounts present, none active, no WSL account', [account('host-1', 'host')], null, true], + ['host account selected', [account('host-1', 'host')], 'host-1', true], + ['WSL-only, normalized to no host selection', [account('wsl-1', 'wsl')], null, false] + ] as const)('resolves %s', (_name, claudeManagedAccounts, activeId, expected) => { + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [...claudeManagedAccounts], + activeClaudeManagedAccountIdsByRuntime: { host: activeId, wsl: {} } + }) + ) + ).toBe(expected) + }) + + /** THE discriminator, and the whole of this rule. With nothing selected for the host runtime the + * settings alone cannot distinguish honest deselection from the WSL-only steady state, because + * `pruneInvalidClaudeRuntimeSelection` empties the host slot in the second case and persists it. + * So the presence of ANY WSL-bound account decides. Simplifying this to "none active -> + * supported" re-opens the auth-identity misrepresentation this gate exists to prevent. */ + it('splits none-active on whether a WSL-bound account exists at all', () => { + const noneActive = (accounts: ReturnType[]) => + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: accounts, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: {} } + }) + ) + + expect(noneActive([account('host-1', 'host')])).toBe(true) + expect(noneActive([account('host-1', 'host'), account('host-2', 'host')])).toBe(true) + expect(noneActive([account('wsl-1', 'wsl')])).toBe(false) + // Mixed list still refuses: the WSL account is present and nothing is selected. + expect(noneActive([account('host-1', 'host'), account('wsl-1', 'wsl')])).toBe(false) + }) + + /** The gate and the auth policy must resolve the SAME account. A legacy settings blob carries the + * selection only in the flat `activeClaudeManagedAccountId`, which is where the accessor's + * fall-through lives — reading the runtime map directly silently disagrees with the policy. */ + it('resolves the same account as the auth policy on a legacy flat selection', () => { + const legacy = settings({ + claudeManagedAccounts: [account('host-1', 'host')], + activeClaudeManagedAccountId: 'host-1' + }) + + expect(getSelectedClaudeAccountIdForTarget(legacy, { runtime: 'host' })).toBe('host-1') + expect(structuredClaudeMatchesActiveManagedAccount(legacy)).toBe(true) + }) + + it('agrees with the auth policy that a legacy flat WSL selection is refused', () => { + const legacy = settings({ + claudeManagedAccounts: [account('wsl-1', 'wsl')], + activeClaudeManagedAccountId: 'wsl-1' + }) + + expect(getSelectedClaudeAccountIdForTarget(legacy, { runtime: 'host' })).toBe('wsl-1') + expect(structuredClaudeMatchesActiveManagedAccount(legacy)).toBe(false) + }) +}) diff --git a/src/main/native-chat/claude-structured-managed-account-support.ts b/src/main/native-chat/claude-structured-managed-account-support.ts new file mode 100644 index 00000000000..dccf6216bda --- /dev/null +++ b/src/main/native-chat/claude-structured-managed-account-support.ts @@ -0,0 +1,61 @@ +import type { GlobalSettings } from '../../shared/global-settings-types' +import { getSelectedClaudeAccountIdForTarget } from '../claude-accounts/runtime-selection' + +export type ClaudeManagedAccountGateSettings = Pick< + GlobalSettings, + | 'claudeManagedAccounts' + | 'activeClaudeManagedAccountId' + | 'activeClaudeManagedAccountIdsByRuntime' +> + +/** + * A structured Claude session launches against the ambient Claude config, which the account service + * keeps in sync with the selected HOST account. A WSL-bound managed account lives inside the distro + * and is never synced there, so such a session would authenticate as whatever the ambient identity + * happens to be while the UI names the WSL account — the user is told one identity and given + * another. Refuse the structured path there and let the terminal-backed one, which resolves the + * account per runtime, handle that account shape. + * + * Reads the selection through the same accessor the auth policy uses. Resolving it any other way + * lets the two disagree, and a session admitted by this gate would then run under a policy computed + * from a different account than the one approved here. + * + * Unknown answers refuse, and only genuinely unknown ones: settings that cannot be read at all, or + * an active selection this cannot resolve. An install with no managed accounts — the list empty or + * never written — claims no identity and is fine. + */ +export function structuredClaudeMatchesActiveManagedAccount( + settings: ClaudeManagedAccountGateSettings | null | undefined +): boolean { + if (!settings) { + return false + } + // Absent is the same answer as empty — this user has no managed Claude accounts, so nothing + // claims an identity and ambient auth is the truth. Only settings that cannot be READ are + // unknown, and those refuse above. The auth policy reads the list the same way. + const accounts = settings.claudeManagedAccounts ?? [] + if (accounts.length === 0) { + return true + } + const activeHostId = getSelectedClaudeAccountIdForTarget(settings, { runtime: 'host' }) + if (!activeHostId) { + // Nothing selected for the host runtime is two different states that the settings cannot tell + // apart after the fact: honest deselection, where ambient auth is the truth and the UI names no + // identity, and the WSL-only case, where the prune emptied the host slot and persisted null + // while the UI still names the WSL account. The presence of any WSL-bound account decides. + return !accounts.some((candidate) => candidate.managedAuthRuntime === 'wsl') + } + const active = accounts.find((candidate) => candidate.id === activeHostId) + return active ? active.managedAuthRuntime !== 'wsl' : false +} + +/** Reads the gate's settings, answering null when they cannot be read so callers refuse. */ +export function readClaudeManagedAccountGateSettings( + getSettings: () => ClaudeManagedAccountGateSettings +): ClaudeManagedAccountGateSettings | null { + try { + return getSettings() + } catch { + return null + } +} diff --git a/src/main/native-chat/session-file-resolver-claude-roots.test.ts b/src/main/native-chat/session-file-resolver-claude-roots.test.ts new file mode 100644 index 00000000000..87ebd570280 --- /dev/null +++ b/src/main/native-chat/session-file-resolver-claude-roots.test.ts @@ -0,0 +1,97 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const scanned = vi.hoisted(() => ({ dirs: [] as string[], hits: {} as Record })) +vi.mock('../ai-vault/session-scanner-discovery', () => ({ + walkSessionFiles: async (dir: string) => { + scanned.dirs.push(dir) + const hit = scanned.hits[dir] + return hit ? [hit] : [] + } +})) + +import { homedir } from 'node:os' +import { join } from 'node:path' +import { resolveSessionFilePath } from './session-file-resolver' + +const DEFAULT_ROOT = join(homedir(), '.claude', 'projects') +const CONFIG_DIR = '/opt/claude-home' +const CONFIG_ROOT = join(CONFIG_DIR, 'projects') + +let previousConfigDir: string | undefined + +beforeEach(() => { + previousConfigDir = process.env.CLAUDE_CONFIG_DIR + scanned.dirs = [] + scanned.hits = {} +}) + +afterEach(() => { + if (previousConfigDir === undefined) { + delete process.env.CLAUDE_CONFIG_DIR + } else { + process.env.CLAUDE_CONFIG_DIR = previousConfigDir + } +}) + +/** + * Honouring CLAUDE_CONFIG_DIR fixed new sessions but would otherwise hide every + * transcript written before the user adopted the variable. The Codex resolver in this + * same file already searches managed-then-default and de-dupes; Claude does the same. + */ +describe('claude transcript roots', () => { + it('searches the config-dir root first, then the default home', async () => { + process.env.CLAUDE_CONFIG_DIR = CONFIG_DIR + + await resolveSessionFilePath('claude', 'session-1') + + expect(scanned.dirs).toEqual([CONFIG_ROOT, DEFAULT_ROOT]) + }) + + it('still finds history written before CLAUDE_CONFIG_DIR was adopted', async () => { + process.env.CLAUDE_CONFIG_DIR = CONFIG_DIR + const legacy = join(DEFAULT_ROOT, '-repos-old', 'session-1.jsonl') + scanned.hits[DEFAULT_ROOT] = legacy + + await expect(resolveSessionFilePath('claude', 'session-1')).resolves.toBe(legacy) + }) + + it('prefers the config-dir root when both hold the session', async () => { + process.env.CLAUDE_CONFIG_DIR = CONFIG_DIR + scanned.hits[CONFIG_ROOT] = join(CONFIG_ROOT, '-repos-new', 'session-1.jsonl') + scanned.hits[DEFAULT_ROOT] = join(DEFAULT_ROOT, '-repos-old', 'session-1.jsonl') + + await expect(resolveSessionFilePath('claude', 'session-1')).resolves.toBe( + scanned.hits[CONFIG_ROOT] + ) + // The default root is never reached, so the common case pays for one scan. + expect(scanned.dirs).toEqual([CONFIG_ROOT]) + }) + + it('scans one root when the variable is unset', async () => { + delete process.env.CLAUDE_CONFIG_DIR + + await resolveSessionFilePath('claude', 'session-1') + + expect(scanned.dirs).toEqual([DEFAULT_ROOT]) + }) + + it('de-dupes when CLAUDE_CONFIG_DIR names the default home', async () => { + process.env.CLAUDE_CONFIG_DIR = join(homedir(), '.claude') + + await resolveSessionFilePath('claude', 'session-1') + + expect(scanned.dirs).toEqual([DEFAULT_ROOT]) + }) + + it('honours an explicit root override without adding fallbacks', async () => { + process.env.CLAUDE_CONFIG_DIR = CONFIG_DIR + // The account-home callers (structured-claude-runtime-adapter, the host handoff) + // know the exact tree their session pinned; a fallback there could resolve a + // different account's transcript. + await resolveSessionFilePath('claude', 'session-1', { + claudeProjectsDir: '/accounts/pinned/projects' + }) + + expect(scanned.dirs).toEqual(['/accounts/pinned/projects']) + }) +}) diff --git a/src/main/native-chat/session-file-resolver.test.ts b/src/main/native-chat/session-file-resolver.test.ts index 584d8a25a9d..04946f5b464 100644 --- a/src/main/native-chat/session-file-resolver.test.ts +++ b/src/main/native-chat/session-file-resolver.test.ts @@ -1,9 +1,12 @@ import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { dirname, join } from 'node:path' -import { afterEach, describe, expect, it } from 'vitest' +import { afterEach, describe, expect, it, vi } from 'vitest' -import { ClaudeTranscriptTailIncompleteError } from '../claude/claude-transcript-branch-proof' +import { + ClaudeTranscriptTailIncompleteError, + readClaudeTranscriptLeafWithReproof +} from '../claude/claude-transcript-branch-proof' import { readClaudeTranscriptLeafUuid, resolveSessionFilePath } from './session-file-resolver' let tempRoots: string[] = [] @@ -139,6 +142,263 @@ describe('resolveSessionFilePath', () => { ) }) + it('rejects non-transcript and sidechain UUIDs as the durable leaf', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-leaf-filter-') + const transcript = join(root, 'session.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'sidechain-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + isSidechain: true + }, + { type: 'result', uuid: 'result-frame', parentUuid: 'main-user', sessionId: 'session-1' }, + { + type: 'system', + subtype: 'init', + uuid: 'init-frame', + parentUuid: null, + sessionId: 'session-1' + }, + { type: 'stream_event', uuid: 'stream-frame', parentUuid: null, sessionId: 'session-1' }, + { type: 'last-prompt', leafUuid: 'sidechain-assistant', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1')).rejects.toThrow( + 'marker leaf is missing from the session graph' + ) + }) + + it('rejects a main leaf whose ancestry crosses a subagent sidechain', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-sidechain-ancestry-') + const transcript = join(root, 'session.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'sidechain-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + isSidechain: true + }, + { + type: 'assistant', + uuid: 'main-after-sidechain', + parentUuid: 'sidechain-assistant', + sessionId: 'session-1' + }, + { type: 'last-prompt', leafUuid: 'main-after-sidechain', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1')).rejects.toThrow( + 'not on the main transcript' + ) + }) + + it('rejects a main leaf whose ancestry crosses a parent-tool-use sidechain', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-parent-tool-ancestry-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'subagent-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + parent_tool_use_id: 'tool-use-1' + }, + { + type: 'assistant', + uuid: 'main-after-sidechain', + parentUuid: 'subagent-assistant', + sessionId: 'session-1' + }, + { type: 'last-prompt', leafUuid: 'main-after-sidechain', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1')).rejects.toThrow( + 'not on the main transcript' + ) + }) + + it('rejects a previous cursor descended from a parent-tool-use sidechain', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-parent-tool-cursor-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'subagent-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + parent_tool_use_id: 'tool-use-1' + }, + { + type: 'assistant', + uuid: 'main-after-sidechain', + parentUuid: 'subagent-assistant', + sessionId: 'session-1' + }, + { type: 'last-prompt', leafUuid: 'main-after-sidechain', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect( + readClaudeTranscriptLeafUuid(transcript, 'session-1', 'main-after-sidechain') + ).rejects.toThrow('not on the main transcript') + }) + + it('rejects a latest marker descended from a parent-tool-use cursor sidechain', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-parent-tool-cursor-descendant-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'subagent-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + parent_tool_use_id: 'tool-use-1' + }, + { + type: 'assistant', + uuid: 'main-after-sidechain', + parentUuid: 'subagent-assistant', + sessionId: 'session-1' + }, + { + type: 'assistant', + uuid: 'latest-after-sidechain', + parentUuid: 'main-after-sidechain', + sessionId: 'session-1' + }, + { type: 'last-prompt', leafUuid: 'latest-after-sidechain', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect( + readClaudeTranscriptLeafUuid(transcript, 'session-1', 'main-after-sidechain') + ).rejects.toThrow('not on the main transcript') + }) + + it('rejects a post-snapshot descendant whose parent row was observed later', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-post-snapshot-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { + type: 'assistant', + uuid: 'descendant', + parentUuid: 'previous', + sessionId: 'session-1' + }, + { type: 'assistant', uuid: 'previous', parentUuid: null, sessionId: 'session-1' }, + { type: 'last-prompt', leafUuid: 'descendant', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1', 'previous')).rejects.toThrow( + 'parent row follows descendant' + ) + }) + + it('does not re-prove a divergent sibling after the sampled cursor rejects', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-sibling-reproof-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'root', parentUuid: null, sessionId: 'session-1' }, + { type: 'assistant', uuid: 'old', parentUuid: 'root', sessionId: 'session-1' }, + { type: 'assistant', uuid: 'new', parentUuid: 'root', sessionId: 'session-1' }, + { type: 'last-prompt', leafUuid: 'new', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + const calls: (string | null)[] = [] + const readTranscriptLeaf = async ({ + previousLeafUuid + }: { + previousLeafUuid: string | null + }) => { + calls.push(previousLeafUuid) + return readClaudeTranscriptLeafUuid(transcript, 'session-1', previousLeafUuid) + } + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1', 'old')).rejects.toThrow( + 'sibling branch' + ) + + await expect( + readClaudeTranscriptLeafWithReproof({ + readTranscriptLeaf, + claudeConfigDir: '/accounts/claude', + providerSessionId: 'session-1', + previousLeafUuid: 'old' + }) + ).rejects.toThrow('sibling branch') + expect(calls).toEqual(['old']) + }) + + it('does not accept a divergent sibling after a truncated-tail reproof', async () => { + const calls: (string | null)[] = [] + const readTranscriptLeaf = vi.fn( + async ({ previousLeafUuid }: { previousLeafUuid: string | null }) => { + calls.push(previousLeafUuid) + if (calls.length === 1) { + throw new ClaudeTranscriptTailIncompleteError() + } + return 'divergent-sibling' + } + ) + + await expect( + readClaudeTranscriptLeafWithReproof({ + readTranscriptLeaf, + claudeConfigDir: '/accounts/claude', + providerSessionId: 'session-1', + previousLeafUuid: 'old' + }) + ).rejects.toBeInstanceOf(ClaudeTranscriptTailIncompleteError) + expect(calls).toEqual(['old']) + }) + it('globs Claude project subdirs for .jsonl', async () => { const root = await makeRoot('orca-native-chat-resolve-claude-') const claudeProjectsDir = join(root, 'claude-projects') @@ -410,3 +670,42 @@ describe('resolveSessionFilePath', () => { expect(resolved).toBe(target) }) }) + +// Mobile native chat resolves with no root override (transcript-read-cache.ts:104), +// while the account home a structured Claude session pins is +// `CLAUDE_CONFIG_DIR || ~/.claude` (runtime-paths.ts:15). When the two disagree the +// CLI writes one place and mobile reads another, and the chat goes dark with no +// wire-level error — so the default root has to honour the same variable. +describe('the default Claude transcript root mobile falls back to', () => { + it('follows CLAUDE_CONFIG_DIR, the same variable the pinned account home follows', async () => { + const configDir = await makeRoot('orca-native-chat-claude-config-dir-') + const slugDir = join(configDir, 'projects', '-repos-workspace-1') + await mkdir(slugDir, { recursive: true }) + const transcript = join(slugDir, 'session-under-config-dir.jsonl') + await writeFile(transcript, '', 'utf8') + const previous = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = configDir + + try { + // No `claudeProjectsDir` override: exactly the call mobile makes. + await expect(resolveSessionFilePath('claude', 'session-under-config-dir')).resolves.toBe( + transcript + ) + } finally { + restoreEnv('CLAUDE_CONFIG_DIR', previous) + } + }) + + it('ignores a blank CLAUDE_CONFIG_DIR rather than resolving against the filesystem root', async () => { + const previous = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = ' ' + + try { + await expect( + resolveSessionFilePath('claude', 'session-that-does-not-exist') + ).resolves.toBeNull() + } finally { + restoreEnv('CLAUDE_CONFIG_DIR', previous) + } + }) +}) diff --git a/src/main/native-chat/session-file-resolver.ts b/src/main/native-chat/session-file-resolver.ts index 0ee73fddc8d..12d2e615742 100644 --- a/src/main/native-chat/session-file-resolver.ts +++ b/src/main/native-chat/session-file-resolver.ts @@ -31,8 +31,20 @@ import { proveClaudeTranscriptBranch } from '../claude/claude-transcript-branch- // the remote main resolves its local home, so we never hardcode an absolute // user path — homedir()/CODEX_HOME resolution stays runtime-relative and is // computed per call (not at module load) so it tracks the live home. -function claudeProjectsDir(): string { - return join(homedir(), '.claude', 'projects') +// Why CLAUDE_CONFIG_DIR and not just homedir(): a structured Claude session pins its +// account home to `CLAUDE_CONFIG_DIR || ~/.claude` (claude-accounts/runtime-paths.ts), +// and the CLI writes its transcript under whatever home it was given. Mobile native chat +// resolves with no root override, so a default that ignored the variable read a different +// tree than the CLI wrote — a silent blackout, not an error. +// Why both roots and not just that one: adopting the variable would otherwise hide every +// transcript written before it was set. Same managed-then-default shape as +// codexSessionsDirs below, de-duped so the usual case still scans once. +function claudeProjectsDirs(): string[] { + const candidates = [ + join(process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude'), 'projects'), + join(homedir(), '.claude', 'projects') + ] + return candidates.filter((dir, index) => candidates.indexOf(dir) === index) } // Why: Orca launches Codex with ORCA_CODEX_HOME pointing at its own managed @@ -173,9 +185,11 @@ async function resolveSessionFileById( } if (transcriptAgent === 'claude') { + // An explicit root is the caller naming the exact account tree its session pinned; + // adding a fallback there could resolve a different account's transcript. return resolveClaudeSessionFile( trimmedId, - options.claudeProjectsDir ?? claudeProjectsDir(), + options.claudeProjectsDir ? [options.claudeProjectsDir] : claudeProjectsDirs(), signal ) } @@ -205,16 +219,22 @@ async function resolveSessionFileById( async function resolveClaudeSessionFile( sessionId: string, - projectsDir: string, + projectsDirs: readonly string[], signal?: AbortSignal ): Promise { const targetName = `${sessionId}.jsonl` - const files = await walkSessionFiles(projectsDir, 'claude', [], { - extensions: new Set(['.jsonl']), - filePredicate: (path) => basename(path) === targetName, - signal - }) - return files[0] ?? null + for (const projectsDir of projectsDirs) { + // No existence pre-check: walkSessionFiles already yields [] for a missing root. + const files = await walkSessionFiles(projectsDir, 'claude', [], { + extensions: new Set(['.jsonl']), + filePredicate: (path) => basename(path) === targetName, + signal + }) + if (files[0]) { + return files[0] + } + } + return null } async function resolveCodexSessionFile( diff --git a/src/main/native-chat/structured-agent-session-create-support.test.ts b/src/main/native-chat/structured-agent-session-create-support.test.ts new file mode 100644 index 00000000000..ab19d369bac --- /dev/null +++ b/src/main/native-chat/structured-agent-session-create-support.test.ts @@ -0,0 +1,83 @@ +import { describe, expect, it } from 'vitest' +import type { AgentSessionExecutionLocation } from '../../shared/agent-session-record' +import type { ClaudeManagedAccountGateSettings } from './claude-structured-managed-account-support' +import { resolveStructuredAgentSessionCreateSupport } from './structured-agent-session-create-support' + +const LOCAL: AgentSessionExecutionLocation = { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' +} + +function managedAccount(id: string, managedAuthRuntime: 'host' | 'wsl') { + return { + id, + email: `${id}@example.com`, + managedAuthPath: `/managed/${id}`, + managedAuthRuntime, + authMethod: 'subscription-oauth' as const, + createdAt: 0, + updatedAt: 0, + lastAuthenticatedAt: 0 + } +} + +const HOST_SELECTED: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('host-1', 'host')], + activeClaudeManagedAccountId: 'host-1', + activeClaudeManagedAccountIdsByRuntime: { host: 'host-1', wsl: {} } +} + +const WSL_ONLY: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('wsl-1', 'wsl')], + activeClaudeManagedAccountId: null, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-1' } } +} + +function support( + overrides: Partial[0]> = {} +) { + return resolveStructuredAgentSessionCreateSupport({ + agent: 'claude', + location: LOCAL, + adapterSupportsCreate: true, + getSettings: () => HOST_SELECTED, + ...overrides + }) +} + +describe('resolveStructuredAgentSessionCreateSupport', () => { + it('supports Claude under a selected host account', () => { + expect(support()).toEqual({ supported: true }) + }) + + it('refuses Claude under a WSL-only managed account', () => { + expect(support({ getSettings: () => WSL_ONLY })).toEqual({ supported: false, reason: 'wsl' }) + }) + + it('fails closed for Claude when the settings throw', () => { + expect( + support({ + getSettings: () => { + throw new Error('no store') + } + }) + ).toEqual({ supported: false, reason: 'wsl' }) + }) + + it('leaves Codex to the adapter answer under the same WSL-only account', () => { + expect(support({ agent: 'codex', getSettings: () => WSL_ONLY })).toEqual({ supported: true }) + }) + + it.each([ + ['remote', { ...LOCAL, executionHostId: 'ssh:host-a' }, 'remote'], + ['wsl workspace', { ...LOCAL, wslDistro: 'Ubuntu' }, 'wsl'], + ['unsupported agent', LOCAL, 'agent'] + ] as const)('keeps the adapter refusal reason for %s', (_name, location, reason) => { + expect(support({ adapterSupportsCreate: false, location })).toEqual({ + supported: false, + reason + }) + }) +}) diff --git a/src/main/native-chat/structured-agent-session-create-support.ts b/src/main/native-chat/structured-agent-session-create-support.ts new file mode 100644 index 00000000000..9b96a1af4be --- /dev/null +++ b/src/main/native-chat/structured-agent-session-create-support.ts @@ -0,0 +1,48 @@ +import type { AgentSessionExecutionLocation } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import { + readClaudeManagedAccountGateSettings, + structuredClaudeMatchesActiveManagedAccount, + type ClaudeManagedAccountGateSettings +} from './claude-structured-managed-account-support' + +export type StructuredAgentSessionCreateSupport = { + supported: boolean + reason?: 'agent' | 'remote' | 'wsl' +} + +/** + * The create-support verdict, kept out of the runtime class file because that file is `@ts-nocheck` + * — a call site there is not typechecked, so an auth-identity decision written inline would compile + * however wrong it was. The runtime hands over the two facts it owns and this decides. + */ +export function resolveStructuredAgentSessionCreateSupport(input: { + agent: 'claude' | 'codex' + location: AgentSessionExecutionLocation + adapterSupportsCreate: boolean + getSettings: () => ClaudeManagedAccountGateSettings +}): StructuredAgentSessionCreateSupport { + if (!input.adapterSupportsCreate) { + return { + supported: false, + reason: + input.location.executionHostId !== LOCAL_EXECUTION_HOST_ID + ? 'remote' + : input.location.wslDistro + ? 'wsl' + : 'agent' + } + } + // Claude only: Codex resolves its account on a different path, so its answer is untouched here. + // `wsl` is the closest existing reason — the cause is a WSL-bound account rather than a WSL + // workspace — and no client reads the field, so it stays as-is. + if ( + input.agent === 'claude' && + !structuredClaudeMatchesActiveManagedAccount( + readClaudeManagedAccountGateSettings(input.getSettings) + ) + ) { + return { supported: false, reason: 'wsl' } + } + return { supported: true } +} diff --git a/src/main/native-chat/transcript-line-decoders-claude.ts b/src/main/native-chat/transcript-line-decoders-claude.ts index f819656bdb2..5202035bbc6 100644 --- a/src/main/native-chat/transcript-line-decoders-claude.ts +++ b/src/main/native-chat/transcript-line-decoders-claude.ts @@ -3,6 +3,8 @@ import { NATIVE_CHAT_INTERRUPTED_STATUS_TEXT, type NativeChatBlock, + type NativeChatEditPatch, + type NativeChatEditPatchHunk, type NativeChatMessage } from '../../shared/native-chat-types' import { @@ -15,6 +17,59 @@ import { imageSourcePathFromText } from '../../shared/native-chat-image-transcri import { claudeContentBlocks } from './transcript-record-blocks' import { claudeInterruptedMessageId } from './transcript-turn-markers' +const MAX_EDIT_PATCH_HUNKS = 40 +const MAX_EDIT_PATCH_HUNK_LINES = 400 + +/** Claude reports an edit as a snippet pair on the call, which cannot locate the + * change in the file. The result record carries the hunks it resolved against + * the real file, so keep them for the renderer's line-number gutter. */ +function claudeEditPatch(record: Record): NativeChatEditPatch | null { + const result = asRecord(record.toolUseResult) + const raw = result?.structuredPatch + if (!Array.isArray(raw) || raw.length === 0) { + return null + } + const hunks: NativeChatEditPatchHunk[] = [] + for (const entry of raw.slice(0, MAX_EDIT_PATCH_HUNKS)) { + const hunk = asRecord(entry) + const lines = hunk?.lines + if ( + typeof hunk?.oldStart !== 'number' || + typeof hunk.newStart !== 'number' || + !Array.isArray(lines) + ) { + continue + } + hunks.push({ + oldStart: hunk.oldStart, + oldLines: typeof hunk.oldLines === 'number' ? hunk.oldLines : 0, + newStart: hunk.newStart, + newLines: typeof hunk.newLines === 'number' ? hunk.newLines : 0, + lines: lines + .slice(0, MAX_EDIT_PATCH_HUNK_LINES) + .flatMap((line) => (typeof line === 'string' ? [line] : [])) + }) + } + if (hunks.length === 0) { + return null + } + const filePath = extractString(result?.filePath) + return { ...(filePath ? { filePath } : {}), hunks } +} + +/** Attaches the resolved hunks to the record's tool result, which is the only + * block in a Claude result turn. */ +function withEditPatch(blocks: NativeChatBlock[], patch: NativeChatEditPatch): NativeChatBlock[] { + let attached = false + return blocks.map((block) => { + if (attached || block.type !== 'tool-result') { + return block + } + attached = true + return { ...block, editPatch: patch } + }) +} + export function decodeClaudeTranscriptLine( line: string, fallbackId: string @@ -41,7 +96,9 @@ export function decodeClaudeTranscriptLine( } } const message = asRecord(record.message) - const decodedBlocks = claudeContentBlocks(message?.content) + const editPatch = claudeEditPatch(record) + const contentBlocks = claudeContentBlocks(message?.content) + const decodedBlocks = editPatch ? withEditPatch(contentBlocks, editPatch) : contentBlocks if (decodedBlocks.length === 0) { return null } diff --git a/src/main/native-chat/transcript-line-decoders-codex.ts b/src/main/native-chat/transcript-line-decoders-codex.ts index c4b0ddbf5ba..d70bd7a7c80 100644 --- a/src/main/native-chat/transcript-line-decoders-codex.ts +++ b/src/main/native-chat/transcript-line-decoders-codex.ts @@ -202,6 +202,10 @@ function codexTurnItemBlocks(content: unknown): NativeChatBlock[] { return blocks } +/** The argument payload is passed through exactly as it arrived. Decoding it + * here would change the shape every `.input` consumer sees — including the ask + * surface, which reads a question shape out of any tool's input — so the one + * consumer that needs structure decodes it for itself. */ function codexCallInput(payload: Record): unknown { if (payload.arguments !== undefined) { return payload.arguments diff --git a/src/main/native-chat/transcript-reader-codex-history-mode.test.ts b/src/main/native-chat/transcript-reader-codex-history-mode.test.ts index 5b0f3dc014f..5d18bec8c85 100644 --- a/src/main/native-chat/transcript-reader-codex-history-mode.test.ts +++ b/src/main/native-chat/transcript-reader-codex-history-mode.test.ts @@ -2,6 +2,7 @@ import { mkdtemp, rm, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, describe, expect, it } from 'vitest' +import { extractPendingAsk } from '../../shared/native-chat-ask' import { decodeCodexTranscriptLine } from './transcript-line-decoders-codex' import { readNativeChatTranscript } from './transcript-reader' import { readNativeChatTranscriptTail } from './transcript-tail-reader' @@ -247,4 +248,43 @@ describe('Codex transcript history modes', () => { blocks: [{ type: 'tool-result', output: 'ok' }] }) }) + + it('passes an argument payload through untouched, so no consumer changes shape', () => { + const call = decodeCodexTranscriptLine( + JSON.stringify({ + type: 'response_item', + payload: { + type: 'function_call', + id: 'call-2', + name: 'shell', + arguments: '{"command":["bash","-lc","echo hi"]}' + } + }), + 'fallback-args' + ) + + expect(call?.blocks[0]).toMatchObject({ + type: 'tool-call', + name: 'shell', + input: '{"command":["bash","-lc","echo hi"]}' + }) + }) + + it('does not raise a question card from an unrelated tool that carries a questions payload', () => { + const call = decodeCodexTranscriptLine( + JSON.stringify({ + type: 'response_item', + payload: { + type: 'function_call', + id: 'call-3', + name: 'some_mcp_tool', + arguments: '{"questions":[{"question":"Which branch?","options":["main","dev"]}]}' + } + }), + 'fallback-questions' + ) + + expect(call).not.toBeNull() + expect(extractPendingAsk(call ? [call] : [])).toBeNull() + }) }) diff --git a/src/main/orca-chromium-process-pids.ts b/src/main/orca-chromium-process-pids.ts index b2613e42b79..3282f3a4d66 100644 --- a/src/main/orca-chromium-process-pids.ts +++ b/src/main/orca-chromium-process-pids.ts @@ -12,6 +12,12 @@ import { recordCoalescedDurableCrashBreadcrumb } from './crash-reporting/durable * Empty on a Node host and empty on failure: that is "no refusal proven", never * "safe to kill" — callers must keep every other guard they already have. * + * The other direction is real too, and bounded by design: `getAppMetrics()` can + * still list a renderer Electron has not finished reaping, so on Windows a pid + * already recycled onto an unrelated child of ours reads as `own` and its tree + * walk is refused. That is why a refusal only blocks the pid-addressed walk and + * every gated site still kills its own root through the child handle. + * * Why failure stays open rather than refusing everything: a refusal is not free. * `terminateWindowsProcessTree` resolves without killing, and * `killSourceControlAgentProcess` returns that straight to a caller that then diff --git a/src/main/own-chromium-tree-kill-guard.test.ts b/src/main/own-chromium-tree-kill-guard.test.ts index bd3b1674e18..7b9c30687bc 100644 --- a/src/main/own-chromium-tree-kill-guard.test.ts +++ b/src/main/own-chromium-tree-kill-guard.test.ts @@ -13,7 +13,12 @@ import { import { readOrcaChromiumProcessPids } from './orca-chromium-process-pids' import { classifyWindowsTreeKillTarget } from './windows-pty-root-identity' import { terminateWindowsProcessTree } from './windows-process-tree-kill' -import { admitSelfInitiatedTreeKill } from './own-chromium-tree-kill-guard' +import { + admitSelfInitiatedTreeKill, + installMainProcessTreeKillGate +} from './own-chromium-tree-kill-guard' +import { killCodexAppServerProcessTree } from './codex/codex-app-server-session' +import { setProcessTreeKillGate } from '../shared/child-process/process-tree-kill-gate' import { resetSelfInitiatedTreeKillLogForTest } from './crash-reporting/self-initiated-tree-kill-log' import { clearCrashBreadcrumbsForTest, @@ -57,6 +62,7 @@ beforeEach(() => { setActiveSink({ push: () => {}, flush: () => {}, close: () => {} }) clearCrashBreadcrumbsForTest() resetSelfInitiatedTreeKillLogForTest() + installMainProcessTreeKillGate() }) afterEach(() => { @@ -66,6 +72,7 @@ afterEach(() => { vi.restoreAllMocks() _resetTracerForTests() clearCrashBreadcrumbsForTest() + setProcessTreeKillGate(null) }) describe('refusing to tree-kill our own Chromium processes', () => { @@ -143,6 +150,38 @@ describe('refusing to tree-kill our own Chromium processes', () => { ) }) + it('refuses the codex app-server deadline kill against one of our own pids', () => { + const spawnImpl = vi.fn(() => ({ on: vi.fn(), unref: vi.fn() })) + const child = { pid: RENDERER_PID, kill: vi.fn() } + + killCodexAppServerProcessTree(child as never, { + platform: 'win32', + spawnImpl: spawnImpl as never + }) + + // The deadline timer fires on `child.pid` alone; a reaped-then-recycled pid + // is the stale-pid mechanism this gate exists to stop. + expect(spawnImpl).not.toHaveBeenCalled() + expect(getCrashBreadcrumbSnapshot()).toEqual([ + expect.objectContaining({ name: 'self_tree_kill_refused_own_chromium' }) + ]) + }) + + it('still lets the codex app-server deadline kill reach a foreign pid', () => { + const killer = { on: vi.fn(), unref: vi.fn() } + const spawnImpl = vi.fn(() => killer) + + killCodexAppServerProcessTree({ pid: 7777, kill: vi.fn() } as never, { + platform: 'win32', + spawnImpl: spawnImpl as never + }) + + expect(spawnImpl).toHaveBeenCalledWith('taskkill', ['/pid', '7777', '/t', '/f'], { + stdio: 'ignore', + windowsHide: true + }) + }) + /** * Fail-open is the deliberate choice — see `orca-chromium-process-pids.ts` for * why refusing everything is worse — so the crumb is the only thing that keeps diff --git a/src/main/own-chromium-tree-kill-guard.ts b/src/main/own-chromium-tree-kill-guard.ts index 4daedf4faa4..24b6a4b7327 100644 --- a/src/main/own-chromium-tree-kill-guard.ts +++ b/src/main/own-chromium-tree-kill-guard.ts @@ -4,16 +4,23 @@ import { type SelfInitiatedTreeKillScope } from './crash-reporting/self-initiated-tree-kill-log' import { readOrcaChromiumProcessPids } from './orca-chromium-process-pids' +import { setProcessTreeKillGate } from '../shared/child-process/process-tree-kill-gate' /** - * Gate every main-process tree-kill through one decision: refuse the pid when - * Electron is currently accounting for it, otherwise put it on the record. + * Gate every main-process tree-kill through one decision: refuse a pid-addressed + * walk when Electron is currently accounting for the pid, otherwise put the + * kill on the record. * * Why a shared gate rather than a check inside `terminateWindowsProcessTree`: - * the codex and claude account-login teardowns run their own `taskkill /T /F` - * with different lifetimes (one sync, one with its own timeout ladder), so a - * guard that only lived in the tree-kill helper would cover one of three - * families. Returns false when the caller must not kill. + * five other families in main run their own `taskkill /T /F` with different + * lifetimes (sync, fire-and-forget, timeout ladder), and three more live in + * `src/shared` and reach this through `process-tree-kill-gate`, so a guard that + * only lived in the tree-kill helper would cover one of nine. + * `main-process-tree-kill-gate.test.ts` holds that set closed by counting `/pid` + * call sites against gate admissions per file, not by file. Returns false + * when the caller must not walk that pid's tree; the caller still kills its own + * root through the child handle (`refused-tree-kill-root-termination.test.ts`), + * so a refusal is never a process leak. * * Electron main only, by construction. `terminateWindowsProcessTree` also runs * in the standalone daemon (the `pty-descendant-sweep` site), where @@ -30,11 +37,26 @@ export function admitSelfInitiatedTreeKill(target: { }): boolean { // Why: no PTY root, codex root or git child is ever one of our own Chromium // processes, so a pid that is means the caller is about to kill a renderer, - // the GPU or the browser itself (#10680). - if (readOrcaChromiumProcessPids().has(target.pid)) { - recordRefusedOwnChromiumTreeKill(target) - return false + // the GPU or the browser itself (#10680). Only the pid-addressed scope can + // land there: a POSIX group holds only what Orca put in it, so that arm is + // recorded and admitted like every other group kill in main, and a stale + // `getAppMetrics()` entry cannot orphan a macOS/Linux tree. + const isOwnChromiumPid = + target.scope === 'win-taskkill-tree' && readOrcaChromiumProcessPids().has(target.pid) + try { + if (isOwnChromiumPid) { + recordRefusedOwnChromiumTreeKill(target) + } else { + recordSelfInitiatedTreeKill(target) + } + } catch { + // Recording must never turn a successful termination into a failed one, and + // never flip the decision: it is taken above, before anything can throw. } - recordSelfInitiatedTreeKill(target) - return true + return !isOwnChromiumPid +} + +/** Hands the gate to the shared choke points, which cannot import main. */ +export function installMainProcessTreeKillGate(): void { + setProcessTreeKillGate((kill) => admitSelfInitiatedTreeKill(kill)) } diff --git a/src/main/persistence/restoring-sessions/pane-key-remapping.test.ts b/src/main/persistence/restoring-sessions/pane-key-remapping.test.ts index c0359efd612..2c73045d901 100644 --- a/src/main/persistence/restoring-sessions/pane-key-remapping.test.ts +++ b/src/main/persistence/restoring-sessions/pane-key-remapping.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it } from 'vitest' import { makePaneKey } from '../../../shared/stable-pane-id' import { + remapAcknowledgedAgentPaneKeys, remapActivityClearedAtPaneKeys, remapManuallyUnreadTurnPaneKeys } from './pane-key-remapping' @@ -30,4 +31,16 @@ describe('remapManuallyUnreadTurnPaneKeys', () => { changed: false }) }) + + it('returns the caller map untouched when no key needs remapping', () => { + const remap = new Map([['tab-1', new Map([[STABLE_LEAF_ID, STABLE_LEAF_ID]])]]) + const stable = { [makePaneKey('tab-1', STABLE_LEAF_ID)]: 1 } + + const result = remapAcknowledgedAgentPaneKeys(stable, remap) + + // Why identity and not just equality: this runs on every session write against a map that + // grows with every pane ever opened, so a rebuilt-then-discarded copy is pure garbage. + expect(result.acknowledgements).toBe(stable) + expect(result.changed).toBe(false) + }) }) diff --git a/src/main/persistence/restoring-sessions/pane-key-remapping.ts b/src/main/persistence/restoring-sessions/pane-key-remapping.ts index 4f3440f087e..8e43184b0c5 100644 --- a/src/main/persistence/restoring-sessions/pane-key-remapping.ts +++ b/src/main/persistence/restoring-sessions/pane-key-remapping.ts @@ -3,51 +3,50 @@ import { isTerminalLeafId, makePaneKey, parsePaneKey } from '../../../shared/sta type PaneLeafRemap = Map> +/** Resolves the pane key a legacy entry should move to, or `null` when it stays put. */ +function resolveRemappedPaneKey( + paneKey: string, + leafIdByInputLeafIdByTabId: PaneLeafRemap +): string | null { + if (parsePaneKey(paneKey)) { + return null + } + const delimiter = paneKey.indexOf(':') + if (delimiter <= 0 || delimiter === paneKey.length - 1) { + return null + } + const tabId = paneKey.slice(0, delimiter) + const remappedLeafId = leafIdByInputLeafIdByTabId.get(tabId)?.get(paneKey.slice(delimiter + 1)) + // makePaneKey cannot throw here: tabId is non-empty and colon-free by construction. + return remappedLeafId && isTerminalLeafId(remappedLeafId) + ? makePaneKey(tabId, remappedLeafId) + : null +} + function remapPaneKeys( values: Record | undefined, leafIdByInputLeafIdByTabId: PaneLeafRemap ): { values: Record | undefined; changed: boolean } { - if (!values || Object.keys(values).length === 0) { + // Why the classify-first pass: these maps grow with every pane ever opened and this runs on + // every session write, but post-migration no key is ever rewritten. Rebuilding the whole + // object only to discard it was pure garbage; the rewrite below is unchanged. + if ( + !values || + !Object.keys(values).some( + (paneKey) => resolveRemappedPaneKey(paneKey, leafIdByInputLeafIdByTabId) !== null + ) + ) { return { values, changed: false } } - let changed = false const next: Record = {} - const setValue = (paneKey: string, value: T): void => { - const existing = next[paneKey] - next[paneKey] = existing === undefined ? value : (Math.max(existing, value) as T) - } for (const [paneKey, value] of Object.entries(values)) { - const parsed = parsePaneKey(paneKey) - if (parsed) { - setValue(paneKey, value) - continue - } - - const delimiter = paneKey.indexOf(':') - if (delimiter <= 0 || delimiter === paneKey.length - 1) { - setValue(paneKey, value) - continue - } - - const tabId = paneKey.slice(0, delimiter) - const legacyLeafId = paneKey.slice(delimiter + 1) - const remappedLeafId = leafIdByInputLeafIdByTabId.get(tabId)?.get(legacyLeafId) - if (!remappedLeafId || !isTerminalLeafId(remappedLeafId)) { - setValue(paneKey, value) - continue - } - - try { - // Carry values over when a legacy leaf is promoted to a UUID. - setValue(makePaneKey(tabId, remappedLeafId), value) - changed = true - } catch { - setValue(paneKey, value) - } + // Carry values over when a legacy leaf is promoted to a UUID; keep the max on collision. + const target = resolveRemappedPaneKey(paneKey, leafIdByInputLeafIdByTabId) ?? paneKey + const existing = next[target] + next[target] = existing === undefined ? value : (Math.max(existing, value) as T) } - - return { values: next, changed } + return { values: next, changed: true } } export function remapAcknowledgedAgentPaneKeys( diff --git a/src/main/providers/local-pty-finalize-environment.ts b/src/main/providers/local-pty-finalize-environment.ts index 0f7d3454a09..1b385639e94 100644 --- a/src/main/providers/local-pty-finalize-environment.ts +++ b/src/main/providers/local-pty-finalize-environment.ts @@ -109,6 +109,9 @@ export function finalizeLocalPtySpawnEnvironment(args: { codexStartupCommand !== undefined && supportsPosixShellStartupCommand(shell) ? codexStartupCommand : undefined + // Why no line-editor widening here (unlike the daemon and relay): a Codex + // startup command this provider wraps is run by the wrapper's own prompt + // hook, never written into the PTY, so there is no early write to double-echo. const waitsForShellReady = Boolean(spawn.command) && (!isCodexStartupCommand || codexRequiresShellReady) return getShellLaunchConfig( diff --git a/src/main/providers/local-pty-foreground-inspection-cheap-tier.test.ts b/src/main/providers/local-pty-foreground-inspection-cheap-tier.test.ts new file mode 100644 index 00000000000..37d4cf3c4d0 --- /dev/null +++ b/src/main/providers/local-pty-foreground-inspection-cheap-tier.test.ts @@ -0,0 +1,138 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type * as ProcessTableSnapshotReader from '../../shared/process-table-snapshot-reader' + +const { cheapSnapshotMock, fullSnapshotMock, resolveMock } = vi.hoisted(() => ({ + cheapSnapshotMock: vi.fn(), + fullSnapshotMock: vi.fn(), + resolveMock: vi.fn() +})) + +vi.mock('../../shared/cheap-process-table-snapshot-reader', () => ({ + getCheapProcessTableSnapshot: cheapSnapshotMock +})) +vi.mock('../../shared/process-table-snapshot-reader', async (importOriginal) => ({ + ...(await importOriginal()), + getProcessTableSnapshot: fullSnapshotMock +})) +vi.mock('./agent-foreground-process', () => ({ + resolveAgentForegroundProcessWithAvailability: resolveMock, + confirmShellForegroundProcess: vi.fn() +})) + +import { getLocalPtyForegroundProcess } from './local-pty-foreground-inspection' +import { ptyLastRecognizedForeground, ptyProcesses, ptyShellName } from './local-pty-provider-state' + +const SHELL_PID = 4242 +const AGENT_PID = 4300 +const ID = 'pty-1' + +type Table = 'agent' | 'shell-only' +let table: Table = 'agent' + +function rows(): Record[] { + const tpgid = table === 'agent' ? AGENT_PID : SHELL_PID + const out: Record[] = [ + { + pid: SHELL_PID, + ppid: 1, + pgid: SHELL_PID, + tpgid, + stat: table === 'agent' ? 'Ss' : 'Ss+', + tty: 'ttys004', + startTime: 'Thu Sep 3 16:02:01 2026', + command: '-zsh' + } + ] + if (table === 'agent') { + out.push({ + pid: AGENT_PID, + ppid: SHELL_PID, + pgid: AGENT_PID, + tpgid, + stat: 'S+', + tty: 'ttys004', + startTime: 'Thu Sep 3 16:02:05 2026', + command: 'node /usr/local/bin/claude' + }) + } + return out +} + +describe('local POSIX provider cheap-tier revalidation', () => { + let platform: PropertyDescriptor | undefined + const proc = { pid: SHELL_PID, process: 'node' } + + beforeEach(() => { + platform = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: 'darwin' }) + table = 'agent' + proc.process = 'node' + cheapSnapshotMock.mockReset() + cheapSnapshotMock.mockImplementation(async () => rows()) + fullSnapshotMock.mockReset() + fullSnapshotMock.mockImplementation(async () => rows()) + resolveMock.mockReset() + resolveMock.mockImplementation(async () => ({ + available: true, + processName: table === 'agent' ? 'claude' : 'zsh' + })) + ptyProcesses.set(ID, proc as never) + ptyShellName.set(ID, 'zsh') + ptyLastRecognizedForeground.delete(ID) + }) + + afterEach(() => { + ptyProcesses.delete(ID) + ptyShellName.delete(ID) + ptyLastRecognizedForeground.delete(ID) + if (platform) { + Object.defineProperty(process, 'platform', platform) + } + }) + + it('a pane with NO recognized anchor never consults the cheap tier', async () => { + table = 'shell-only' + proc.process = 'zsh' + for (let i = 0; i < 3; i += 1) { + expect(await getLocalPtyForegroundProcess(ID)).toBe('zsh') + } + expect(cheapSnapshotMock).not.toHaveBeenCalled() + expect(resolveMock).toHaveBeenCalledTimes(3) + expect(ptyLastRecognizedForeground.get(ID)).toBeUndefined() + }) + + it('once recognized, an unchanged pane re-proves the agent from the cheap tier without a full scan', async () => { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + expect(resolveMock).toHaveBeenCalledTimes(1) + expect(ptyLastRecognizedForeground.get(ID)?.steady?.fingerprint).toEqual(expect.any(String)) + for (let i = 0; i < 3; i += 1) { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + } + expect(cheapSnapshotMock).toHaveBeenCalledTimes(3) + expect(resolveMock).toHaveBeenCalledTimes(1) + }) + + it('an agent exit changes the fingerprint, escalates to the full scan, and clears the anchor', async () => { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + table = 'shell-only' + expect(await getLocalPtyForegroundProcess(ID)).toBe('zsh') + expect(resolveMock).toHaveBeenCalledTimes(2) + expect(ptyLastRecognizedForeground.get(ID)).toBeUndefined() + }) + + it('a changed node-pty foreground name escalates without consulting the cheap tier', async () => { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + proc.process = 'zsh' + table = 'shell-only' + expect(await getLocalPtyForegroundProcess(ID)).toBe('zsh') + expect(cheapSnapshotMock).not.toHaveBeenCalled() + expect(resolveMock).toHaveBeenCalledTimes(2) + }) + + it('a cheap capture failure falls through to the full scan', async () => { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + cheapSnapshotMock.mockRejectedValueOnce(new Error('ps died')) + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + expect(resolveMock).toHaveBeenCalledTimes(2) + }) +}) diff --git a/src/main/providers/local-pty-foreground-inspection.ts b/src/main/providers/local-pty-foreground-inspection.ts index c42c06d9121..d4a717a9de2 100644 --- a/src/main/providers/local-pty-foreground-inspection.ts +++ b/src/main/providers/local-pty-foreground-inspection.ts @@ -1,8 +1,11 @@ import { recognizeAgentProcessFromCommandLine } from '../../shared/agent-process-recognition' +import { getCheapProcessTableSnapshot } from '../../shared/cheap-process-table-snapshot-reader' +import { getProcessTableSnapshot } from '../../shared/process-table-snapshot-reader' import { confirmShellForegroundProcess, resolveAgentForegroundProcessWithAvailability } from './agent-foreground-process' +import { buildPaneProcessFingerprint } from './posix-pane-foreground-fingerprint' import { resolveForegroundFallbackProcess } from './local-pty-launch-helpers' import { ptyAgentForegroundContextPaths, @@ -35,6 +38,34 @@ export async function hasLocalPtyChildProcesses(id: string): Promise { } } +/** + * POSIX twin of the Windows job-membership short-circuit below: a pane that already holds a + * recognized agent re-proves it from the cheap `ps` tier when the subtree fingerprint is + * unchanged. Panes with no anchor never get here, so start discovery is untouched. + */ +async function revalidateCachedPosixAgent( + proc: { pid: number }, + cachedEntry: { + name: string + steady?: { fingerprint: string; fallbackProcess: string | null } | null + }, + fallbackProcess: string | null +): Promise { + const steady = cachedEntry.steady + if (!steady || steady.fallbackProcess !== fallbackProcess) { + return false + } + try { + const observed = await buildPaneProcessFingerprint( + await getCheapProcessTableSnapshot(), + proc.pid + ) + return observed !== null && observed === steady.fingerprint + } catch { + return false + } +} + export async function getLocalPtyForegroundProcess(id: string): Promise { const proc = ptyProcesses.get(id) if (!proc) { @@ -89,6 +120,18 @@ export async function getLocalPtyForegroundProcess(id: string): Promise { + if (process.platform === 'win32') { + return null + } + try { + const fingerprint = await buildPaneProcessFingerprint(await getProcessTableSnapshot(), shellPid) + return fingerprint === null ? null : { fingerprint, fallbackProcess } + } catch { + return null + } +} + export async function confirmLocalPtyForegroundProcess(id: string): Promise { const proc = ptyProcesses.get(id) if (!proc) { diff --git a/src/main/providers/local-pty-provider-spawn-session.test.ts b/src/main/providers/local-pty-provider-spawn-session.test.ts index 7d998e69544..9dfa08d81bf 100644 --- a/src/main/providers/local-pty-provider-spawn-session.test.ts +++ b/src/main/providers/local-pty-provider-spawn-session.test.ts @@ -174,7 +174,10 @@ describe('LocalPtyProvider', () => { id: 'serve-session-1', incarnationId: first.incarnationId, pid: 12345, - isReattach: true + isReattach: true, + // Why published: this attach really moved the PTY, unlike daemon/relay attach, so main + // must record 120x40 rather than preserving the size it held for the session. + attachedGrid: { cols: 120, rows: 40 } }) expect(mockProc.resize).toHaveBeenCalledWith(120, 40) expect(spawnMock).not.toHaveBeenCalled() diff --git a/src/main/providers/local-pty-provider-state.ts b/src/main/providers/local-pty-provider-state.ts index 5e87502b1f6..9e54ac859b0 100644 --- a/src/main/providers/local-pty-provider-state.ts +++ b/src/main/providers/local-pty-provider-state.ts @@ -43,9 +43,16 @@ export const ptyAgentForegroundContextPaths = new Map() // Why: remember the last recognized agent foreground so a degraded scan doesn't report the shell and look like an exit. // `pid` anchors the identity to the row that proved it (null when ambiguous); // `at` is the last confirmation, so unanchored job evidence -- only a superset -- cannot hold it forever. +// `steady` (POSIX) is the pane fingerprint the recognizing capture proved plus node-pty's name at +// that moment; a cheap capture matching it re-proves the identity without the full table. export const ptyLastRecognizedForeground = new Map< string, - { name: string; pid: number | null; at: number } + { + name: string + pid: number | null + at: number + steady?: { fingerprint: string; fallbackProcess: string | null } | null + } >() export const ptyTerminalHandle = new Map() export const ptyWorktreeId = new Map() diff --git a/src/main/providers/local-pty-spawn-state.ts b/src/main/providers/local-pty-spawn-state.ts index bf1bea6bcb8..2ab145f6c39 100644 --- a/src/main/providers/local-pty-spawn-state.ts +++ b/src/main/providers/local-pty-spawn-state.ts @@ -52,8 +52,10 @@ export function reattachLocalPty(id: string, cols: number, rows: number): PtySpa if (!existing) { return null } + let resized = false try { existing.resize(cols, rows) + resized = true } catch { /* Existing PTY may reject resize during teardown; still return the live handle. */ } @@ -62,6 +64,8 @@ export function reattachLocalPty(id: string, cols: number, rows: number): PtySpa ...(ptyIncarnations.has(id) ? { incarnationId: ptyIncarnations.get(id) } : {}), pid: existing.pid, ...(ptyWslDistroById.has(id) ? { wslDistro: ptyWslDistroById.get(id) ?? null } : {}), - isReattach: true + isReattach: true, + // Why: unlike daemon/relay attach, this one really moved the live PTY to the caller's grid. + ...(resized ? { attachedGrid: { cols, rows } } : {}) } } diff --git a/src/main/providers/posix-pane-foreground-fingerprint.test.ts b/src/main/providers/posix-pane-foreground-fingerprint.test.ts new file mode 100644 index 00000000000..76cb68f79dc --- /dev/null +++ b/src/main/providers/posix-pane-foreground-fingerprint.test.ts @@ -0,0 +1,187 @@ +import { describe, expect, it } from 'vitest' +import { + buildPaneProcessFingerprint, + type PaneFingerprintRow +} from './posix-pane-foreground-fingerprint' + +const SHELL = 4242 +const AGENT = 4300 +const OTHER_PANE = 9000 + +type Row = PaneFingerprintRow + +const shell = (over: Partial = {}): Row => ({ + pid: SHELL, + ppid: 1, + pgid: SHELL, + tpgid: AGENT, + stat: 'Ss', + startTime: 'Thu Sep 3 16:02:01 2026', + ...over +}) +const agent = (over: Partial = {}): Row => ({ + pid: AGENT, + ppid: SHELL, + pgid: AGENT, + tpgid: AGENT, + stat: 'S+', + startTime: 'Thu Sep 3 16:02:05 2026', + ...over +}) +const child = (pid: number, ppid: number, over: Partial = {}): Row => ({ + pid, + ppid, + pgid: AGENT, + tpgid: AGENT, + stat: 'S+', + startTime: `Thu Sep 3 16:03:${String(pid % 60).padStart(2, '0')} 2026`, + ...over +}) +const foreign = (): Row => ({ + pid: OTHER_PANE, + ppid: 1, + pgid: OTHER_PANE, + tpgid: OTHER_PANE, + stat: 'Ss+', + startTime: 'Thu Sep 3 12:00:00 2026' +}) + +const fp = (rows: Row[]): Promise => + buildPaneProcessFingerprint(rows, SHELL, { platform: 'darwin' }) + +describe('buildPaneProcessFingerprint', () => { + const baseline = [foreign(), shell(), agent()] + + it('is stable across captures that differ only in scheduler state, row order, and foreign panes', async () => { + const a = await fp(baseline) + expect(a).not.toBeNull() + // R vs S: a working agent flips this every tick and it says nothing about the pane. + expect(await fp([agent({ stat: 'R+' }), shell({ stat: 'Ss' }), foreign()])).toBe(a) + // The shell going idle-vs-runnable, or a foreign pane starting/exiting, is not our business. + expect(await fp([shell({ stat: 'Rs' }), agent()])).toBe(a) + // lstart padding differs between column sets; both must stamp identically. + expect(await fp([shell({ startTime: 'Thu Sep 3 16:02:01 2026' }), agent()])).toBe(a) + }) + + describe('escalates (fingerprint changes) on every completion-relevant transition', () => { + it('agent exit: the recognized pid vanishes from the subtree', async () => { + const before = await fp(baseline) + expect(await fp([foreign(), shell({ tpgid: SHELL, stat: 'Ss+' })])).not.toBe(before) + }) + + it('exit-and-replace: the same pid is reused by a new process with a new start time', async () => { + const before = await fp(baseline) + expect(await fp([shell(), agent({ startTime: 'Thu Sep 3 16:09:00 2026' })])).not.toBe(before) + }) + + it('Ctrl-Z: the agent stops and the shell takes the terminal back', async () => { + const before = await fp(baseline) + expect(await fp([shell({ tpgid: SHELL, stat: 'Ss+' }), agent({ stat: 'T' })])).not.toBe( + before + ) + }) + + it('bg: the stopped job resumes in the background, foreground stays with the shell', async () => { + const stopped = await fp([shell({ tpgid: SHELL, stat: 'Ss+' }), agent({ stat: 'T' })]) + const backgrounded = await fp([shell({ tpgid: SHELL, stat: 'Ss+' }), agent({ stat: 'S' })]) + expect(backgrounded).not.toBe(stopped) + expect(backgrounded).not.toBe(await fp(baseline)) + }) + + it('child churn: a subprocess appearing or disappearing under the agent', async () => { + const before = await fp(baseline) + const withChild = await fp([shell(), agent(), child(4310, AGENT)]) + expect(withChild).not.toBe(before) + expect(await fp([shell(), agent(), child(4310, AGENT), child(4311, 4310)])).not.toBe( + withChild + ) + // A child exec'ing away from the group (setsid / disown) is also a change. + expect(await fp([shell(), agent(), child(4310, AGENT, { pgid: 4310 })])).not.toBe(withChild) + }) + + it('shell replaced: same pid, different start time', async () => { + const before = await fp(baseline) + expect(await fp([shell({ startTime: 'Thu Sep 3 17:00:00 2026' }), agent()])).not.toBe(before) + }) + }) + + describe('refuses to fingerprint an unfenced pane (caller must take the full capture)', () => { + it('root shell missing from the capture', async () => { + expect(await fp([foreign(), agent()])).toBeNull() + }) + + it('root shell has no start marker', async () => { + expect(await fp([shell({ startTime: undefined }), agent()])).toBeNull() + }) + + it('root shell has no job-control columns', async () => { + expect(await fp([shell({ pgid: undefined, tpgid: undefined }), agent()])).toBeNull() + }) + }) + + describe('Linux', () => { + it('reads /proc start times for the pane subtree only and ignores ps start markers', async () => { + const asked: number[] = [] + const read = async (pid: number): Promise => { + asked.push(pid) + return pid === SHELL ? '1000' : pid === AGENT ? '2000' : null + } + const rows = [foreign(), shell({ startTime: undefined }), agent({ startTime: undefined })] + const a = await buildPaneProcessFingerprint(rows, SHELL, { + platform: 'linux', + readLinuxStartTime: read + }) + expect(a).toContain(`${SHELL}@1000`) + expect(a).toContain(`${AGENT}@2000`) + expect(asked.sort()).toEqual([SHELL, AGENT].sort()) + // An exit-and-replace changes only the /proc start ticks. + const replaced = await buildPaneProcessFingerprint(rows, SHELL, { + platform: 'linux', + readLinuxStartTime: async (pid) => (pid === AGENT ? '2500' : read(pid)) + }) + expect(replaced).not.toBe(a) + }) + + it('refuses when the root /proc entry cannot be read', async () => { + expect( + await buildPaneProcessFingerprint([shell(), agent()], SHELL, { + platform: 'linux', + readLinuxStartTime: async () => null + }) + ).toBeNull() + }) + it('refuses when a DESCENDANT start marker cannot be read', async () => { + // Why: the start marker is what makes a pid comparison recycle-safe. Stamping a missing + // one as a placeholder let two captures that both failed to read it compare equal across + // a recycled pid, so a vanished agent looked unchanged and the cheap tier kept serving + // its name. Refusing sends the caller to the full capture. + const rows = [shell(), agent()] + + expect( + await buildPaneProcessFingerprint(rows, SHELL, { + platform: 'linux', + readLinuxStartTime: async (pid) => (pid === SHELL ? '2400' : null) + }) + ).toBeNull() + }) + + it('does not let a recycled descendant pid reuse a fingerprint', async () => { + // Both captures fail to read the descendant marker; the pid is reused by a different + // process in between. Equal fingerprints here would mask the agent's exit. + const readNoDescendant = async (pid: number): Promise => + pid === SHELL ? '2400' : null + const before = await buildPaneProcessFingerprint([shell(), agent()], SHELL, { + platform: 'linux', + readLinuxStartTime: readNoDescendant + }) + const after = await buildPaneProcessFingerprint( + [shell(), agent({ stat: 'S+', pgid: AGENT })], + SHELL, + { platform: 'linux', readLinuxStartTime: readNoDescendant } + ) + + expect(before).toBeNull() + expect(after).toBeNull() + }) + }) +}) diff --git a/src/main/providers/posix-pane-foreground-fingerprint.ts b/src/main/providers/posix-pane-foreground-fingerprint.ts new file mode 100644 index 00000000000..ac1539300e2 --- /dev/null +++ b/src/main/providers/posix-pane-foreground-fingerprint.ts @@ -0,0 +1,93 @@ +import { readFile } from 'node:fs/promises' +import { collectDescendantsFromIndex, getProcessTableIndex } from '../../shared/process-table-index' +import { parseLinuxProcStatStartTime } from '../../shared/process-table-snapshot-reader' + +/** The job-control columns both `ps` tiers carry; `command`/`tty` are deliberately absent. */ +export type PaneFingerprintRow = { + pid: number + ppid: number + pgid?: number + tpgid?: number + stat: string + startTime?: string +} + +export type PaneFingerprintDeps = { + platform?: NodeJS.Platform + /** Linux: `/proc//stat` field 22, read for the pane subtree only. */ + readLinuxStartTime?: (pid: number) => Promise +} + +/** + * Only the job-control bits of `stat`. The scheduler letter (R/S/D/I/U) flips every tick + * on a working agent and says nothing about whether the pane changed hands; stopped, + * zombie, and foreground-group membership do. + */ +function jobControlState(stat: string): string { + const head = stat[0] ?? '' + const lifecycle = head === 'T' || head === 't' ? 'T' : head === 'Z' ? 'Z' : '' + return lifecycle + (stat.includes('+') ? '+' : '') +} + +async function readLinuxProcStartTime(pid: number): Promise { + try { + return parseLinuxProcStatStartTime(await readFile(`/proc/${pid}/stat`, 'utf8')) + } catch { + return null + } +} + +/** + * A per-pane summary of everything the cheap `ps` tier can see: the root shell's identity + * (pid + start marker) and terminal foreground group, and every descendant's identity, group, + * and job-control state. Two captures with equal fingerprints describe the same pane + * subtree, so the name resolved from the last full capture still holds. + * + * Null when the root is missing or unfenced (no start marker, no group columns): callers + * must then take the full capture rather than trust a comparison that could not be made. + */ +export async function buildPaneProcessFingerprint( + rows: readonly PaneFingerprintRow[], + rootPid: number, + deps: PaneFingerprintDeps = {} +): Promise { + const platform = deps.platform ?? process.platform + const index = getProcessTableIndex(rows) + const root = index.byPid.get(rootPid) + if (!root || root.pgid === undefined || root.tpgid === undefined) { + return null + } + const descendants = collectDescendantsFromIndex(index, rootPid) + const subtree = [root, ...descendants] + let startTimes: ReadonlyMap + if (platform === 'linux') { + const read = deps.readLinuxStartTime ?? readLinuxProcStartTime + const entries = await Promise.all( + subtree.map(async (row) => [row.pid, await read(row.pid)] as const) + ) + startTimes = new Map(entries) + } else { + // Collapse `lstart` padding (`Sep 3`) so both column sets stamp identically. + startTimes = new Map( + subtree.map((row) => [row.pid, row.startTime?.replace(/\s+/g, ' ') ?? null] as const) + ) + } + const rootStart = startTimes.get(rootPid) + if (!rootStart) { + return null + } + // Why every member, not just the root: a start marker is what makes a pid comparison + // recycle-safe. Stamping a missing one as a placeholder would let two captures that both + // failed to read it compare equal across a recycled pid, so a vanished agent could look + // unchanged. Refusing the fingerprint sends the caller to the full capture instead. + const members: string[] = [] + for (const row of descendants) { + const startTime = startTimes.get(row.pid) + if (!startTime) { + return null + } + members.push(`${row.pid}@${startTime}:${row.pgid ?? '?'}:${jobControlState(row.stat)}`) + } + members.sort() + return `${rootPid}@${rootStart}#${root.tpgid}:${jobControlState(root.stat)}|${members.join(',')}` +} diff --git a/src/main/providers/pty-process-inspection.ts b/src/main/providers/pty-process-inspection.ts index 59d910b2238..2c316ae05f9 100644 --- a/src/main/providers/pty-process-inspection.ts +++ b/src/main/providers/pty-process-inspection.ts @@ -23,6 +23,9 @@ type CompletionSensitivePtyProvider = IPtyProvider & { export type PtyProcessInspectionOptions = { expectedIncarnationId?: PtyIncarnationId scanChildProcesses?: boolean + /** A self-correcting cadence poll that reads only the process name: licenses a host to answer + * from a cheap capture and OMIT evidence. Never set by a caller that consumes evidence. */ + steadyState?: boolean } export async function inspectPtyProviderProcess( diff --git a/src/main/providers/pty-spawn-result.ts b/src/main/providers/pty-spawn-result.ts index 43e9665de45..90b41d9656a 100644 --- a/src/main/providers/pty-spawn-result.ts +++ b/src/main/providers/pty-spawn-result.ts @@ -16,6 +16,11 @@ export type PtySpawnResult = { sourceActivation?: PtySourceReceivingActivation /** The provider observed this exact spawn exit before returning its spawn result. */ exitedBeforeSpawnReply?: true + /** Whether the execution host armed the shell-ready marker for a renderer-delivered startup + * command. `false` means the host looked and did not (fish, sh, Windows) so the client must + * not wait; absent means the host predates the field and the client keeps its own guess. + * Never collapse absent into `false`. */ + shellReadyArmed?: boolean /** OS-level pid of the shell process, when available at spawn time. * Why: the memory collector needs this to walk each PTY's process * subtree. Daemon-backed providers return it from the RPC result; @@ -57,6 +62,10 @@ export type PtySpawnResult = { snapshotTerminalOwner?: TerminalOwner /** True when the spawn reattached to an existing daemon session. */ isReattach?: boolean + /** Grid the PTY is proven to be at once this spawn settled. Only providers whose attach + * applies the requested size set it; daemon/relay attach leave the live grid alone, so main + * must not read the requested dims back as a measurement (see `resolveCommittedPtySize`). */ + attachedGrid?: { cols: number; rows: number } /** Last OSC title tracked by the daemon session the snapshot came from. * Seeds main's terminal title records after a relaunch; never replayed * into a terminal. */ diff --git a/src/main/providers/windows-foreground-process-rows.ts b/src/main/providers/windows-foreground-process-rows.ts index 5f462649e6c..e8320a6d00a 100644 --- a/src/main/providers/windows-foreground-process-rows.ts +++ b/src/main/providers/windows-foreground-process-rows.ts @@ -108,6 +108,26 @@ export async function queryWindowsPaneProcessInventory( } } +/** + * The descendant walk over rows the caller already read. + * + * Why exported: a caller that needs a field this module's projection drops — + * process creation time, for a PID-reuse-safe teardown snapshot — would + * otherwise read the whole table a second time to get it. + * Null when the root is absent, which is a stale or filtered snapshot rather + * than a root with no descendants. + */ +export function windowsDescendantsFromRows( + rows: Row[], + rootPid: number +): (Row & { depth: number })[] | null { + const index = getProcessTableIndex(rows) + if (!index.byPid.has(rootPid)) { + return null + } + return collectDescendantsFromIndex(index, rootPid).sort((a, b) => b.depth - a.depth) +} + /** Test-only: clear the shared snapshot so one case's rows never serve the next. */ export function resetWindowsProcessRowsSnapshotForTests(): void { resetWindowsProcessTableForTests() diff --git a/src/main/pty-descendant-exit-verification.ts b/src/main/pty-descendant-exit-verification.ts index c0ed223704a..4c8471fa955 100644 --- a/src/main/pty-descendant-exit-verification.ts +++ b/src/main/pty-descendant-exit-verification.ts @@ -21,50 +21,143 @@ function waitForDelay(ms: number): Promise { function matchingSnapshotRows( snapshot: DescendantSnapshot, - table: readonly ProcessTableRow[] + table: readonly ProcessTableRow[], + rejectDuplicatePids = false ): ProcessTableRow[] { const expected = new Map(snapshot.descendants.map((row) => [row.pid, row])) - return table.filter((live) => { - const row = expected.get(live.pid) - return row?.startedAt === live.startedAt && row.pgid === live.pgid + const rowsByPid = new Map() + for (const live of table) { + const rows = rowsByPid.get(live.pid) + if (rows) { + rows.push(live) + } else { + rowsByPid.set(live.pid, [live]) + } + } + return [...expected.entries()].flatMap(([pid, row]) => { + const rows = rowsByPid.get(pid) + if (rejectDuplicatePids && rows?.length !== 1) { + // Duplicate PID rows make this non-atomic process-table read ambiguous; + // never signal or count either identity as proof of liveness. + return [] + } + return (rows ?? []).filter((live) => live.startedAt === row.startedAt && live.pgid === row.pgid) }) } +function hasDuplicateSnapshotPids( + snapshot: DescendantSnapshot, + table: readonly ProcessTableRow[] +): boolean { + const expected = new Set(snapshot.descendants.map((row) => row.pid)) + const counts = new Map() + for (const live of table) { + if (expected.has(live.pid)) { + counts.set(live.pid, (counts.get(live.pid) ?? 0) + 1) + } + } + return [...counts.values()].some((count) => count > 1) +} + type VerificationDeps = TerminateDeps & { verifyMs?: number + /** Revalidate identities before signaling; used by Claude's close proof. */ + requireIdentityBeforeSignal?: boolean } +/** + * Orca's verdict vocabulary for a snapshotted tree, with no synonyms: `live` is + * an identity-matched descendant still observed at the deadline; `unverifiable` + * is a table that could not be read, which is never evidence either way. + */ +export type DescendantTreeVerdict = 'exited' | 'live' | 'unverifiable' + /** An unreadable process table is never proof that a stopped descendant exited. */ export async function terminateDescendantSnapshotAndWait( snapshot: DescendantSnapshot, deps: VerificationDeps = {} ): Promise { + return (await terminateDescendantSnapshotWithVerdict(snapshot, deps)) === 'exited' +} + +/** Signals the snapshot, then reports what the last table read observed. */ +export async function terminateDescendantSnapshotWithVerdict( + snapshot: DescendantSnapshot, + deps: VerificationDeps = {} +): Promise { const sendSignal = deps.sendSignal ?? sendDescendantSignal const readTable = deps.readTable ?? readProcessTable const graceMs = deps.graceMs ?? DESCENDANT_KILL_GRACE_MS const verifyMs = deps.verifyMs ?? DESCENDANT_KILL_VERIFY_MS const deadline = Date.now() + verifyMs - for (const row of snapshot.descendants) { - sendSignal(row.pid, 'SIGTERM') - } let forced = false + let signalled = !deps.requireIdentityBeforeSignal + let missingObservations = 0 + if (signalled) { + for (const row of snapshot.descendants) { + sendSignal(row.pid, 'SIGTERM') + } + } while (Date.now() < deadline) { const capture = await readProcessTableBeforeDeadline( readTable, deps.timeoutMs ?? DESCENDANT_SNAPSHOT_TIMEOUT_MS ) - if (!capture) { - return false - } - const live = matchingSnapshotRows(snapshot, capture.rows) - if (live.length === 0) { - return true - } - if (!forced && Date.now() >= deadline - verifyMs + graceMs) { - forced = true - for (const row of live) { - if (hasUnambiguousStartIdentity(row, snapshot.capturedAtMs)) { - sendSignal(row.pid, 'SIGKILL') + // A read that missed its own deadline is not an answer, and surrendering on + // the first slow one spends none of the window this verification was given: + // on a loaded host that reported a tree unverifiable without ever seeing it. + if (capture) { + if (deps.requireIdentityBeforeSignal && hasDuplicateSnapshotPids(snapshot, capture.rows)) { + // A duplicate target pid is an ambiguous non-atomic read. Do not signal + // either row and do not turn that uncertainty into an exited verdict. + await waitForDelay(50) + continue + } + const live = matchingSnapshotRows(snapshot, capture.rows, deps.requireIdentityBeforeSignal) + if (live.length === 0) { + // Before a signal has been sent, an empty identity match means the + // snapshotted descendants already exited or were replaced. Signalling + // those old numeric pids would be unsafe. + if (deps.requireIdentityBeforeSignal) { + // A single process-table read can race a fork or return a partial + // view; require two bounded absences before claiming the tree gone. + missingObservations += 1 + if (missingObservations < 2) { + await waitForDelay(50) + continue + } + } + return 'exited' + } + missingObservations = 0 + if (!signalled) { + // Revalidate every identity immediately before the first signal. A PID + // can be recycled between the original walk and close, so never signal + // from the stale snapshot alone. + for (const row of live) { + sendSignal(row.pid, 'SIGTERM') + } + signalled = true + } + if (!forced && Date.now() >= deadline - verifyMs + graceMs) { + forced = true + for (const row of live) { + // A row a walk re-derived from a live root is ours whatever second it + // was born in, which start time alone can never establish for one born + // in its own capture second. Rows no walk re-derived still answer to + // the second-resolution fence, which is all the evidence they have. + // Scoped to the identity-revalidating callers; the same argument holds + // for the rest, but widening it is a deliberate change of its own. + if ( + (deps.requireIdentityBeforeSignal === true && + snapshot.reDerivedPids?.has(row.pid) === true) || + hasUnambiguousStartIdentity( + row, + snapshot.capturedAtMsByPid?.[String(row.pid)] ?? snapshot.capturedAtMs + ) + ) { + sendSignal(row.pid, 'SIGKILL') + } } } } @@ -74,5 +167,19 @@ export async function terminateDescendantSnapshotAndWait( readTable, deps.timeoutMs ?? DESCENDANT_SNAPSHOT_TIMEOUT_MS ) - return finalCapture !== null && matchingSnapshotRows(snapshot, finalCapture.rows).length === 0 + if (!finalCapture) { + return 'unverifiable' + } + if (deps.requireIdentityBeforeSignal && hasDuplicateSnapshotPids(snapshot, finalCapture.rows)) { + return 'unverifiable' + } + const finalLive = matchingSnapshotRows( + snapshot, + finalCapture.rows, + deps.requireIdentityBeforeSignal + ) + if (finalLive.length > 0) { + return 'live' + } + return deps.requireIdentityBeforeSignal && missingObservations < 2 ? 'unverifiable' : 'exited' } diff --git a/src/main/pty-descendant-termination.test.ts b/src/main/pty-descendant-termination.test.ts index e1255a678d8..0c4bea81306 100644 --- a/src/main/pty-descendant-termination.test.ts +++ b/src/main/pty-descendant-termination.test.ts @@ -15,7 +15,10 @@ import { type ProcessTableCapture, type ProcessTableRow } from './pty-descendant-termination' -import { terminateDescendantSnapshotAndWait } from './pty-descendant-exit-verification' +import { + terminateDescendantSnapshotAndWait, + terminateDescendantSnapshotWithVerdict +} from './pty-descendant-exit-verification' const CAPTURED_AT_MS = Date.parse('Tue Jul 14 12:00:00 2026') @@ -53,7 +56,14 @@ function snapshot( rootPgid: number | null = 10, capturedAtMs = CAPTURED_AT_MS ) { - return { rootPgid, descendants, capturedAtMs } + return { + ...(rootPgid === null ? {} : { root: { pid: 10, startedAt: 'Mon Jul 13 12:54:47 2026' } }), + rootPgid, + descendants, + capturedAtMs, + // Everything a walk returns was re-derived by it. + ...(rootPgid === null ? {} : { reDerivedPids: new Set(descendants.map((row) => row.pid)) }) + } } describe('parseProcessTable', () => { @@ -298,6 +308,32 @@ describe('terminateDescendantSnapshot', () => { expect(sendSignal).not.toHaveBeenCalled() expect(vi.getTimerCount()).toBe(0) }) + + it("uses each row's capture boundary when escalating a merged snapshot", async () => { + const oldBoundary = CAPTURED_AT_MS + 900 + const refreshBoundary = CAPTURED_AT_MS + 2_100 + const retained = row(20, 10, 20, 'Tue Jul 14 12:00:00 2026') + const fresh = row(30, 10, 30, 'Tue Jul 14 12:00:01 2026') + const sendSignal = vi.fn() + terminateDescendantSnapshot( + { + ...snapshot([retained, fresh], 10, refreshBoundary), + capturedAtMsByPid: { '20': oldBoundary, '30': refreshBoundary } + }, + { + sendSignal, + readTable: vi.fn().mockResolvedValue(tableCapture([retained, fresh])) + } + ) + sendSignal.mockClear() + + await vi.advanceTimersByTimeAsync(DESCENDANT_KILL_GRACE_MS) + + // PID 20 was retained from the earlier capture and is still in its + // capture second; PID 30 was newly observed by the refresh and is old + // enough for a bounded forced cleanup. + expect(sendSignal.mock.calls).toEqual([[30, 'SIGKILL']]) + }) }) describe('terminateDescendantSnapshotAndWait', () => { @@ -333,14 +369,114 @@ describe('terminateDescendantSnapshotAndWait', () => { it('does not claim exit when the verification table is unavailable', async () => { const sendSignal = vi.fn() - const result = await terminateDescendantSnapshotAndWait(snapshot([row(20, 10, 20)]), { + const pending = terminateDescendantSnapshotAndWait(snapshot([row(20, 10, 20)]), { sendSignal, - readTable: vi.fn().mockRejectedValue(new Error('ps exploded')) + readTable: vi.fn().mockRejectedValue(new Error('ps exploded')), + verifyMs: 200 }) + await vi.advanceTimersByTimeAsync(400) - expect(result).toBe(false) + await expect(pending).resolves.toBe(false) expect(sendSignal).toHaveBeenCalledWith(20, 'SIGTERM') }) + + it('keeps polling past a read that missed its deadline rather than surrendering', async () => { + const survivor = row(20, 10, 20) + const readTable = vi + .fn() + // A loaded host can miss one read's deadline with the window still open. + .mockRejectedValueOnce(new Error('ps timed out')) + .mockResolvedValueOnce(tableCapture([survivor])) + .mockResolvedValueOnce(tableCapture([])) + const sendSignal = vi.fn() + + const pending = terminateDescendantSnapshotWithVerdict(snapshot([survivor]), { + sendSignal, + readTable, + graceMs: 0, + verifyMs: 2_000 + }) + await vi.advanceTimersByTimeAsync(500) + + await expect(pending).resolves.toBe('exited') + expect(sendSignal.mock.calls).toEqual([ + [20, 'SIGTERM'], + [20, 'SIGKILL'] + ]) + }) + + it('names a survivor seen at the deadline live, never unverifiable', async () => { + const survivor = row(20, 10, 20) + const pending = terminateDescendantSnapshotWithVerdict(snapshot([survivor]), { + sendSignal: vi.fn(), + readTable: vi.fn().mockResolvedValue(tableCapture([survivor])), + graceMs: 0, + verifyMs: 100 + }) + await vi.advanceTimersByTimeAsync(200) + + await expect(pending).resolves.toBe('live') + }) + + it('names an unreadable verification table unverifiable', async () => { + const pending = terminateDescendantSnapshotWithVerdict(snapshot([row(20, 10, 20)]), { + sendSignal: vi.fn(), + readTable: vi.fn().mockRejectedValue(new Error('ps exploded')), + verifyMs: 200 + }) + await vi.advanceTimersByTimeAsync(400) + + await expect(pending).resolves.toBe('unverifiable') + }) + + it('does not signal a recycled descendant when identity validation is required', async () => { + const sendSignal = vi.fn() + const recycled = row(20, 10, 20, 'Tue Jul 14 13:00:00 2026') + const pending = terminateDescendantSnapshotWithVerdict( + snapshot([row(20, 10, 20, 'Tue Jul 14 12:00:00 2026')]), + { + sendSignal, + readTable: vi.fn().mockResolvedValue(tableCapture([recycled])), + requireIdentityBeforeSignal: true, + verifyMs: 100 + } + ) + + await vi.advanceTimersByTimeAsync(200) + await expect(pending).resolves.toBe('exited') + expect(sendSignal).not.toHaveBeenCalled() + }) + + it('uses row-scoped boundaries for forced cleanup in the exit verifier', async () => { + const oldBoundary = CAPTURED_AT_MS + 900 + const refreshBoundary = CAPTURED_AT_MS + 2_100 + const retained = row(20, 10, 20, 'Tue Jul 14 12:00:00 2026') + const fresh = row(30, 10, 30, 'Tue Jul 14 12:00:01 2026') + const sendSignal = vi.fn() + const pending = terminateDescendantSnapshotWithVerdict( + { + ...snapshot([retained, fresh], 10, refreshBoundary), + capturedAtMsByPid: { '20': oldBoundary, '30': refreshBoundary }, + // What a merge produces: only the refresh re-derived 30; 20 is retained. + reDerivedPids: new Set([30]) + }, + { + sendSignal, + readTable: vi.fn().mockResolvedValue(tableCapture([retained, fresh])), + requireIdentityBeforeSignal: true, + graceMs: 0, + verifyMs: 100 + } + ) + await vi.advanceTimersByTimeAsync(200) + + await expect(pending).resolves.toBe('live') + expect(sendSignal.mock.calls).toEqual([ + [20, 'SIGTERM'], + [30, 'SIGTERM'], + [30, 'SIGKILL'] + ]) + }) }) describe('createProcessTableSnapshotReader', () => { diff --git a/src/main/pty-descendant-termination.ts b/src/main/pty-descendant-termination.ts index bf254d03b56..4f56c3e977b 100644 --- a/src/main/pty-descendant-termination.ts +++ b/src/main/pty-descendant-termination.ts @@ -21,12 +21,25 @@ export type ProcessTableRow = { startedAt: string } +export type PosixProcessIdentity = Pick + export type DescendantSnapshot = { + /** Identity of the root observed in the same process-table capture. */ + root?: PosixProcessIdentity rootPgid: number | null descendants: ProcessTableRow[] - /** Wall-clock boundary for deciding whether ps's second-resolution lstart - * can safely distinguish this process from a later PID reuse. */ + /** Wall-clock boundary for an unmerged snapshot (or legacy callers). */ capturedAtMs: number + /** Per-PID identity boundaries for merged captures. */ + capturedAtMsByPid?: Readonly> + /** + * PIDs this walk re-derived from a live root. A ppid walk only reaches what + * the root actually parents, so membership is proof of ownership that owes + * nothing to `lstart`'s one-second resolution: a stranger would have to have + * been forked into our own tree, and then it is not a stranger. Rows a merge + * retained from an earlier walk are absent, and still answer to start time. + */ + reDerivedPids?: ReadonlySet } export type ProcessTableCapture = { @@ -156,9 +169,13 @@ export function collectDescendantRows( ): DescendantSnapshot { const childrenByPpid = new Map() let rootRow: ProcessTableRow | null = null + let duplicateRoot = false for (const row of table) { if (row.pid === rootPid) { - rootRow = row + // A non-atomic process-table read can contain both an old and a recycled + // root row. There is no safe identity to retain in that case. + duplicateRoot = rootRow !== null + rootRow ??= row continue } const siblings = childrenByPpid.get(row.ppid) @@ -172,7 +189,7 @@ export function collectDescendantRows( // An absent root has already exited — its real descendants reparent to pid 1 and // become unreachable by ppid, so any rows still pointing at the vacated PID are a // PID-reuse coincidence. Sweeping them could signal an unrelated process, so bail. - if (!rootRow) { + if (!rootRow || duplicateRoot) { return { rootPgid: null, descendants: [], capturedAtMs } } const descendants: ProcessTableRow[] = [] @@ -191,7 +208,13 @@ export function collectDescendantRows( queue.push(child.pid) } } - return { rootPgid: rootRow.pgid, descendants, capturedAtMs } + return { + root: { pid: rootRow.pid, startedAt: rootRow.startedAt }, + rootPgid: rootRow.pgid, + descendants, + capturedAtMs, + reDerivedPids: new Set(descendants.map((row) => row.pid)) + } } type SnapshotDeps = { @@ -316,7 +339,11 @@ export type TerminateDeps = { } export function hasUnambiguousStartIdentity(row: ProcessTableRow, capturedAtMs: number): boolean { - const startedAtMs = Date.parse(row.startedAt) + return hasUnambiguousStartTime(row.startedAt, capturedAtMs) +} + +export function hasUnambiguousStartTime(startedAt: string, capturedAtMs: number): boolean { + const startedAtMs = Date.parse(startedAt) if (!Number.isFinite(startedAtMs)) { return false } @@ -364,7 +391,10 @@ export function terminateDescendantSnapshot( for (const row of snapshot.descendants) { const live = liveTargets.get(row.pid) if ( - hasUnambiguousStartIdentity(row, snapshot.capturedAtMs) && + hasUnambiguousStartIdentity( + row, + snapshot.capturedAtMsByPid?.[String(row.pid)] ?? snapshot.capturedAtMs + ) && live?.startedAt === row.startedAt && live.pgid === row.pgid ) { diff --git a/src/main/refused-tree-kill-root-termination.test.ts b/src/main/refused-tree-kill-root-termination.test.ts new file mode 100644 index 00000000000..ada4b5942a9 --- /dev/null +++ b/src/main/refused-tree-kill-root-termination.test.ts @@ -0,0 +1,220 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { spawnMock, execFileMock, queryWindowsProcessDescendantsMock } = vi.hoisted(() => ({ + spawnMock: vi.fn(), + execFileMock: vi.fn(), + queryWindowsProcessDescendantsMock: vi.fn() +})) + +vi.mock('node:child_process', async (importOriginal) => ({ + ...(await importOriginal>()), + spawn: spawnMock, + execFile: execFileMock +})) +vi.mock('electron', () => ({ ipcMain: { handle: vi.fn(), on: vi.fn() } })) +vi.mock('./providers/windows-foreground-process-rows', () => ({ + queryWindowsProcessDescendants: queryWindowsProcessDescendantsMock +})) + +import { + getAppEnvironment, + hasAppEnvironment, + setAppEnvironment, + type AppEnvironment +} from '../shared/app-environment' +import { installMainProcessTreeKillGate } from './own-chromium-tree-kill-guard' +import { setProcessTreeKillGate } from '../shared/child-process/process-tree-kill-gate' +import { resetSelfInitiatedTreeKillLogForTest } from './crash-reporting/self-initiated-tree-kill-log' +import { + clearCrashBreadcrumbsForTest, + getCrashBreadcrumbSnapshot +} from './crash-reporting/crash-breadcrumb-store' +import { _resetTracerForTests, setActiveSink } from './observability/tracer' +import { terminateNotebookProcessTree } from './ipc/notebook' +import { killLocalPrecheckProcessTree } from './automations/precheck-runner' +import { killRecipeProcess } from '../shared/ephemeral-vm-recipe-process' +import { killSpawnedCommandTree } from './git/command-runner/spawned-command-tree-kill' +import { killCodexAppServerProcessTree } from './codex/codex-app-server-session' +import { signalProcessTree } from '../shared/child-process/process-tree-termination' +import { killSourceControlAgentProcess } from './text-generation/source-control-local-process' +import { terminateCodexTurnProcesses } from './codex/codex-structured-turn-processes' + +/** A pid Electron reports as one of ours: every gate below must refuse it. */ +const RENDERER_PID = 1001 + +function appEnvironment(): AppEnvironment { + return { + getPath: () => process.cwd(), + getAppPath: () => process.cwd(), + getVersion: () => '0.0.0-test', + isPackaged: () => false, + onWillQuit: () => {}, + exit: () => {}, + getAppMetrics: (() => [ + { pid: RENDERER_PID, type: 'Tab' } + ]) as unknown as AppEnvironment['getAppMetrics'] + } +} + +let previousEnvironment: AppEnvironment | null = null +let previousPlatform: PropertyDescriptor | undefined + +function setPlatform(platform: NodeJS.Platform): void { + Object.defineProperty(process, 'platform', { value: platform, configurable: true }) +} + +beforeEach(() => { + previousEnvironment = hasAppEnvironment() ? getAppEnvironment() : null + previousPlatform = Object.getOwnPropertyDescriptor(process, 'platform') + setAppEnvironment(appEnvironment()) + setActiveSink(null) + clearCrashBreadcrumbsForTest() + resetSelfInitiatedTreeKillLogForTest() + installMainProcessTreeKillGate() + spawnMock.mockReset() + execFileMock.mockReset() + queryWindowsProcessDescendantsMock.mockReset() + spawnMock.mockReturnValue({ on: vi.fn(), once: vi.fn(), unref: vi.fn(), kill: vi.fn() }) +}) + +afterEach(() => { + setProcessTreeKillGate(null) + if (previousPlatform) { + Object.defineProperty(process, 'platform', previousPlatform) + } + if (previousEnvironment) { + setAppEnvironment(previousEnvironment) + } + _resetTracerForTests() +}) + +/** + * A refusal must never become a process leak. The gate only blocks the + * pid-addressed tree walk; the root kill is addressed by the child handle, so it + * cannot reach the recycled pid we refused, and skipping it would report a + * timed-out command as stopped while its tree keeps running. + */ +describe('a refused tree-kill still terminates the root it owns', () => { + it('kills the git command root when the tree walk is refused', async () => { + setPlatform('win32') + const child = { pid: RENDERER_PID, kill: vi.fn() } + + await killSpawnedCommandTree(child as never) + + expect(spawnMock).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledTimes(1) + }) + + it('kills the notebook cell root when the tree walk is refused', () => { + setPlatform('win32') + const child = { pid: RENDERER_PID, kill: vi.fn() } + + expect(terminateNotebookProcessTree(child as never)).toBeNull() + + expect(spawnMock).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledTimes(1) + }) + + it('kills the automation precheck root when the tree walk is refused', () => { + setPlatform('win32') + const child = { pid: RENDERER_PID, kill: vi.fn() } + + expect(killLocalPrecheckProcessTree(child as never)).toBeNull() + + expect(spawnMock).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledTimes(1) + }) + + it('kills the ephemeral-VM recipe root when the tree walk is refused', () => { + setPlatform('win32') + const child = { pid: RENDERER_PID, kill: vi.fn() } + + killRecipeProcess(child as never, true) + + expect(spawnMock).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('kills the codex app-server root when the deadline tree walk is refused', () => { + const child = { pid: RENDERER_PID, kill: vi.fn() } + + killCodexAppServerProcessTree(child as never, { + platform: 'win32', + spawnImpl: spawnMock as never + }) + + expect(spawnMock).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('kills the commit-message agent root when the tree walk is refused', async () => { + setPlatform('win32') + const child = { pid: RENDERER_PID, kill: vi.fn() } + + await killSourceControlAgentProcess(child as never) + + expect(execFileMock).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('kills the runProcess root when the Windows arm of the shared choke point is refused', async () => { + setPlatform('win32') + const windowsChild = { pid: RENDERER_PID, kill: vi.fn(), exitCode: null, signalCode: null } + + await expect(signalProcessTree(windowsChild as never, 'SIGKILL')).resolves.toBe(false) + expect(spawnMock).not.toHaveBeenCalled() + expect(windowsChild.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('still signals the POSIX process group: a group only holds what Orca put in it', async () => { + // Same contract as main and as the other three POSIX group arms in main + // (claude-login, codex teardown, PTY sweep): record, never refuse. A stale + // `getAppMetrics()` entry must not orphan a macOS/Linux tree. + setPlatform('linux') + const posixChild = { pid: RENDERER_PID, kill: vi.fn(), exitCode: null, signalCode: null } + const processKill = vi.spyOn(process, 'kill').mockImplementation(() => true) + + await expect(signalProcessTree(posixChild as never, 'SIGKILL')).resolves.toBe(true) + expect(processKill).toHaveBeenCalledWith(-RENDERER_PID, 'SIGKILL') + expect(posixChild.kill).not.toHaveBeenCalled() + expect(getCrashBreadcrumbSnapshot()).toEqual([ + expect.objectContaining({ + name: 'self_tree_kill', + data: expect.objectContaining({ pid: RENDERER_PID, scope: 'posix-process-group' }) + }) + ]) + processKill.mockRestore() + }) +}) + +/** + * The one gated site with nothing to fall back to: the roots it kills are found + * by a process-table walk, not spawned here, so there is no child handle. A + * refusal must then be visible — the refusal crumb is written and the turn is + * reported as not cancelled — rather than resolving as if the tree had gone. + */ +describe('a refused tree-kill with no handle to fall back to', () => { + it('reports the codex turn as not cancelled and records the refused added root', async () => { + const appServerPid = 500 + const addedRoot = { + pid: RENDERER_PID, + ppid: appServerPid, + name: 'node.exe', + command: 'node', + depth: 1 + } + queryWindowsProcessDescendantsMock.mockResolvedValue([addedRoot]) + + await expect( + terminateCodexTurnProcesses(appServerPid, { platform: 'win32', identities: new Map() }) + ).resolves.toBe(false) + + expect(execFileMock).not.toHaveBeenCalled() + expect(getCrashBreadcrumbSnapshot()).toEqual([ + expect.objectContaining({ + name: 'self_tree_kill_refused_own_chromium', + data: expect.objectContaining({ pid: RENDERER_PID, site: 'codex-turn-added-roots' }) + }) + ]) + }) +}) diff --git a/src/main/repo-worktrees.ts b/src/main/repo-worktrees.ts index 228f303bfa2..f5d67523286 100644 --- a/src/main/repo-worktrees.ts +++ b/src/main/repo-worktrees.ts @@ -1,6 +1,11 @@ import type { Repo } from '../shared/repo-types' import type { GitWorktreeInfo } from '../shared/worktree/types' -import { listWorktreeGraph, listWorktrees, listWorktreesStrict } from './git/worktree' +import { + listWorktreeGraph, + listWorktrees, + listWorktreesSharedStrictAllowingTrueEmpty, + listWorktreesStrict +} from './git/worktree' import { isFolderRepo } from '../shared/repo-kind' import { getRepoExecutionHostId, LOCAL_EXECUTION_HOST_ID } from '../shared/execution-host' import { resolveGitRouteForHost } from './providers/execution-host-provider-dispatch' @@ -42,6 +47,29 @@ export function createFolderWorktree(repo: Repo): GitWorktreeInfo { export async function listRepoWorktrees( repo: Repo, options?: LocalRepoWorktreeListOptions +): Promise { + return listRoutedRepoWorktrees(repo, options, listWorktrees) +} + +/** + * The detected scan's listing: a Git or host failure rejects instead of softening to `[]`, so a + * failed scan cannot be published as an authoritative empty listing and prune the repo's worktrees + * (#1158's retention guard only fires when the listing admits it failed). + */ +export async function listRepoWorktreesForDetectedScan( + repo: Repo, + options?: LocalRepoWorktreeListOptions +): Promise { + return listRoutedRepoWorktrees(repo, options, listWorktreesSharedStrictAllowingTrueEmpty) +} + +async function listRoutedRepoWorktrees( + repo: Repo, + options: LocalRepoWorktreeListOptions | undefined, + listLocal: ( + repoPath: string, + options?: LocalRepoWorktreeListOptions + ) => Promise ): Promise { if (isFolderRepo(repo)) { return [createFolderWorktree(repo)] @@ -66,8 +94,8 @@ export async function listRepoWorktrees( return await route.provider.listWorktrees(repo.path) } return hasLocalRepoWorktreeListOptions(options) - ? await listWorktrees(repo.path, options) - : await listWorktrees(repo.path) + ? await listLocal(repo.path, options) + : await listLocal(repo.path) } /** diff --git a/src/main/runtime/agent-session-acquisition-failure-settlement.ts b/src/main/runtime/agent-session-acquisition-failure-settlement.ts index 7ad20397813..c790a149f31 100644 --- a/src/main/runtime/agent-session-acquisition-failure-settlement.ts +++ b/src/main/runtime/agent-session-acquisition-failure-settlement.ts @@ -4,10 +4,28 @@ import { type AgentSessionOperationOutcome } from '../../shared/agent-session-operation-ledger' import { nextAgentSessionFence } from '../../shared/agent-session-next-fence' -import type { AgentSessionRecord } from '../../shared/agent-session-record' +import type { + AgentSessionDeathEvidence, + AgentSessionRecord +} from '../../shared/agent-session-record' import { assertFence, withLease } from './agent-session-lease-transitions' import type { AgentSessionStoreState } from './agent-session-record-store-file' +/** + * How the failed attempt's provider process was accounted for. + * - `exit-proven`: cleanup observed the whole tree gone. + * - `root-exit-observed`: the owner root's exit was observed first-hand, so the + * identity this lease is keyed on is dead, but its descendants could not be + * verified. Releases the lease and says exactly that, claiming nothing more. + * - `processless`: the attempt failed before a process existed. + * - `unproven`: nothing about the process was observed; the reservation latches. + */ +export type AgentSessionAcquisitionExitProof = + | 'exit-proven' + | 'root-exit-observed' + | 'processless' + | 'unproven' + export type AgentSessionFailedAcquisitionSettlement = { sessionId: string fence: number @@ -15,7 +33,7 @@ export type AgentSessionFailedAcquisitionSettlement = { callerKey: string operationId: string outcome: Extract - exitProof: 'exit-proven' | 'processless' | 'unproven' + exitProof: AgentSessionAcquisitionExitProof now: number } @@ -83,11 +101,18 @@ export function settleFailedAgentSessionPostAcquisitionAttachment( claimStatus: 'released', lastRenewedAt: args.now, handoffOperationId: null, - deathEvidence: { - kind: 'exit-observed', - detail: 'post-acquisition cleanup proved no provider child remains', - observedAt: args.now - } + deathEvidence: + args.exitProof === 'root-exit-observed' + ? { + kind: 'exit-observed', + detail: 'the provider process exited; its descendants were not verifiable', + observedAt: args.now + } + : { + kind: 'exit-observed', + detail: 'post-acquisition cleanup proved no provider child remains', + observedAt: args.now + } }) state.records.set(args.sessionId, next) state.operations = settleAgentSessionOperation(state.operations, args) @@ -127,18 +152,29 @@ function settleFailedLease( claimStatus: 'released', lastRenewedAt: args.now, handoffOperationId: null, - deathEvidence: - args.exitProof === 'processless' - ? { - kind: 'pid-absent', - detail: 'reservation failed before spawn', - observedAt: args.now - } - : { - // Cleanup proved no child of this attempt remains; it may never have spawned. - kind: 'exit-observed', - detail: 'acquisition cleanup proved no provider child remains', - observedAt: args.now - } + deathEvidence: acquisitionDeathEvidence(args.exitProof, args.now) }) } + +/** Records only what was observed: never a tree claim the cleanup did not make. */ +function acquisitionDeathEvidence( + exitProof: AgentSessionAcquisitionExitProof, + observedAt: number +): AgentSessionDeathEvidence { + if (exitProof === 'processless') { + return { kind: 'pid-absent', detail: 'reservation failed before spawn', observedAt } + } + if (exitProof === 'root-exit-observed') { + return { + kind: 'exit-observed', + detail: 'the provider process exited; its descendants were not verifiable', + observedAt + } + } + // Cleanup proved no child of this attempt remains; it may never have spawned. + return { + kind: 'exit-observed', + detail: 'acquisition cleanup proved no provider child remains', + observedAt + } +} diff --git a/src/main/runtime/agent-session-launch-env-backfill.test.ts b/src/main/runtime/agent-session-launch-env-backfill.test.ts new file mode 100644 index 00000000000..c95705f2aa5 --- /dev/null +++ b/src/main/runtime/agent-session-launch-env-backfill.test.ts @@ -0,0 +1,90 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { AgentSessionRecordStore } from './agent-session-record-store' +import type { AgentSessionReserveRequest } from './agent-session-reservation-admission' + +const NOW = 1_800_000_000_000 +const SESSION = 'session-launch-env' +let directory: string + +function request(overrides: Partial = {}): AgentSessionReserveRequest { + return { + sessionId: SESSION, + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }, + provider: 'claude', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/home/dev/.claude' }, + runtimeKind: 'native', + expectedFence: null, + spawnToken: 'spawn-a', + claimKeyId: 'key-1', + handoffOperationId: null, + probe: { outcome: 'reservation-unused' }, + operation: { + callerKey: 'client-1', + operationId: `${NOW}-00000000000000000000000000000001`, + fingerprint: 'fp-1' + }, + now: NOW, + ...overrides + } +} + +beforeEach(async () => { + directory = await mkdtemp(join(tmpdir(), 'orca-agent-session-launch-env-')) +}) + +afterEach(async () => { + await rm(directory, { recursive: true, force: true }) +}) + +describe('legacy agent session launch environment', () => { + it('durably pins the first environment resolved by a current reservation', async () => { + const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + await store.reserveOwner(request()) + await store.reserveOwner( + request({ + expectedFence: 1, + spawnToken: 'spawn-b', + launchEnv: { ANTHROPIC_AUTH_TOKEN: 'pinned-token' }, + operation: { + callerKey: 'client-1', + operationId: `${NOW}-00000000000000000000000000000002`, + fingerprint: 'fp-2' + } + }) + ) + + const reopened = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + expect( + (reopened.getRecord(SESSION) as { launchEnv?: Record } | null)?.launchEnv + ).toBeUndefined() + }) + + it('rejects an environment that could not be reloaded before writing it', async () => { + const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + const launchEnv = Object.fromEntries( + Array.from({ length: 257 }, (_, index) => [`KEY_${index}`, 'value']) + ) + + await expect(store.reserveOwner(request({ launchEnv }))).rejects.toThrow( + 'agent_session_launch_env_invalid' + ) + expect(store.getRecord(SESSION)).toBeNull() + }) + + it('rejects an overlong environment key before writing it', async () => { + const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + + await expect( + store.reserveOwner(request({ launchEnv: { ['K'.repeat(513)]: 'value' } })) + ).rejects.toThrow('agent_session_launch_env_invalid') + expect(store.getRecord(SESSION)).toBeNull() + }) +}) diff --git a/src/main/runtime/agent-session-record-options.test.ts b/src/main/runtime/agent-session-record-options.test.ts index a1dfb9ccdef..5795763d96c 100644 --- a/src/main/runtime/agent-session-record-options.test.ts +++ b/src/main/runtime/agent-session-record-options.test.ts @@ -31,6 +31,20 @@ it('fails option hydration before ownership can be proved', async () => { ).rejects.toThrow('model list unavailable') }) +it('drops provider-rejected persisted options before the next owner proof', async () => { + await expect( + readNativeSessionOptions({ + adapter: { + readOptions: async () => ({ models: [], current: { model: 'provider-model' } }), + readOptionRestoreFailures: () => ['permissionMode'] + }, + sessionId: SESSION, + fence: 2, + priorOptions: { permissionMode: 'retired-mode', other: 'keep' } + }) + ).resolves.toEqual({ model: 'provider-model', other: 'keep' }) +}) + it('persists resumed provider options atomically with owner proof', async () => { const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) const reserved = await store.reserveOwner({ diff --git a/src/main/runtime/agent-session-resume-args.test.ts b/src/main/runtime/agent-session-resume-args.test.ts new file mode 100644 index 00000000000..db4d0b07b8d --- /dev/null +++ b/src/main/runtime/agent-session-resume-args.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, it } from 'vitest' +import { resolveAgentSessionResumeArgs } from './agent-session-resume-args' + +describe('agent session resume arguments', () => { + it('keeps the session creation arguments after mutable defaults change', () => { + expect( + resolveAgentSessionResumeArgs({ + persistedArgs: ['--model', 'claude-created'], + defaultArgs: '--model claude-current', + shell: 'posix' + }) + ).toBe("'--model' 'claude-created'") + }) + + it('keeps an explicit empty snapshot when defaults are toggled off', () => { + expect( + resolveAgentSessionResumeArgs({ + persistedArgs: [], + defaultArgs: '--dangerously-skip-permissions', + shell: 'posix' + }) + ).toBe('') + }) + + it('uses current defaults for legacy records without a snapshot', () => { + expect( + resolveAgentSessionResumeArgs({ + defaultArgs: '--dangerously-skip-permissions', + shell: 'posix' + }) + ).toBe('--dangerously-skip-permissions') + }) +}) diff --git a/src/main/runtime/agent-session-resume-args.ts b/src/main/runtime/agent-session-resume-args.ts new file mode 100644 index 00000000000..dc276726dc6 --- /dev/null +++ b/src/main/runtime/agent-session-resume-args.ts @@ -0,0 +1,17 @@ +import type { AgentSessionLaunchArgs } from '../../shared/agent-session-record' +import { quoteStartupArg, type AgentStartupShell } from '../../shared/tui-agent-startup-shell' + +export function resolveAgentSessionResumeArgs(input: { + requestArgs?: string | null + persistedArgs?: AgentSessionLaunchArgs + defaultArgs?: string | null + shell: AgentStartupShell +}): string | null | undefined { + if (input.requestArgs !== undefined) { + return input.requestArgs + } + if (input.persistedArgs !== undefined) { + return input.persistedArgs.map((arg) => quoteStartupArg(arg, input.shell)).join(' ') + } + return input.defaultArgs +} diff --git a/src/main/runtime/claude-structured-session-integration.test.ts b/src/main/runtime/claude-structured-session-integration.test.ts new file mode 100644 index 00000000000..2e42d595b04 --- /dev/null +++ b/src/main/runtime/claude-structured-session-integration.test.ts @@ -0,0 +1,746 @@ +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../shared/agent-session-mutation-envelope' +import type { AgentJournalRenderItem } from '../../shared/agent-session-journal-types' +import type { AgentSessionSubscribeEvent } from '../../shared/agent-session-wire' +import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../shared/protocol-version' +import type { + ClaudeStreamJsonConnection, + ClaudeStreamJsonConnectionHandlers, + ClaudeStreamJsonLaunch, + openClaudeStreamJsonConnection +} from '../claude/claude-stream-json-connection' +import { claudeSessionIdForOrcaSession } from '../claude/claude-structured-launch-resolution' +import { + CLAUDE_SPAWN_TOKEN_ENV, + claudeProviderHandleLink +} from '../claude/claude-structured-owner-identity' +import { attachFingerprintFields } from '../native-chat/agent-session-wire/structured-agent-session-attach' +import { getStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' +import type { + StructuredAgentSessionHandoffTransport, + StructuredTuiOwner +} from '../native-chat/agent-session-wire/structured-agent-session-handoff-types' +import type { OrcaRuntimeService } from './orca-runtime' +import type { RpcRequest, RpcResponse } from './rpc/core' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { RpcDispatcher } from './rpc/dispatcher' +import { STRUCTURED_AGENT_SESSION_METHODS } from './rpc/methods/structured-agent-session' +import { + ensureStructuredAgentSessionHost, + stopStructuredAgentSessionRuntime, + waitForStructuredAgentSessionRecovery +} from './structured-agent-session-runtime' + +const SESSION = 'claude-integration-1' +const PROVIDER_SESSION = claudeSessionIdForOrcaSession(SESSION) +const WORKSPACE = 'workspace-claude' +// Why 'runtime': this file exercises the Claude structured integration over agentSession.*, not the +// mobile surface — nothing here asserts anything mobile-specific, and its sibling integration +// suites use 'runtime' too. Mobile additionally requires the experimental structured-chat setting, +// which structured-agent-session.test.ts pins in both its satisfied and refused states. +const CLIENT = { + clientKind: 'runtime' as const, + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] +} + +const { readClaudeTranscriptLeafUuid, resolveSessionFilePath } = vi.hoisted(() => ({ + readClaudeTranscriptLeafUuid: vi.fn(), + resolveSessionFilePath: vi.fn() +})) + +vi.mock('../native-chat/session-file-resolver', () => ({ + readClaudeTranscriptLeafUuid, + resolveSessionFilePath +})) + +type FakeClaudeConnection = Omit & { + closed: boolean + exitVerdict: ClaudeStreamJsonConnection['exitVerdict'] + launch: ClaudeStreamJsonLaunch + handlers: ClaudeStreamJsonConnectionHandlers + calls: { subtype: string; params?: Record }[] + sent: Record[] +} + +function fakeClaude() { + const connections: FakeClaudeConnection[] = [] + let initializeAccount: unknown + /** A child that dies during start, with the close verdict its ladder observed. */ + let selfExit: { message: string; exitVerdict: ClaudeStreamJsonConnection['exitVerdict'] } | null = + null + const openConnection = (async (launch, handlers = {}) => { + const connection: FakeClaudeConnection = { + launch, + handlers, + calls: [], + sent: [], + pid: 4321 + connections.length, + closed: false, + initializationResult: async () => { + connection.calls.push({ subtype: 'initialize' }) + if (selfExit) { + handlers.onExit?.(new Error(selfExit.message)) + return { models: [] } + } + handlers.onMessage?.({ + type: 'system', + subtype: 'init', + session_id: PROVIDER_SESSION, + ...(connections.length === 0 ? { uuid: 'init-leaf' } : {}), + model: 'claude-sonnet-5', + apiKeySource: 'none' + }) + return { + models: [{ value: 'sonnet', displayName: 'Sonnet' }], + ...(initializeAccount === undefined ? {} : { account: initializeAccount }) + } + }, + getSettings: async () => { + connection.calls.push({ subtype: 'get_settings' }) + return { env: {} } + }, + supportedModels: async () => { + connection.calls.push({ subtype: 'list_models' }) + return [{ value: 'sonnet', displayName: 'Sonnet' }] + }, + setModel: async (model) => { + connection.calls.push({ subtype: 'set_model', params: { model } }) + }, + setPermissionMode: async (mode) => { + connection.calls.push({ subtype: 'set_permission_mode', params: { mode } }) + }, + applyFlagSettings: async (settings) => { + connection.calls.push({ subtype: 'apply_flag_settings', params: { settings } }) + }, + interrupt: async () => { + connection.calls.push({ subtype: 'interrupt', params: {} }) + return undefined + }, + cancelAsyncMessage: async () => {}, + stopTask: async (taskId) => { + connection.calls.push({ subtype: 'stop_task', params: { taskId } }) + }, + send: async (message) => { + connection.sent.push(message) + if (message.type === 'user') { + handlers.onMessage?.({ ...message, uuid: 'user-1' }) + } + }, + exitVerdict: selfExit?.exitVerdict ?? { root: 'live', tree: 'unverifiable' }, + close: async () => { + connection.closed = true + return selfExit === null + } + } + connections.push(connection) + return connection + }) as typeof openClaudeStreamJsonConnection + const live = (): FakeClaudeConnection => { + const connection = connections.at(-1) + if (!connection) { + throw new Error('no Claude connection') + } + return connection + } + return { + connections, + openConnection, + live, + setInitializeAccount: (account: unknown) => { + initializeAccount = account + }, + setSelfExit: (exit: typeof selfExit) => { + selfExit = exit + } + } +} + +let operations = 0 +// Keep IDs unique without making each assertion depend on a wall-clock tick. +const TEST_OPERATION_TIMESTAMP = Date.now().toString() + +function operationId(): string { + operations += 1 + return `${TEST_OPERATION_TIMESTAMP}-${operations.toString(16).padStart(32, '0')}` +} + +function envelope(method: string, fields: Record, fence: number | null) { + return { + sessionId: SESSION, + clientOperationId: operationId(), + expectedRuntimeFence: fence, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method, + sessionId: SESSION, + fields + }) + } +} + +function createIntentParams() { + const worktree = `id:${WORKSPACE}` + const fields = { worktree, agent: 'claude' } + return { envelope: envelope('agentSession.create', fields, null), ...fields } +} + +function ensureParams(fence: number) { + const params = { + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: WORKSPACE, + workspaceKind: 'git-worktree' as const + }, + provider: 'claude' as const, + agent: 'claude', + accountHome: { variable: 'CLAUDE_CONFIG_DIR' as const, path: join(root, 'claude-home') }, + runtimeKind: 'native' as const, + providerHandle: { + kind: 'claude' as const, + sessionId: PROVIDER_SESSION, + leafUuid: 'assistant-leaf' + } + } + const base = { + sessionId: SESSION, + clientOperationId: operationId(), + expectedRuntimeFence: fence, + payloadFingerprint: '' + } + return { + ...params, + envelope: { + ...base, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.attach', + sessionId: SESSION, + fields: attachFingerprintFields({ ...params, envelope: base } as never) + }) + } + } +} + +function leaseOf(sessionId: string): { + claimStatus: string + runtimeFence: number + handoffStage: string | null + deathEvidence: { kind: string; detail: string } | null +} { + const host = getStructuredAgentSessionHost() as unknown as { + deps: { store: { getRecord: (id: string) => { lease: ReturnType } } } + } + return host.deps.store.getRecord(sessionId).lease +} + +function handoffParams(direction: 'to-native' | 'to-tui', fence: number) { + const fields = { direction, mode: 'now' as const, action: 'start' as const } + return { + envelope: envelope('agentSession.requestHandoff', fields, fence), + ...fields + } +} + +let claude: ReturnType +let root: string +let dispatcher: RpcDispatcher +let cleanups: Map void> +let tuiOwner: StructuredTuiOwner | null +let transcriptPath: string +/** Managed-account state and configured overlay this host installs, per test. */ +let claudeAuthPolicy: ClaudeStructuredAuthPolicy +let claudeLaunchEnv: Record + +async function call(method: string, params: unknown): Promise { + const replies: RpcResponse[] = [] + const request: RpcRequest = { id: `req-${operations}`, authToken: 'token', method, params } + await dispatcher.dispatchStreaming(request, (raw) => replies.push(JSON.parse(raw)), CLIENT) + if (!replies[0]) { + throw new Error(`no reply for ${method}`) + } + return replies[0] +} + +async function ok(method: string, params: unknown): Promise { + const response = await call(method, params) + expect(response, JSON.stringify(response)).toMatchObject({ ok: true }) + const result = (response as { result: { ok: boolean; value?: T } }).result + expect(result).toMatchObject({ ok: true }) + return result.value as T +} + +async function subscribe(): Promise { + const frames: AgentSessionSubscribeEvent[] = [] + await dispatcher.dispatchStreaming( + { + id: 'subscribe-1', + authToken: 'token', + method: 'agentSession.subscribe', + params: { sessionId: SESSION } + }, + (raw) => { + const response = JSON.parse(raw) as { ok: boolean; result?: AgentSessionSubscribeEvent } + if (response.ok && response.result) { + frames.push(response.result) + } + }, + CLIENT + ) + return frames +} + +function itemsOf(frames: AgentSessionSubscribeEvent[]): AgentJournalRenderItem[] { + const items = new Map() + for (const frame of frames) { + const rows = + frame.type === 'snapshot' || frame.type === 'reset' + ? frame.page.items + : frame.type === 'batch' + ? frame.batch.items + : [] + for (const row of rows) { + items.set(row.itemId, row) + } + } + return [...items.values()] +} + +function textOf(item: AgentJournalRenderItem): string { + return item.body?.kind === 'message' + ? item.body.blocks.map((block) => (block.type === 'text' ? block.text : '')).join('') + : '' +} + +beforeEach(async () => { + operations = 0 + claudeAuthPolicy = { stripAuthEnv: false } + claudeLaunchEnv = { + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test' + } + root = await mkdtemp(join(tmpdir(), 'orca-claude-structured-integration-')) + transcriptPath = join(root, 'claude-home', 'projects', 'workspace', `${PROVIDER_SESSION}.jsonl`) + await mkdir(join(root, 'claude-home', 'projects', 'workspace'), { recursive: true }) + resolveSessionFilePath.mockResolvedValue(transcriptPath) + // The production branch proof returns the latest descendant of the prior + // cursor; mirror that contract so structured close does not regress to a + // stale mocked head. + readClaudeTranscriptLeafUuid.mockImplementation( + async (_path: string, _providerSessionId: string, previousLeafUuid?: string | null) => + previousLeafUuid ?? 'init-leaf' + ) + claude = fakeClaude() + tuiOwner = null + cleanups = new Map() + const handoffTransport: StructuredAgentSessionHandoffTransport = { + hostLabel: 'Scripted Claude host', + launchTui: async ({ record, fence, spawnToken }) => { + const head = record.providerHandleChain.at(-1)?.handle + tuiOwner = { + terminal: { + handle: 'term-claude-tui', + tabId: 'tab-claude-tui', + paneKey: 'tab-claude-tui:leaf-claude-tui', + ptyId: 'pty-claude-tui' + }, + process: { + hostId: 'local', + pid: 7331, + processStartTimeMs: 100, + spawnToken + }, + link: claudeProviderHandleLink({ + sessionId: PROVIDER_SESSION, + leafUuid: head?.provider === 'claude' ? head.leafUuid : null, + resumed: true, + fence, + observedAt: 1 + }), + transcriptPath + } + return tuiOwner + }, + reproveTuiOwner: async ({ owner }) => { + if (owner.link.handle.provider !== 'claude' || !owner.transcriptPath) { + return owner + } + return { + ...owner, + link: claudeProviderHandleLink({ + sessionId: owner.link.handle.sessionId, + leafUuid: await readClaudeTranscriptLeafUuid(owner.transcriptPath), + resumed: true, + fence: owner.link.mintedAtFence, + observedAt: 1 + }) + } + }, + recoverTuiOwner: async () => { + if (!tuiOwner) { + throw new Error('scripted TUI owner missing') + } + return tuiOwner + }, + stopRecoveredOwner: async () => {}, + waitForTuiExit: async (owner) => ({ transcriptPath: owner.transcriptPath }), + waitForTuiIdleOrExit: async () => 'idle', + tuiStatus: () => 'idle', + stopFailedTuiLaunch: async () => {} + } + const runtime = { + getRuntimeId: () => 'runtime-1', + getStructuredAgentSessionCreateSupport: async () => ({ supported: true }), + resolveStructuredAgentSessionCreateIntent: async (input: { envelope: unknown }) => ({ + ...ensureParams(1), + envelope: input.envelope, + providerHandle: undefined + }), + publishStructuredAgentSessionTab: vi.fn(), + ensureStructuredAgentSessionHost: () => + ensureStructuredAgentSessionHost({ + stateDirectory: root, + hostId: 'local', + claimKeyId: 'key-1', + resolveWorkspacePath: async (workspaceId) => `/repos/${workspaceId}`, + resolveCodexCommand: () => '/usr/local/bin/codex', + resolveClaudeCommand: () => '/usr/local/bin/claude', + readProcessStartTime: async (pid: number) => pid * 10, + resolveClaudeLaunchEnv: () => claudeLaunchEnv, + resolveClaudeAuthPolicy: () => claudeAuthPolicy, + openClaudeConnection: claude.openConnection, + handoffTransport + }).then(() => undefined), + registerSubscriptionCleanup: (id: string, dispose: () => void) => cleanups.set(id, dispose), + cleanupSubscription: (id: string) => cleanups.get(id)?.(), + cleanupSubscriptionsByPrefix: () => {} + } + dispatcher = new RpcDispatcher({ + runtime: runtime as unknown as OrcaRuntimeService, + methods: STRUCTURED_AGENT_SESSION_METHODS + }) +}) + +afterEach(async () => { + vi.unstubAllEnvs() + await stopStructuredAgentSessionRuntime() + await rm(root, { recursive: true, force: true }) +}) + +describe('a structured Claude session over agentSession.*', () => { + it('strips ambient Anthropic auth from the child once a managed account is pinned', async () => { + claudeAuthPolicy = { stripAuthEnv: true } + claudeLaunchEnv = { ANTHROPIC_BASE_URL: 'https://gateway.example.test' } + vi.stubEnv('ANTHROPIC_API_KEY', 'sk-ant-SHELL-LEAK') + vi.stubEnv('ANTHROPIC_AUTH_TOKEN', 'tok-SHELL-LEAK') + + await ok<{ fence: number }>('agentSession.create', createIntentParams()) + + const env = claude.live().launch.env + expect(env).not.toHaveProperty('ANTHROPIC_API_KEY') + expect(env).not.toHaveProperty('ANTHROPIC_AUTH_TOKEN') + expect(env).toMatchObject({ + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + CLAUDE_CONFIG_DIR: join(root, 'claude-home') + }) + }) + + it('refuses a create whose configured env overrides the pinned managed account auth', async () => { + claudeAuthPolicy = { stripAuthEnv: true } + // The default overlay carries ANTHROPIC_AUTH_TOKEN, which the terminal path + // refuses at spawn-env.ts:25 rather than letting it beat the pinned account. + const refused = await call('agentSession.create', createIntentParams()) + + expect(JSON.stringify(refused)).toContain('explicit Anthropic auth environment') + // Refused before spawn: no provider child was ever opened. + expect(claude.connections).toHaveLength(0) + }) + + it('durably returns actionable sign-in guidance when initialization has no credentials', async () => { + claude.setInitializeAccount({ apiProvider: 'firstParty', tokenSource: 'none' }) + const params = createIntentParams() + + const first = await call('agentSession.create', params) + const retry = await call('agentSession.create', params) + + expect(first).toMatchObject({ + ok: true, + result: { + ok: false, + refusal: { + code: 'agent_session_operation_invalid', + message: expect.stringMatching(/not signed in.*Claude CLI.*CLAUDE_CONFIG_DIR/s) + } + } + }) + expect((retry as { result: unknown }).result).toEqual((first as { result: unknown }).result) + expect(claude.connections).toHaveLength(1) + }) + + it('releases a session whose CLI self-exited during create, with its diagnostic intact', async () => { + claude.setSelfExit({ + message: 'claude stream-json exited (code 1): claude: not signed in', + // The root's death is first-hand; its descendants were never snapshottable. + exitVerdict: { root: 'exited', tree: 'unverifiable' } + }) + + const failed = await call('agentSession.create', createIntentParams()) + + expect(JSON.stringify(failed)).toContain('claude: not signed in') + const lease = leaseOf(SESSION) + // Latching here would refuse every later attach with agent_session_ownership_unknown, + // wedging a user who only needs to sign in. + expect(lease).toMatchObject({ claimStatus: 'released', handoffStage: null }) + expect(lease.deathEvidence).toMatchObject({ + kind: 'exit-observed', + detail: 'the provider process exited; its descendants were not verifiable' + }) + + claude.setSelfExit(null) + // Signing in and reopening the chat works: the reservation was not latched. + await ok<{ fence: number }>('agentSession.ensure', ensureParams(lease.runtimeFence)) + }) + + it('keeps a session reserved when a descendant of the failed start was seen alive', async () => { + claude.setSelfExit({ + message: 'claude stream-json exited (code 1): claude: not signed in', + exitVerdict: { root: 'exited', tree: 'live' } + }) + + await call('agentSession.create', createIntentParams()) + + // A live descendant still holds the provider session: releasing would hand a + // second writer to it. + expect(leaseOf(SESSION)).toMatchObject({ + claimStatus: 'reserved', + handoffStage: 'manual-recovery' + }) + claude.setSelfExit(null) + }) + + it('routes a published Claude first-hand exit through fenced host reconciliation', async () => { + await ok<{ fence: number }>('agentSession.create', createIntentParams()) + const connection = claude.live() + connection.exitVerdict = { root: 'exited', tree: 'unverifiable' } + connection.handlers.onExit?.(new Error('claude stream-json exited (code 1): crashed')) + + // Claude publishes an exit only after its close ladder and transcript write, + // so the recovery barrier — not a wall-clock poll — is what says it landed. + await waitForStructuredAgentSessionRecovery() + expect(leaseOf(SESSION)).toMatchObject({ claimStatus: 'released', handoffStage: null }) + }) + + it('creates, sends, streams, approves, interrupts, and resumes from the chain head', async () => { + vi.stubEnv('ANTHROPIC_API_KEY', 'sk-ant-SHELL-LEAK') + const created = await ok<{ fence: number }>('agentSession.create', createIntentParams()) + expect(claude.live().launch.options).toMatchObject({ sessionId: PROVIDER_SESSION }) + expect(claude.live().launch.options.resume).toBeUndefined() + expect(claude.live().launch.env).toMatchObject({ + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + CLAUDE_CONFIG_DIR: join(root, 'claude-home'), + [CLAUDE_SPAWN_TOKEN_ENV]: expect.any(String) + }) + // System auth: the user's own shell key is their sign-in, exactly as on the + // terminal path, and the configured overlay still wins over it. + expect(claude.live().launch.env).toMatchObject({ ANTHROPIC_API_KEY: 'sk-ant-SHELL-LEAK' }) + expect(claude.live().launch.env?.PATH ?? claude.live().launch.env?.Path).toBeTruthy() + const history = await call('agentSession.history', { + sessionId: SESSION, + direction: 'tail', + limit: 1 + }) + expect(history).toMatchObject({ + ok: true, + result: { providerSession: { key: 'session_id', id: PROVIDER_SESSION } } + }) + const stream = await subscribe() + + const body = { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'List files' }] } + const sent = await ok<{ + submission: { dispatchState: string; providerItemId: string | null } + }>('agentSession.send', { + envelope: envelope('agentSession.send', { body }, created.fence), + body + }) + expect(sent.submission).toMatchObject({ + dispatchState: 'accepted', + providerItemId: `claude:${PROVIDER_SESSION}:user-1` + }) + + claude.live().handlers.onMessage?.({ + type: 'stream_event', + session_id: PROVIDER_SESSION, + uuid: 'assistant-leaf', + event: { type: 'content_block_delta', delta: { type: 'text_delta', text: 'Two files.' } } + }) + claude.live().handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION, + uuid: 'assistant-leaf', + parent_tool_use_id: null, + message: { role: 'assistant', content: [{ type: 'text', text: 'Two files.' }] } + }) + claude.live().handlers.onMessage?.({ + type: 'result', + subtype: 'success', + session_id: PROVIDER_SESSION, + uuid: 'result-frame-uuid' + }) + claude.live().handlers.onMessage?.({ + type: 'stream_event', + session_id: PROVIDER_SESSION, + uuid: 'stream-event-frame-uuid', + event: { type: 'message_stop' } + }) + await getStructuredAgentSessionHost()?.flushStreamedEvents(SESSION) + expect(itemsOf(stream).find((item) => textOf(item) === 'Two files.')?.itemId).toBe( + `claude:${PROVIDER_SESSION}:assistant-leaf` + ) + + const answeredPermission = Promise.resolve( + claude.live().handlers.canUseTool?.('Bash', { command: 'ls' }, { + requestId: 'permission-1', + toolUseID: 'tool-1', + signal: new AbortController().signal + } as never) + ) + await getStructuredAgentSessionHost()?.flushStreamedEvents(SESSION) + const approval = itemsOf(stream).find((item) => item.body?.kind === 'approval') + expect(approval?.body).toMatchObject({ title: 'Allow Bash?', detail: '{"command":"ls"}' }) + await ok('agentSession.respondToApproval', { + envelope: envelope( + 'agentSession.respondTo:approval', + { + itemId: approval?.itemId, + expectedRevision: approval?.revision, + optionId: 'allow' + }, + created.fence + ), + itemId: approval?.itemId, + expectedRevision: approval?.revision, + optionId: 'allow' + }) + // Answering resolves the SDK's own canUseTool callback with the allow decision. + await expect(answeredPermission).resolves.toMatchObject({ + behavior: 'allow', + toolUseID: 'tool-1' + }) + + await expect( + ok('agentSession.cancel', { + envelope: envelope('agentSession.cancel', { turnId: 'user-1' }, created.fence), + turnId: 'user-1' + }) + ).resolves.toMatchObject({ turnId: 'user-1', cancelled: true }) + expect(claude.live().calls.at(-1)).toMatchObject({ subtype: 'interrupt' }) + + const host = getStructuredAgentSessionHost() as unknown as { + deps: { + store: { + getRecord: (sessionId: string) => { + providerHandleChain: { handle: { provider: string; leafUuid?: string | null } }[] + } + } + } + } + expect(host.deps.store.getRecord(SESSION).providerHandleChain.at(-1)?.handle).toMatchObject({ + provider: 'claude', + leafUuid: null + }) + const old = claude.live() + const resumed = await ok<{ fence: number }>('agentSession.ensure', ensureParams(created.fence)) + expect(resumed.fence).toBe(created.fence + 1) + expect(old.closed).toBe(true) + expect(resolveSessionFilePath).toHaveBeenCalledWith('claude', PROVIDER_SESSION, { + claudeProjectsDir: join(root, 'claude-home', 'projects') + }) + expect(claude.live().launch.options).toMatchObject({ + resume: PROVIDER_SESSION, + resumeSessionAt: 'assistant-leaf' + }) + expect(host.deps.store.getRecord(SESSION).providerHandleChain.at(-1)).toMatchObject({ + handle: { + provider: 'claude', + sessionId: PROVIDER_SESSION, + leafUuid: 'assistant-leaf' + }, + origin: 'resumed' + }) + }) + + it('completes a scripted native to TUI to native cycle with provider-history rehydration', async () => { + const created = await ok<{ fence: number }>('agentSession.create', createIntentParams()) + await writeFile( + transcriptPath, + [ + { + type: 'user', + uuid: 'native-user', + message: { role: 'user', content: [{ type: 'text', text: 'NATIVE_USER' }] } + }, + { + type: 'assistant', + uuid: 'native-assistant', + message: { role: 'assistant', content: [{ type: 'text', text: 'NATIVE_ASSISTANT' }] } + }, + { + type: 'user', + uuid: 'tui-user', + message: { role: 'user', content: [{ type: 'text', text: 'TUI_USER' }] } + }, + { + type: 'assistant', + uuid: 'tui-assistant', + message: { role: 'assistant', content: [{ type: 'text', text: 'TUI_ASSISTANT' }] } + }, + { type: 'last-prompt', leafUuid: 'tui-assistant' } + ] + .map((entry) => JSON.stringify(entry)) + .join('\n') + ) + + await ok('agentSession.requestHandoff', handoffParams('to-tui', created.fence)) + const host = getStructuredAgentSessionHost()! + // No poll: the request enqueues the flow on the session's serialized chain before it returns, + // so this status read is already ordered behind it. Polling only added a wall-clock deadline + // that a loaded runner missed, abandoning a live flow into the suite's teardown. + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'tui', phase: 'idle' }) + expect(claude.connections[0]?.closed).toBe(true) + + const tuiFence = ( + host as unknown as { + deps: { store: { getRecord: (id: string) => { lease: { runtimeFence: number } } } } + } + ).deps.store.getRecord(SESSION).lease.runtimeFence + readClaudeTranscriptLeafUuid.mockResolvedValueOnce('tui-assistant') + await ok('agentSession.requestHandoff', handoffParams('to-native', tuiFence)) + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'native', phase: 'idle' }) + + const frames = await subscribe() + const texts = itemsOf(frames).map(textOf).filter(Boolean) + expect(texts).toEqual( + expect.arrayContaining(['NATIVE_USER', 'NATIVE_ASSISTANT', 'TUI_USER', 'TUI_ASSISTANT']) + ) + expect(new Set(texts).size).toBe(texts.length) + expect(claude.connections).toHaveLength(2) + expect(claude.live().launch.options).toMatchObject({ resume: PROVIDER_SESSION }) + const record = ( + host as unknown as { + deps: { + store: { + getRecord: (id: string) => { + providerHandleChain: { handle: { provider: string; leafUuid?: string | null } }[] + } + } + } + } + ).deps.store.getRecord(SESSION) + expect(record.providerHandleChain.at(-1)?.handle).toMatchObject({ + provider: 'claude', + leafUuid: 'tui-assistant' + }) + }) +}) diff --git a/src/main/runtime/orca-runtime-create-pty-headless-terminal-state.ts b/src/main/runtime/orca-runtime-create-pty-headless-terminal-state.ts index 59fe493118a..da5d824ef62 100644 --- a/src/main/runtime/orca-runtime-create-pty-headless-terminal-state.ts +++ b/src/main/runtime/orca-runtime-create-pty-headless-terminal-state.ts @@ -163,6 +163,17 @@ export class OrcaRuntimeWithCreatePtyHeadlessTerminalState extends OrcaRuntimeWi }) } + /** Public: reflow an already-created model onto a grid the PROVIDER proved — a reattach learns + * the live session's real size only from its spawn reply, after live bytes may have lazily + * created the model at the 80x24 default. Not onExternalPtyResize: nothing measured a pane + * here, so the renderer-geometry baselines behind mobile take-back must stay untouched. */ + reflowHeadlessTerminalToPtyGrid(ptyId: string, cols: number, rows: number): void { + if (cols <= 0 || rows <= 0) { + return + } + this.resizeHeadlessTerminal(ptyId, cols, rows) + } + // Public: desktop-initiated clears (ipc/pty.ts) must also drop this mobile // mirror or a resubscribing mobile client resurrects the cleared scrollback. async clearHeadlessTerminalBuffer(ptyId: string): Promise { diff --git a/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts b/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts index afbb70fc286..f70cf033718 100644 --- a/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts +++ b/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts @@ -17,6 +17,9 @@ import { resolveTuiAgentLaunchArgs, resolveTuiAgentLaunchEnv } from '../../shared/tui-agent-launch-defaults' +import type { AgentSessionLaunchArgs } from '../../shared/agent-session-record' +import { resolveStartupShell } from '../../shared/tui-agent-startup-shell' +import { resolveAgentSessionResumeArgs } from './agent-session-resume-args' export class OrcaRuntimeWithGetAgentSessionExecutionNamespace extends OrcaRuntimeWithResolveWorktreeRemovalTarget { protected getAgentSessionExecutionNamespace( @@ -88,7 +91,12 @@ export class OrcaRuntimeWithGetAgentSessionExecutionNamespace extends OrcaRuntim async ensureAgentSession( request: RuntimeEnsureAgentSessionRequest, _caller: RuntimeAgentSessionRpcCaller = {}, - handoffAuthority?: { spawnToken: string; providerRoot: string; sessionId: string } + handoffAuthority?: { + spawnToken: string + providerRoot: string + sessionId: string + launchArgs?: AgentSessionLaunchArgs + } ): Promise { if (request.kind === 'automatic') { // Legacy renderer sleep records are migration evidence, not host authority. @@ -134,10 +142,12 @@ export class OrcaRuntimeWithGetAgentSessionExecutionNamespace extends OrcaRuntim agent: request.agent, providerSession: identity.providerSession, cmdOverrides: settings.agentCmdOverrides ?? {}, - agentArgs: - request.agentArgs !== undefined - ? request.agentArgs - : resolveTuiAgentLaunchArgs(request.agent, settings.agentDefaultArgs), + agentArgs: resolveAgentSessionResumeArgs({ + requestArgs: request.agentArgs, + persistedArgs: handoffAuthority?.launchArgs, + defaultArgs: resolveTuiAgentLaunchArgs(request.agent, settings.agentDefaultArgs), + shell: resolveStartupShell(platform, shell) + }), agentEnv: { ...resolveTuiAgentLaunchEnv(request.agent, settings.agentDefaultEnv), ...(handoffAuthority && request.agent === 'codex' @@ -148,6 +158,7 @@ export class OrcaRuntimeWithGetAgentSessionExecutionNamespace extends OrcaRuntim }, ompResumeFilePath: request.ompResumeFilePath, sessionOptions: this.toAgentSessionOptions(request.launchPreferences), + sessionOptionsOverrideAgentArgs: Boolean(request.launchPreferences), platform, shell, isRemote diff --git a/src/main/runtime/orca-runtime-get-worktree-ps.ts b/src/main/runtime/orca-runtime-get-worktree-ps.ts index 42c9c7ff6d3..06391e2ae83 100644 --- a/src/main/runtime/orca-runtime-get-worktree-ps.ts +++ b/src/main/runtime/orca-runtime-get-worktree-ps.ts @@ -10,6 +10,7 @@ import { } from './runtime-worktree-ps-activity' import { attachRuntimeWorktreeAgentRows } from './runtime-worktree-agent-rows' import { compareWorktreePs } from './runtime-worktree-status-projection' +import type { AgentSessionRecord } from '../../shared/agent-session-record' import type { Repo } from '../../shared/repo-types' import { enrichMissingRepoGitRemoteIdentities } from '../repo-git-remote-identity-enrichment' import { ensureStructuredAgentSessionHost as installStructuredAgentSessionHost } from './structured-agent-session-runtime' @@ -21,9 +22,11 @@ import { resolveTuiAgentLaunchEnv } from '../../shared/tui-agent-launch-defaults' import { resolveLocalWindowsAgentStartupShell } from '../../shared/windows-terminal-shell' +import { resolveStartupShell, tokenizeStartupCommand } from '../../shared/tui-agent-startup-shell' import { resolveCodexStructuredAppServerArgs } from '../codex/codex-structured-app-server-args' import type { StructuredAgentSessionHandoffTransport } from '../native-chat/agent-session-wire/structured-agent-session-handoff-types' import { hostname } from 'node:os' +import { claudeStructuredAuthPolicyForSettings } from '../claude-accounts/claude-structured-auth-policy' import { probeAgentSessionProcessIdentity } from './agent-session-process-identity-probe' import { structuredAgentSessionTabId } from '../../shared/structured-agent-session-projection' @@ -144,13 +147,47 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStructuredAgent // in a plain folder lands in the folder rather than failing to resolve. resolveWorkspacePath: async (workspaceId) => (await this.resolveRuntimeFileTarget(`id:${workspaceId}`)).worktree.path, - resolveLaunchArgs: () => this.resolveConfiguredCodexStructuredArgs(), + resolveLaunchArgs: (provider) => this.resolveConfiguredStructuredLaunchArgs(provider), resolveLaunchEnvOverlay: () => resolveTuiAgentLaunchEnv('codex', this.requireStore().getSettings().agentDefaultEnv), + resolveClaudeLaunchEnv: () => + resolveTuiAgentLaunchEnv('claude', this.requireStore().getSettings().agentDefaultEnv), + resolveClaudeAuthPolicy: () => + claudeStructuredAuthPolicyForSettings(this.requireStore().getSettings()), + // Same gate and same settings as agentSession.createSupport, re-read on every acquisition. + getClaudeManagedAccountGateSettings: () => this.requireStore().getSettings(), handoffTransport: this.createStructuredAgentSessionHandoffTransport() }) } + // Why the provider is honoured rather than assumed: Codex app-server flags are not + // Claude CLI flags, and prepending them to `claude` makes it exit on an unknown option. + protected resolveConfiguredStructuredLaunchArgs( + provider: AgentSessionRecord['provider'] + ): string[] { + if (provider === 'claude') { + return this.resolveConfiguredClaudeStructuredArgs() + } + return this.resolveConfiguredCodexStructuredArgs() + } + + protected resolveConfiguredClaudeStructuredArgs(): string[] { + const settings = this.requireStore().getSettings() + const shell = resolveStartupShell( + process.platform, + resolveLocalWindowsAgentStartupShell({ + platform: process.platform, + isRemote: false, + terminalWindowsShell: settings.terminalWindowsShell + }) + ) + const tokenized = tokenizeStartupCommand( + resolveTuiAgentLaunchArgs('claude', settings.agentDefaultArgs), + shell + ) + return tokenized.ok ? tokenized.tokens : [] + } + protected resolveConfiguredCodexStructuredArgs(): string[] { const settings = this.requireStore().getSettings() const shell = resolveLocalWindowsAgentStartupShell({ @@ -195,7 +232,7 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStructuredAgent tuiStatus: (owner) => this.structuredTuiStatus(owner), closeTuiOwner: (owner) => this.closeStructuredTuiOwner(owner), revealNativeSession: async ({ workspaceId, sessionId, agent = 'codex', adoptedTerminal }) => { - if (adoptedTerminal || agent !== 'codex') { + if (adoptedTerminal || (agent !== 'codex' && agent !== 'claude')) { return } await this.publishStructuredAgentSessionTab({ diff --git a/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts b/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts index 5700b71d9a5..aef04bde6bc 100644 --- a/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts +++ b/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts @@ -3,7 +3,10 @@ import { OrcaRuntimeWithStopStructuredSessionProcess } from './orca-runtime-stop import type { AgentSessionOwnerBinding } from '../../shared/agent-session-host-authority' import { agentSessionOwnerBindingsEqual } from '../../shared/claimed-agent-pty-owner-snapshot' import { resolvePinnedCodexRolloutProof } from '../codex/codex-tui-rollout-proof' +import { supportsCodexStructuredLocation } from '../codex/codex-structured-location-support' +import { supportsClaudeStructuredLocation } from '../claude/claude-structured-location-support' import { getStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' +import { resolveStructuredAgentSessionCreateSupport } from '../native-chat/structured-agent-session-create-support' import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' import type { AgentStatusIpcPayload } from '../../shared/agent-status-types' import { getLocalProjectWorktreeGitOptions } from '../project-runtime-git-options' @@ -12,6 +15,8 @@ import { getSystemCodexHomePath } from '../codex/codex-home-paths' import { resolveTuiAgentLaunchEnv } from '../../shared/tui-agent-launch-defaults' import { hasPersistedStructuredAgentSessionStore as hasPersistedStructuredAgentSessionStoreOnDisk } from './structured-agent-session-runtime' import { getProfileUserDataPath } from '../orca-profiles/profile-storage-paths' +import { homedir } from 'node:os' +import { join } from 'node:path' export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends OrcaRuntimeWithStopStructuredSessionProcess { protected async resolveRecoveredStructuredTuiTranscript(input: { @@ -45,22 +50,18 @@ export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends Orca async getStructuredAgentSessionCreateSupport( worktreeSelector: string, - agent: 'codex' + agent: 'claude' | 'codex' ): Promise<{ supported: boolean; reason?: 'agent' | 'remote' | 'wsl' }> { const location = await this.resolveStructuredAgentSessionLocation(worktreeSelector) - await this.ensureStructuredAgentSessionHost() - if (getStructuredAgentSessionHost()?.supportsCreate(location, agent)) { - return { supported: true } - } - return { - supported: false, - reason: - location.executionHostId !== LOCAL_EXECUTION_HOST_ID - ? 'remote' - : location.wslDistro - ? 'wsl' - : 'agent' - } + return resolveStructuredAgentSessionCreateSupport({ + agent, + location, + adapterSupportsCreate: + agent === 'claude' + ? supportsClaudeStructuredLocation(location) + : supportsCodexStructuredLocation(location), + getSettings: () => this.requireStore().getSettings() + }) } protected hasProviderSessionObservationSource(): boolean { @@ -108,8 +109,23 @@ export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends Orca async resolveStructuredAgentSessionCreateIntent(input: { envelope: { sessionId: string; clientOperationId: string } worktree: string - agent: 'codex' + agent: 'claude' | 'codex' }): Promise { + if (input.agent === 'claude') { + return this.resolveStructuredAgentSessionIntent(input, async ({ launchEnv, location }) => { + return ( + launchEnv.CLAUDE_CONFIG_DIR?.trim() || + this.accounts + .getClaudeConfigDirectory( + location.wslDistro + ? { runtime: 'wsl', wslDistro: location.wslDistro } + : { runtime: 'host' } + ) + ?.trim() || + join(homedir(), '.claude') + ) + }) + } return this.resolveStructuredAgentSessionIntent(input, async ({ workspacePath, launchEnv }) => { // A create has no process yet, so the current selection is what it must follow. const preparedHome = await this.prepareCodexStructuredLaunchFn?.({ workspacePath, launchEnv }) @@ -126,11 +142,17 @@ export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends Orca input: { envelope: { sessionId: string; clientOperationId: string } worktree: string - agent: 'codex' + agent: 'claude' | 'codex' }, resolveAccountHomePath: (context: { workspacePath: string launchEnv: NodeJS.ProcessEnv + location: { + executionHostId: string + wslDistro: string | null + workspaceId: string + workspaceKind: 'folder' | 'git-worktree' + } }) => string | Promise ): Promise { const support = await this.getStructuredAgentSessionCreateSupport(input.worktree, input.agent) @@ -152,8 +174,8 @@ export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends Orca provider: input.agent, agent: input.agent, accountHome: { - variable: 'CODEX_HOME', - path: await resolveAccountHomePath({ workspacePath, launchEnv }) + variable: input.agent === 'claude' ? 'CLAUDE_CONFIG_DIR' : 'CODEX_HOME', + path: await resolveAccountHomePath({ workspacePath, launchEnv, location }) }, runtimeKind: 'native' } diff --git a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts index 4466594b0dc..5b2c160f2c6 100644 --- a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts +++ b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts @@ -45,7 +45,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu } this.hydrateHeadlessMobileSessionTabsFromWorkspaceSession() for (const session of host?.listSessionTabs() ?? []) { - if (session.agent !== 'codex') { + if (session.agent !== 'codex' && session.agent !== 'claude') { continue } let sessionId = session.sessionId @@ -54,7 +54,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu } await this.publishStructuredAgentSessionTab({ ...session, - agent: 'codex', + agent: session.agent, sessionId, activate: false, notify: false @@ -65,7 +65,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu async publishStructuredAgentSessionTab(input: { workspaceId: string sessionId: string - agent: 'codex' + agent: 'claude' | 'codex' activate: boolean notify?: boolean }): Promise { @@ -105,7 +105,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu const tab: RuntimeMobileSessionAgentTab = { type: 'agent-session', id, - title: 'Codex Chat', + title: input.agent === 'claude' ? 'Claude Chat' : 'Codex Chat', sessionId: input.sessionId, agent: input.agent, isActive: input.activate diff --git a/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts b/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts index 083c677e39f..af1fbc20372 100644 --- a/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts +++ b/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts @@ -52,4 +52,108 @@ describe('structured agent-session create intent', () => { path: '/accounts/selected/home' }) }) + + it('pins the configured Claude launch home without Codex launch preparation', async () => { + const prepareCodexStructuredLaunch = vi.fn() + const runtime = new OrcaRuntimeService( + { + getSettings: () => ({ + agentDefaultEnv: { + claude: { CLAUDE_CONFIG_DIR: '/configured/claude-home' } + } + }) + } as never, + undefined, + { prepareCodexStructuredLaunch } + ) + vi.spyOn(runtime, 'getStructuredAgentSessionCreateSupport').mockResolvedValue({ + supported: true + }) + const internal = runtime as unknown as { + resolveStructuredAgentSessionLocation: (selector: string) => Promise<{ + executionHostId: string + wslDistro: null + workspaceId: string + workspaceKind: 'git-worktree' + }> + resolveRuntimeFileTarget: (selector: string) => Promise<{ + worktree: { path: string } + }> + } + internal.resolveStructuredAgentSessionLocation = vi.fn(async () => ({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' as const + })) + internal.resolveRuntimeFileTarget = vi.fn(async () => ({ + worktree: { path: '/repos/workspace-1' } + })) + + const intent = await runtime.resolveStructuredAgentSessionCreateIntent({ + envelope: { sessionId: 'session-1', clientOperationId: 'operation-1' }, + worktree: 'id:workspace-1', + agent: 'claude' + }) + + expect(prepareCodexStructuredLaunch).not.toHaveBeenCalled() + expect(intent.accountHome).toEqual({ + variable: 'CLAUDE_CONFIG_DIR', + path: '/configured/claude-home' + }) + }) + + it('uses the managed Claude launch home before falling back to ~/.claude', async () => { + const prepareCodexStructuredLaunch = vi.fn() + const getRuntimeConfigDir = vi.fn(() => '/accounts/managed/claude-home') + const runtime = new OrcaRuntimeService( + { + getSettings: () => ({ + agentDefaultEnv: { claude: {} } + }) + } as never, + undefined, + { prepareCodexStructuredLaunch } + ) + runtime.setAccountServices({ + claudeAccounts: { getRuntimeConfigDir } as never, + codexAccounts: {} as never, + rateLimits: {} as never + }) + vi.spyOn(runtime, 'getStructuredAgentSessionCreateSupport').mockResolvedValue({ + supported: true + }) + const internal = runtime as unknown as { + resolveStructuredAgentSessionLocation: (selector: string) => Promise<{ + executionHostId: string + wslDistro: null + workspaceId: string + workspaceKind: 'git-worktree' + }> + resolveRuntimeFileTarget: (selector: string) => Promise<{ + worktree: { path: string } + }> + } + internal.resolveStructuredAgentSessionLocation = vi.fn(async () => ({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' as const + })) + internal.resolveRuntimeFileTarget = vi.fn(async () => ({ + worktree: { path: '/repos/workspace-1' } + })) + + const intent = await runtime.resolveStructuredAgentSessionCreateIntent({ + envelope: { sessionId: 'session-1', clientOperationId: 'operation-1' }, + worktree: 'id:workspace-1', + agent: 'claude' + }) + + expect(getRuntimeConfigDir).toHaveBeenCalledTimes(1) + expect(intent.accountHome).toEqual({ + variable: 'CLAUDE_CONFIG_DIR', + path: '/accounts/managed/claude-home' + }) + }) }) diff --git a/src/main/runtime/orca-runtime-structured-agent-session-launch-args.test.ts b/src/main/runtime/orca-runtime-structured-agent-session-launch-args.test.ts new file mode 100644 index 00000000000..c4333fd0a4d --- /dev/null +++ b/src/main/runtime/orca-runtime-structured-agent-session-launch-args.test.ts @@ -0,0 +1,86 @@ +import { describe, expect, it, vi } from 'vitest' +import { OrcaRuntimeService } from './orca-runtime' + +type InstalledDeps = { + resolveLaunchArgs: (provider: 'claude' | 'codex') => Promise | string[] + resolveLaunchEnvOverlay: () => Record + resolveClaudeLaunchEnv?: () => Record +} + +const { installStructuredAgentSessionHost } = vi.hoisted(() => ({ + installStructuredAgentSessionHost: vi.fn(async (_deps: unknown) => ({}) as never) +})) + +vi.mock('./structured-agent-session-runtime', async (importOriginal) => ({ + ...(await importOriginal()), + ensureStructuredAgentSessionHost: installStructuredAgentSessionHost +})) + +function runtimeWith(settings: Record): OrcaRuntimeService { + return new OrcaRuntimeService({ getSettings: () => settings } as never) +} + +async function installedDeps(settings: Record): Promise { + installStructuredAgentSessionHost.mockClear() + await runtimeWith(settings).ensureStructuredAgentSessionHost() + return installStructuredAgentSessionHost.mock.calls[0]?.[0] as InstalledDeps +} + +describe('structured agent-session launch args wiring', () => { + it('resolves Claude launch args from the Claude agent defaults, not Codex flags', async () => { + const deps = await installedDeps({ + agentDefaultArgs: { + claude: '--dangerously-skip-permissions --model opus', + codex: '--dangerously-bypass-approvals-and-sandbox' + }, + agentDefaultEnv: {} + }) + + expect(await deps.resolveLaunchArgs('claude')).toEqual([ + '--dangerously-skip-permissions', + '--model', + 'opus' + ]) + }) + + it('still resolves Codex app-server args for a Codex session', async () => { + const deps = await installedDeps({ + agentDefaultArgs: { + claude: '--dangerously-skip-permissions', + codex: '--dangerously-bypass-approvals-and-sandbox' + }, + agentDefaultEnv: {} + }) + + const codexArgs = await deps.resolveLaunchArgs('codex') + expect(codexArgs).not.toContain('--dangerously-skip-permissions') + expect(codexArgs.length).toBeGreaterThan(0) + }) + + it('never lets a broken Codex args configuration block a Claude session', async () => { + const deps = await installedDeps({ + agentDefaultArgs: { claude: '--model opus', codex: '--not-a-real-codex-flag' }, + agentDefaultEnv: {} + }) + + expect(await deps.resolveLaunchArgs('claude')).toEqual(['--model', 'opus']) + expect(() => deps.resolveLaunchArgs('codex')).toThrow() + }) + + it('supplies the Claude env overlay so the launch resolver does not fall back to process.env', async () => { + const deps = await installedDeps({ + agentDefaultArgs: {}, + agentDefaultEnv: { + claude: { ORCA_CLAUDE_OVERLAY: 'claude-value' }, + codex: { ORCA_CODEX_OVERLAY: 'codex-value' } + } + }) + + expect(deps.resolveClaudeLaunchEnv).toBeTypeOf('function') + expect(deps.resolveClaudeLaunchEnv?.()).toMatchObject({ + ORCA_CLAUDE_OVERLAY: 'claude-value' + }) + expect(deps.resolveClaudeLaunchEnv?.()).not.toHaveProperty('ORCA_CODEX_OVERLAY') + expect(deps.resolveLaunchEnvOverlay()).toMatchObject({ ORCA_CODEX_OVERLAY: 'codex-value' }) + }) +}) diff --git a/src/main/runtime/orca-runtime-structured-agent-session-launch-tui.ts b/src/main/runtime/orca-runtime-structured-agent-session-launch-tui.ts index 9e70602f91b..89836d1e027 100644 --- a/src/main/runtime/orca-runtime-structured-agent-session-launch-tui.ts +++ b/src/main/runtime/orca-runtime-structured-agent-session-launch-tui.ts @@ -28,7 +28,12 @@ export class OrcaRuntimeWithStructuredAgentSessionLaunchTui extends OrcaRuntimeW presentation: 'background' }, {}, - { spawnToken, providerRoot: record.accountHome.path, sessionId: record.sessionId } + { + spawnToken, + providerRoot: record.accountHome.path, + sessionId: record.sessionId, + ...(record.launchArgs !== undefined ? { launchArgs: record.launchArgs } : {}) + } ) const terminal = launched.terminal let spawnedOwner: StructuredTuiOwner | null = null diff --git a/src/main/runtime/orca-runtime-structured-claude-account-gate.test.ts b/src/main/runtime/orca-runtime-structured-claude-account-gate.test.ts new file mode 100644 index 00000000000..64b451e9ea9 --- /dev/null +++ b/src/main/runtime/orca-runtime-structured-claude-account-gate.test.ts @@ -0,0 +1,112 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { OrcaRuntimeService } from './orca-runtime' +import { setStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' +import type { StructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-host' +import type { ClaudeManagedAccountGateSettings } from '../native-chat/claude-structured-managed-account-support' + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp') } +})) + +function managedAccount(id: string, managedAuthRuntime: 'host' | 'wsl') { + return { + id, + email: `${id}@example.com`, + managedAuthPath: `/managed/${id}`, + managedAuthRuntime, + authMethod: 'subscription-oauth' as const, + createdAt: 0, + updatedAt: 0, + lastAuthenticatedAt: 0 + } +} + +const WSL_ONLY: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('wsl-1', 'wsl')], + activeClaudeManagedAccountId: null, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-1' } } +} + +/** Registered Claude accounts with none selected: ambient auth, and the UI names no host identity, + * so this must reach structured rather than silently falling back to a terminal session. */ +const ACCOUNTS_PRESENT_NONE_ACTIVE: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('host-1', 'host'), managedAccount('host-2', 'host')], + activeClaudeManagedAccountId: null, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: {} } +} + +const HOST_SELECTED: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('host-1', 'host')], + activeClaudeManagedAccountId: 'host-1', + activeClaudeManagedAccountIdsByRuntime: { host: 'host-1', wsl: {} } +} + +function runtimeWithAccounts(claude: ClaudeManagedAccountGateSettings | null): OrcaRuntimeService { + // No store at all is the unreadable-settings case the gate must fail closed on. + const runtime = claude + ? new OrcaRuntimeService({ getSettings: () => claude } as never) + : new OrcaRuntimeService() + const internal = runtime as unknown as { + resolveStructuredAgentSessionLocation: (selector: string) => Promise + ensureStructuredAgentSessionHost: () => Promise + } + internal.resolveStructuredAgentSessionLocation = vi.fn(async () => ({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' as const + })) + // The adapter's own location answer is irrelevant here; pin it supported so only the account + // gate can refuse. + internal.ensureStructuredAgentSessionHost = vi.fn(async () => {}) + setStructuredAgentSessionHost({ + supportsCreate: () => true + } as unknown as StructuredAgentSessionHost) + return runtime +} + +afterEach(() => { + setStructuredAgentSessionHost(null) +}) + +describe('structured Claude managed-account gate', () => { + it('refuses Claude under a WSL-only managed account', async () => { + const runtime = runtimeWithAccounts(WSL_ONLY) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'claude') + ).resolves.toMatchObject({ supported: false }) + }) + + it('supports Claude when accounts are registered but none is selected', async () => { + const runtime = runtimeWithAccounts(ACCOUNTS_PRESENT_NONE_ACTIVE) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'claude') + ).resolves.toMatchObject({ supported: true }) + }) + + it('still supports Claude under a selected host managed account', async () => { + const runtime = runtimeWithAccounts(HOST_SELECTED) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'claude') + ).resolves.toMatchObject({ supported: true }) + }) + + it('fails closed for Claude when the account runtime cannot be determined', async () => { + const runtime = runtimeWithAccounts(null) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'claude') + ).resolves.toMatchObject({ supported: false }) + }) + + /** The gate is Claude's alone: Codex resolves its account separately and this lane must not + * change any Codex answer. */ + it('leaves Codex supported under the same WSL-only Claude account', async () => { + const runtime = runtimeWithAccounts(WSL_ONLY) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'codex') + ).resolves.toMatchObject({ supported: true }) + }) +}) diff --git a/src/main/runtime/orca-runtime-structured-claude-gate-wiring.test.ts b/src/main/runtime/orca-runtime-structured-claude-gate-wiring.test.ts new file mode 100644 index 00000000000..0d9b12f7c52 --- /dev/null +++ b/src/main/runtime/orca-runtime-structured-claude-gate-wiring.test.ts @@ -0,0 +1,64 @@ +import { describe, expect, it, vi } from 'vitest' + +const installed = vi.hoisted(() => ({ deps: null as Record | null })) + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp') } +})) + +vi.mock('./structured-agent-session-runtime', () => ({ + ensureStructuredAgentSessionHost: vi.fn(async (deps: Record) => { + installed.deps = deps + }) +})) + +import { OrcaRuntimeService } from './orca-runtime' +import { + readClaudeManagedAccountGateSettings, + type ClaudeManagedAccountGateSettings +} from '../native-chat/claude-structured-managed-account-support' + +const SETTINGS = { + claudeManagedAccounts: [], + activeClaudeManagedAccountId: null, + agentDefaultEnv: {}, + agentDefaultArgs: {} +} as unknown as ClaudeManagedAccountGateSettings + +function gateSettingsGetter(): (() => ClaudeManagedAccountGateSettings) | undefined { + const deps: Record = installed.deps ?? {} + const get = deps['getClaudeManagedAccountGateSettings'] + return typeof get === 'function' ? (get as () => ClaudeManagedAccountGateSettings) : undefined +} + +/** The runtime class this wiring lives on does not typecheck its own `this` calls, so a broken or + * missing gate hookup compiles clean. Pin it behaviourally instead. */ +describe('structured Claude managed-account gate wiring', () => { + it('hands the host a gate reader that resolves the live settings', async () => { + installed.deps = null + const runtime = new OrcaRuntimeService({ getSettings: () => SETTINGS } as never) + + await runtime.ensureStructuredAgentSessionHost() + + const get = gateSettingsGetter() + expect(typeof get).toBe('function') + expect(get?.()).toBe(SETTINGS) + }) + + /** The installer composes this getter with the fail-closed reader, which is the shape the + * resolver consumes; pin that composition end to end. */ + it('composes into a null answer instead of throwing when settings cannot be read', async () => { + installed.deps = null + const runtime = new OrcaRuntimeService() + + await runtime.ensureStructuredAgentSessionHost() + + const get = gateSettingsGetter() + expect(typeof get).toBe('function') + expect(() => get?.()).toThrow() + expect(readClaudeManagedAccountGateSettings(get!)).toBeNull() + }) +}) diff --git a/src/main/runtime/orca-runtime-structured-session-restore.test.ts b/src/main/runtime/orca-runtime-structured-session-restore.test.ts index 9e752330445..d6ec1e22782 100644 --- a/src/main/runtime/orca-runtime-structured-session-restore.test.ts +++ b/src/main/runtime/orca-runtime-structured-session-restore.test.ts @@ -325,6 +325,56 @@ describe('structured session cold restoration', () => { expect(closed.tabGroups?.[0]?.tabOrder).toEqual(['terminal-tab']) }) + it('publishes restored Claude tabs with the Claude title', async () => { + const runtime = new OrcaRuntimeService() + const publish = vi.spyOn(runtime, 'publishStructuredAgentSessionTab') + const internal = runtime as unknown as { + hasPersistedStructuredAgentSessionStore(): boolean + getKnownWorkspaceSessionWorktreeIds(): Set + hydrateHeadlessMobileSessionTabsFromWorkspaceSession(): Set + refreshMobileSessionPtyRecords(): Promise | null> + ensureStructuredAgentSessionHost(): Promise + } + internal.hasPersistedStructuredAgentSessionStore = () => true + internal.getKnownWorkspaceSessionWorktreeIds = () => new Set() + internal.hydrateHeadlessMobileSessionTabsFromWorkspaceSession = () => new Set() + internal.refreshMobileSessionPtyRecords = async () => new Set() + internal.ensureStructuredAgentSessionHost = async () => undefined + setStructuredAgentSessionHost({ + reconcileRestartLeases: async () => undefined, + restoreReadableSessions: async () => undefined, + listSessionTabs: () => [ + { + sessionId: 'agent-session:agent-session:restored-claude', + workspaceId: 'workspace-1', + agent: 'claude' + } + ] + } as never) + + await runtime.restoreStructuredAgentSessionTabs() + + expect(publish).toHaveBeenCalledWith({ + workspaceId: 'workspace-1', + sessionId: 'restored-claude', + agent: 'claude', + activate: false, + notify: false + }) + + const restored = await runtime.listMobileSessionTabs('id:workspace-1') + expect(restored.tabs).toEqual( + expect.arrayContaining([ + expect.objectContaining({ + type: 'agent-session', + id: 'agent-session:restored-claude', + title: 'Claude Chat', + agent: 'claude' + }) + ]) + ) + }) + it('commits the host close when the renderer already removed the structured tab', async () => { const runtime = new OrcaRuntimeService() runtime.setNotifier({ diff --git a/src/main/runtime/orca-runtime-structured-tui-tab-binding.test.ts b/src/main/runtime/orca-runtime-structured-tui-tab-binding.test.ts new file mode 100644 index 00000000000..d15f754b244 --- /dev/null +++ b/src/main/runtime/orca-runtime-structured-tui-tab-binding.test.ts @@ -0,0 +1,730 @@ +import { createHash } from 'node:crypto' +import { describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionHandoffTransport } from '../native-chat/agent-session-wire/structured-agent-session-handoff-types' +import { createEphemeralAgentSessionClaimSigner } from './agent-session-claim-identity' +import { agentSessionPtyWriteGate } from './agent-session-pty-write-gate' +import { OrcaRuntimeService } from './orca-runtime' + +const { + probeAgentSessionProcessIdentity, + proveCodexTuiRollout, + readClaudeTranscriptLeafUuid, + readStructuredTuiProcessIdentity, + resolveSessionFilePath, + resolvePinnedCodexRolloutProof +} = vi.hoisted(() => ({ + probeAgentSessionProcessIdentity: vi.fn(), + proveCodexTuiRollout: vi.fn(), + readClaudeTranscriptLeafUuid: vi.fn(), + readStructuredTuiProcessIdentity: vi.fn(), + resolveSessionFilePath: vi.fn(), + resolvePinnedCodexRolloutProof: vi.fn() +})) + +vi.mock('./structured-tui-process-identity', () => ({ readStructuredTuiProcessIdentity })) +vi.mock('../codex/codex-tui-rollout-proof', () => ({ + proveCodexTuiRollout, + resolvePinnedCodexRolloutProof +})) +vi.mock('../native-chat/session-file-resolver', () => ({ + readClaudeTranscriptLeafUuid, + resolveSessionFilePath +})) +vi.mock('./agent-session-process-identity-probe', async (importOriginal) => ({ + ...(await importOriginal()), + probeAgentSessionProcessIdentity +})) + +const WORKTREE_ID = 'repo-1::/tmp/structured-handoff' + +function notifier(revealTerminalSession: ReturnType) { + return { + worktreesChanged: vi.fn(), + reposChanged: vi.fn(), + activateWorktree: vi.fn(), + createTerminal: vi.fn(), + revealTerminalSession, + splitTerminal: vi.fn(), + renameTerminal: vi.fn(), + focusTerminal: vi.fn(), + closeTerminal: vi.fn(), + sleepWorktree: vi.fn(), + terminalFitOverrideChanged: vi.fn(), + terminalDriverChanged: vi.fn() + } +} + +describe('structured TUI launch tab binding', () => { + it('recovers a live TUI from durable owner inventory in a fresh runtime', async () => { + const namespace = { + machine: 'native:test', + principal: 'uid:1', + container: 'native', + providerRoot: '/tmp/codex-home' + } + const signer = createEphemeralAgentSessionClaimSigner('profile-test') + const claim = signer.createClaim({ + namespace, + identity: { agent: 'codex', providerSession: { key: 'session_id', id: 'thread-1' } }, + canonicalWorktreeId: WORKTREE_ID + }) + const terminalHandle = 'term_cold_owner' + const leafId = '23013912-13f8-44e5-818f-d40a1ff4e8c5' + resolvePinnedCodexRolloutProof.mockResolvedValue('/tmp/codex-home/sessions/thread-1.jsonl') + const writeAgentSessionProof = vi.fn(() => false) + const runtime = new OrcaRuntimeService(undefined, undefined, { + agentSessionClaimSigner: signer + }) + runtime.setPtyController({ + listProcesses: vi.fn(async () => [ + { + id: 'pty-cold-owner', + incarnationId: 'incarnation-1', + cwd: '/tmp/structured-handoff', + title: 'codex', + worktreeId: WORKTREE_ID, + terminalHandle, + agentSessionOwners: [ + { + claim, + generation: 'generation-1', + phase: 'live' as const, + ptyId: 'pty-cold-owner', + surface: { + worktreeId: WORKTREE_ID, + tabId: 'tab-cold-owner', + leafId, + terminalHandle + } + } + ] + } + ]), + write: () => true, + kill: () => true, + writeAgentSessionProof, + getForegroundProcess: async () => null + }) + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + refreshMobileSessionPtyRecords(): Promise | null> + listResolvedWorktrees(): Promise + resolveTerminalWorkspaceLaunchScope(): Promise<{ + id: string + path: string + connectionId: null + repo: null + folderWorkspace: null + }> + getAgentSessionExecutionNamespace(): typeof namespace + ptysById: Map< + string, + { + launchToken: string | null + launchAgent: string | null + agentSessionOwners: unknown[] + tabId?: string | null + paneKey?: string | null + } + > + } + internal.listResolvedWorktrees = vi.fn(async () => [ + { id: WORKTREE_ID, repoId: 'repo-1', path: '/tmp/structured-handoff' } + ]) + internal.resolveTerminalWorkspaceLaunchScope = vi.fn(async () => ({ + id: WORKTREE_ID, + path: '/tmp/structured-handoff', + connectionId: null, + repo: null, + folderWorkspace: null + })) + internal.getAgentSessionExecutionNamespace = () => namespace + proveCodexTuiRollout.mockResolvedValueOnce({ + transcriptPath: '/tmp/codex-home/sessions/thread-1.jsonl' + }) + probeAgentSessionProcessIdentity.mockResolvedValue({ + outcome: 'identity-matched', + matchedOn: ['process-start-time'] + }) + + await internal.refreshMobileSessionPtyRecords() + const coldPty = internal.ptysById.get('pty-cold-owner')! + expect(coldPty).toMatchObject({ launchToken: null, launchAgent: null }) + expect(coldPty.agentSessionOwners).toHaveLength(1) + const runtimeId = (runtime as unknown as { runtimeId: string }).runtimeId + ;( + runtime as unknown as { + handles: Map< + string, + { + handle: string + runtimeId: string + rendererGraphEpoch: number + worktreeId: string + tabId: string + leafId: string + ptyId: string + ptyGeneration: number + } + > + } + ).handles.set(terminalHandle, { + handle: terminalHandle, + runtimeId, + rendererGraphEpoch: 0, + worktreeId: WORKTREE_ID, + tabId: 'pty:pty-cold-owner', + leafId: 'pty:pty-cold-owner', + ptyId: 'pty-cold-owner', + ptyGeneration: 0 + }) + coldPty.tabId = 'tab-cold-owner' + coldPty.paneKey = `tab-cold-owner:${leafId}` + coldPty.launchToken = 'spawn-token' + coldPty.launchAgent = 'codex' + + const owner = await internal.createStructuredAgentSessionHandoffTransport().recoverTuiOwner({ + sessionId: 'session-1', + location: { workspaceId: WORKTREE_ID, executionHostId: 'local' }, + accountHome: { variable: 'CODEX_HOME', path: namespace.providerRoot }, + providerHandleChain: [{ handle: { provider: 'codex', threadId: 'thread-1' }, observedAt: 1 }], + lease: { + ownerProcess: { + hostId: 'local', + pid: 4243, + processStartTimeMs: 10, + spawnToken: 'spawn-token' + }, + runtimeFence: 3 + } + } as never) + + expect(owner.terminal).toEqual({ + handle: terminalHandle, + tabId: 'tab-cold-owner', + paneKey: `tab-cold-owner:${leafId}`, + ptyId: 'pty-cold-owner' + }) + expect(proveCodexTuiRollout).toHaveBeenCalledWith( + expect.objectContaining({ + codexHome: namespace.providerRoot, + threadId: 'thread-1', + readOutput: expect.any(Function), + write: expect.any(Function) + }) + ) + expect(resolvePinnedCodexRolloutProof).not.toHaveBeenCalled() + expect(writeAgentSessionProof).not.toHaveBeenCalled() + expect(agentSessionPtyWriteGate.boundSessionId('pty-cold-owner')).toBe('session-1') + agentSessionPtyWriteGate.unbindPty('pty-cold-owner') + }) + + it('rebuilds a Claude proving link from current launch-token-bound hook evidence', async () => { + const paneKey = 'tab-claude:leaf-claude' + const spawnToken = 'claude-restart-token' + const sessionId = '019fd532-7c11-7a90-b6de-4e1a2c3d5f61' + const transcriptPath = '/tmp/claude-home/projects/worktree/session.jsonl' + const attestAgentHookCompatibilityAuthority = vi.fn(() => ({ + paneKey, + source: 'hydrated_commitment' as const + })) + const runtime = new OrcaRuntimeService(null, undefined, { + attestAgentHookCompatibilityAuthority, + getAgentProviderSessionRowsForPane: () => [ + { + paneKey, + connectionId: null, + state: 'done', + prompt: '', + agentType: 'claude', + receivedAt: Date.now() + 1000, + stateStartedAt: 10, + providerSession: { key: 'session_id', id: sessionId, transcriptPath } + } + ] + }) + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + ptysById: Map + restoredOrchestrationAuthorityByPtyId: Map + } + internal.ptysById.set('pty-claude', { + ptyId: 'pty-claude', + worktreeId: WORKTREE_ID, + connectionId: null, + tabId: 'tab-claude', + paneKey, + launchToken: spawnToken, + launchAgent: 'claude', + connected: true + }) + resolveSessionFilePath.mockResolvedValue('/tmp/claude-home/projects/worktree/session.jsonl') + readClaudeTranscriptLeafUuid.mockResolvedValue('leaf-before-resume') + const record = { + sessionId: 'session-1', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/tmp/claude-home' }, + providerHandleChain: [ + { + linkId: 'claude-old', + handle: { provider: 'claude', sessionId, leafUuid: 'leaf-before-resume' }, + origin: 'created', + mintedAtFence: 1, + observedAt: 1 + } + ], + lease: { + runtimeFence: 3, + ownerProcess: { + hostId: 'local', + pid: 4343, + processStartTimeMs: 20, + spawnToken + }, + provenHandleLinkId: null + } + } as never + probeAgentSessionProcessIdentity.mockResolvedValue({ + outcome: 'identity-matched', + matchedOn: ['process-start-time'] + }) + + const transport = internal.createStructuredAgentSessionHandoffTransport() + const recovered = await transport.recoverTuiOwner(record) + expect(recovered).toMatchObject({ + transcriptPath, + link: { + handle: { provider: 'claude', sessionId, leafUuid: 'leaf-before-resume' }, + origin: 'resumed', + mintedAtFence: 3 + } + }) + expect(recovered.link.linkId).not.toBe('claude-old') + + const pty = internal.ptysById.get('pty-claude') as { + launchToken: string | null + } + pty.launchToken = null + const dispatchAuthority = runtime.getOrchestrationDispatchAuthority(recovered.terminal.handle)! + internal.restoredOrchestrationAuthorityByPtyId.set('pty-claude', { + ptyId: 'pty-claude', + worktreeId: WORKTREE_ID, + terminalHandle: recovered.terminal.handle, + paneKey: recovered.terminal.paneKey, + processIncarnation: dispatchAuthority.processIncarnation, + hostScope: dispatchAuthority.hostScope + }) + + expect( + runtime.verifyOrchestrationCompatibilityCaller({ + terminalHandle: recovered.terminal.handle, + paneKey, + launchToken: spawnToken + }) + ).toMatchObject({ + paneKey, + terminalHandle: recovered.terminal.handle, + processIncarnation: dispatchAuthority.processIncarnation + }) + expect(attestAgentHookCompatibilityAuthority).toHaveBeenCalledWith({ + paneKey, + launchTokenHash: createHash('sha256').update(spawnToken).digest('hex'), + connectionId: null, + terminalProvenance: 'restored' + }) + attestAgentHookCompatibilityAuthority.mockReturnValueOnce(null as never) + expect( + runtime.verifyOrchestrationCompatibilityCaller({ + terminalHandle: recovered.terminal.handle, + paneKey, + launchToken: spawnToken + }) + ).toBeNull() + expect(agentSessionPtyWriteGate.boundSessionId('pty-claude')).toBe('session-1') + agentSessionPtyWriteGate.unbindPty('pty-claude') + }) + + it('requires restored hook attestation after the runtime restarts', async () => { + const paneKey = 'tab-restored:leaf-restored' + const spawnToken = 'restored-token' + const sessionId = '019fd532-7c11-7a90-b6de-4e1a2c3d5f62' + const transcriptPath = '/tmp/claude-home/projects/worktree/restored.jsonl' + const attestAgentHookCompatibilityAuthority = vi.fn(() => ({ + paneKey, + source: 'hydrated_commitment' as const + })) + const runtime = new OrcaRuntimeService(null, undefined, { + attestAgentHookCompatibilityAuthority, + getAgentProviderSessionRowsForPane: () => [ + { + paneKey, + connectionId: null, + state: 'done', + prompt: '', + agentType: 'claude', + receivedAt: Date.now() + 1000, + stateStartedAt: 10, + providerSession: { key: 'session_id', id: sessionId, transcriptPath } + } + ] + }) + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + ptysById: Map + restoredOrchestrationAuthorityByPtyId: Map + } + internal.ptysById.set('pty-restored', { + ptyId: 'pty-restored', + worktreeId: WORKTREE_ID, + connectionId: null, + tabId: 'tab-restored', + paneKey, + launchToken: spawnToken, + launchAgent: 'claude', + connected: true + }) + resolveSessionFilePath.mockResolvedValue('/tmp/claude-home/projects/worktree/restored.jsonl') + readClaudeTranscriptLeafUuid.mockResolvedValue('leaf-restored') + const record = { + sessionId: 'session-restored', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/tmp/claude-home' }, + providerHandleChain: [ + { + linkId: 'claude-restored', + handle: { provider: 'claude', sessionId, leafUuid: 'leaf-restored' }, + origin: 'created', + mintedAtFence: 1, + observedAt: 1 + } + ], + lease: { + runtimeFence: 4, + ownerProcess: { + hostId: 'local', + pid: 4545, + processStartTimeMs: 30, + spawnToken + }, + provenHandleLinkId: null + } + } as never + + const recovered = await internal + .createStructuredAgentSessionHandoffTransport() + .recoverTuiOwner(record) + const restoredPty = internal.ptysById.get('pty-restored') as { + launchToken: string | null + } + restoredPty.launchToken = null + const dispatchAuthority = runtime.getOrchestrationDispatchAuthority(recovered.terminal.handle)! + internal.restoredOrchestrationAuthorityByPtyId.set('pty-restored', { + ptyId: 'pty-restored', + worktreeId: WORKTREE_ID, + terminalHandle: recovered.terminal.handle, + paneKey: recovered.terminal.paneKey, + processIncarnation: dispatchAuthority.processIncarnation, + hostScope: dispatchAuthority.hostScope + }) + + expect( + runtime.verifyOrchestrationCompatibilityCaller({ + terminalHandle: recovered.terminal.handle, + paneKey, + launchToken: spawnToken + }) + ).toMatchObject({ + paneKey, + terminalHandle: recovered.terminal.handle, + processIncarnation: dispatchAuthority.processIncarnation + }) + expect(attestAgentHookCompatibilityAuthority).toHaveBeenCalledWith({ + paneKey, + launchTokenHash: createHash('sha256').update(spawnToken).digest('hex'), + connectionId: null, + terminalProvenance: 'restored' + }) + attestAgentHookCompatibilityAuthority.mockReturnValueOnce(null as never) + await expect( + runtime.verifyOrchestrationCompatibilityCaller({ + terminalHandle: recovered.terminal.handle, + paneKey, + launchToken: spawnToken + }) + ).toBeNull() + agentSessionPtyWriteGate.unbindPty('pty-restored') + }) + + it('proves the published launch tab before returning its revealed renderer binding', async () => { + let explicitStatus: { + state: 'working' | 'done' + prompt: string + receivedAt: number + stateStartedAt: number + paneKey: string + terminalHandle: string + } | null = null + const revealTerminalSession = vi.fn( + (_worktreeId: string, _options: { tabId?: string; leafId?: string; ptyId?: string }) => + Promise.resolve({ tabId: 'tab-renderer' }) + ) + const runtime = new OrcaRuntimeService( + { + getSettings: () => ({ + disabledTuiAgents: [], + agentCmdOverrides: {}, + agentDefaultArgs: { + codex: '-m gpt-5.6-sol -c model_reasoning_effort=high' + }, + agentDefaultEnv: {} + }) + } as never, + undefined, + { + getAgentStatusSnapshot: () => (explicitStatus ? [explicitStatus as never] : []) + } + ) + runtime.setNotifier(notifier(revealTerminalSession) as never) + const spawn = vi.fn().mockResolvedValue({ id: 'pty-structured', pid: 4242 }) + runtime.setPtyController({ + spawn, + write: () => true, + kill: () => true, + getForegroundProcess: async () => null + }) + + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + resolveTerminalWorkspaceLaunchScope(): Promise<{ + id: string + path: string + connectionId: null + repo: null + folderWorkspace: null + }> + markLocalWorkspaceTrustedForAgent(): void + waitForTerminal(): Promise + waitForAdoptedStructuredTuiProof(): Promise<{ transcriptPath?: string }> + waitForStructuredTuiPtyExit(): Promise + closeTerminal(handle: string): Promise + handles: Map< + string, + { + rendererGraphEpoch: number + tabId: string + leafId: string + } + > + graphStatus: 'ready' + } + internal.resolveTerminalWorkspaceLaunchScope = vi.fn(async () => ({ + id: WORKTREE_ID, + path: '/tmp/structured-handoff', + connectionId: null, + repo: null, + folderWorkspace: null + })) + internal.markLocalWorkspaceTrustedForAgent = vi.fn() + const waitForTerminal = vi.fn(async () => ({})) + internal.waitForTerminal = waitForTerminal + const waitForAdoptedStructuredTuiProof = vi.fn(async () => { + const snapshot = await runtime.listMobileSessionTabs(`id:${WORKTREE_ID}`) + expect(snapshot.tabs).toContainEqual( + expect.objectContaining({ + type: 'terminal', + parentTabId: expect.any(String), + leafId: expect.any(String), + ptyId: 'pty-structured', + terminal: expect.any(String) + }) + ) + expect(revealTerminalSession).not.toHaveBeenCalled() + return { transcriptPath: '/tmp/rollout.jsonl' } + }) + internal.waitForAdoptedStructuredTuiProof = waitForAdoptedStructuredTuiProof + const waitForStructuredTuiPtyExit = vi.fn(async () => {}) + internal.waitForStructuredTuiPtyExit = waitForStructuredTuiPtyExit + const closeTerminal = vi.fn(async () => undefined) + internal.closeTerminal = closeTerminal + readStructuredTuiProcessIdentity.mockResolvedValue({ + hostId: 'local', + pid: 4243, + processStartTimeMs: 10, + spawnToken: 'spawn-token' + }) + probeAgentSessionProcessIdentity.mockResolvedValue({ + outcome: 'identity-matched', + matchedOn: ['process-start-time'] + }) + + const transport = internal.createStructuredAgentSessionHandoffTransport() + const onSpawned = vi.fn(async () => {}) + const owner = await transport.launchTui({ + record: { + sessionId: 'session-1', + location: { workspaceId: WORKTREE_ID, executionHostId: 'local' }, + accountHome: { variable: 'CODEX_HOME', path: '/tmp/codex-home' }, + launchArgs: ['--search'], + options: { model: 'gpt-5.6-terra', effort: 'medium' }, + providerHandleChain: [ + { handle: { provider: 'codex', threadId: 'thread-1' }, observedAt: 1 } + ] + } as never, + fence: 3, + spawnToken: 'spawn-token', + onSpawned + }) + + const reveal = revealTerminalSession.mock.calls[0]?.[1] as { + tabId: string + leafId: string + } + expect(owner.terminal).toMatchObject({ + tabId: 'tab-renderer', + paneKey: `${reveal.tabId}:${reveal.leafId}`, + ptyId: 'pty-structured' + }) + expect(waitForTerminal).toHaveBeenCalledWith( + expect.any(String), + expect.objectContaining({ condition: 'tui-idle' }) + ) + expect(waitForAdoptedStructuredTuiProof).toHaveBeenCalledOnce() + expect(onSpawned).toHaveBeenCalledWith( + expect.objectContaining({ + terminal: expect.objectContaining({ ptyId: 'pty-structured' }), + process: expect.objectContaining({ spawnToken: 'spawn-token' }) + }) + ) + expect(onSpawned.mock.invocationCallOrder[0]).toBeLessThan( + waitForTerminal.mock.invocationCallOrder[0]! + ) + expect(waitForAdoptedStructuredTuiProof.mock.invocationCallOrder[0]).toBeLessThan( + revealTerminalSession.mock.invocationCallOrder[0]! + ) + const launchCommand = spawn.mock.calls[0]?.[0]?.command + expect(launchCommand).toContain("'-m' 'gpt-5.6-terra'") + expect(launchCommand).toContain("'-c' 'model_reasoning_effort=medium'") + expect(launchCommand).toContain("'--search'") + expect(launchCommand).not.toContain('gpt-5.6-sol') + expect(launchCommand).not.toContain('model_reasoning_effort=high') + + Object.assign(internal.handles.get(owner.terminal.handle)!, { + rendererGraphEpoch: -1, + tabId: 'tab-retired', + leafId: 'leaf-retired' + }) + internal.graphStatus = 'ready' + + explicitStatus = { + state: 'working', + prompt: '', + receivedAt: Date.now(), + stateStartedAt: Date.now(), + paneKey: owner.terminal.paneKey, + terminalHandle: owner.terminal.handle + } + expect(transport.tuiStatus(owner)).toBe('busy') + await expect( + transport.waitForTuiIdleOrExit(owner, new AbortController().signal) + ).resolves.toBeNull() + + explicitStatus = { ...explicitStatus, state: 'done', receivedAt: Date.now() } + expect(transport.tuiStatus(owner)).toBe('idle') + await expect(transport.waitForTuiIdleOrExit(owner, new AbortController().signal)).resolves.toBe( + 'idle' + ) + + explicitStatus = null + const livePty = ( + runtime as unknown as { + ptysById: Map< + string, + { + tailBuffer: string[] + tailPartialLine: string + preview: string + lastAgentStatus: null + lastAgentStatusObservedLive: boolean + } + > + } + ).ptysById.get('pty-structured')! + Object.assign(livePty, { + tailBuffer: [ + 'OpenAI Codex (v0.147.0)', + 'model: gpt-5.6-terra', + 'directory: /tmp/structured-handoff' + ], + tailPartialLine: '', + preview: '', + lastAgentStatus: null, + lastAgentStatusObservedLive: false + }) + expect(transport.tuiStatus(owner)).toBe('idle') + await expect(transport.waitForTuiIdleOrExit(owner, new AbortController().signal)).resolves.toBe( + 'idle' + ) + + const pty = ( + runtime as unknown as { + ptysById: Map + } + ).ptysById.get('pty-structured')! + pty.launchToken = null + const persistedRecord = { + sessionId: 'session-1', + providerHandleChain: [{ handle: { provider: 'codex', threadId: 'thread-1' }, observedAt: 1 }], + lease: { ownerProcess: owner.process, provenHandleLinkId: owner.link.linkId } + } as never + + const rebound = await transport.reproveTuiOwner({ record: persistedRecord, owner }) + expect(rebound.terminal).toMatchObject({ + ptyId: 'pty-structured', + tabId: owner.terminal.tabId, + paneKey: owner.terminal.paneKey + }) + expect(rebound.terminal.handle).not.toBe(owner.terminal.handle) + await transport.waitForTuiExit(rebound) + expect(waitForStructuredTuiPtyExit).toHaveBeenCalledWith('pty-structured') + expect(waitForAdoptedStructuredTuiProof).toHaveBeenCalledOnce() + + await expect(transport.closeTuiOwner?.(rebound)).resolves.toEqual({ + transcriptPath: '/tmp/rollout.jsonl' + }) + expect(closeTerminal).toHaveBeenCalledWith(rebound.terminal.handle) + + explicitStatus = null + pty.connected = false + await expect( + transport.waitForTuiIdleOrExit(rebound, new AbortController().signal) + ).resolves.toBe('exited') + await expect(transport.stopFailedTuiLaunch?.(rebound)).resolves.toBeUndefined() + }) + + it('reveals Claude structured native sessions into the mobile graph', async () => { + const runtime = new OrcaRuntimeService() + const publish = vi.spyOn(runtime, 'publishStructuredAgentSessionTab') + const focusEditorTab = vi.fn() + runtime.setNotifier({ focusEditorTab } as never) + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + } + + await internal.createStructuredAgentSessionHandoffTransport().revealNativeSession?.({ + workspaceId: WORKTREE_ID, + sessionId: 'session-claude', + agent: 'claude' + }) + + expect(publish).toHaveBeenCalledWith( + expect.objectContaining({ + workspaceId: WORKTREE_ID, + sessionId: 'session-claude', + agent: 'claude', + activate: false + }) + ) + expect(focusEditorTab).toHaveBeenCalledWith( + 'structured-agent-session-session-claude', + WORKTREE_ID + ) + }) +}) diff --git a/src/main/runtime/orca-runtime-tests/reattach-headless-grid.spec.ts b/src/main/runtime/orca-runtime-tests/reattach-headless-grid.spec.ts new file mode 100644 index 00000000000..43ba687c4bf --- /dev/null +++ b/src/main/runtime/orca-runtime-tests/reattach-headless-grid.spec.ts @@ -0,0 +1,47 @@ +import { describe, expect, it } from 'vitest' +import { createRuntime, syncSinglePty } from '../orca-runtime-test-fixtures.spec' + +describe('headless model grid after a reattach', () => { + it('reflows a model that live bytes created at the 80x24 default onto the PTY grid', async () => { + const runtime = createRuntime() + // No controller size: mirrors a reattach whose real grid main only learns from the reply. + runtime.setPtyController({ + write: () => true, + kill: () => true, + getForegroundProcess: async () => null, + getSize: () => null + }) + syncSinglePty(runtime, 'pty-1') + + runtime.onPtyData('pty-1', 'user@host % claude\r\n', 100) + await expect( + runtime.serializeMainTerminalBuffer('pty-1', { scrollbackRows: 100 }) + ).resolves.toMatchObject({ cols: 80, rows: 24 }) + + runtime.reflowHeadlessTerminalToPtyGrid('pty-1', 211, 57) + + await expect( + runtime.serializeMainTerminalBuffer('pty-1', { scrollbackRows: 100 }) + ).resolves.toMatchObject({ cols: 211, rows: 57, source: 'headless' }) + }) + + it('never creates a model for a PTY that has none', async () => { + const runtime = createRuntime() + runtime.setPtyController({ + write: () => true, + kill: () => true, + getForegroundProcess: async () => null, + getSize: () => null + }) + syncSinglePty(runtime, 'pty-1') + + runtime.reflowHeadlessTerminalToPtyGrid('pty-1', 211, 57) + runtime.onPtyData('pty-1', 'hello\r\n', 100) + + // Why it matters: commit reflows every reattach, and pre-creating here would defeat the + // renderer-authority gate that deliberately leaves the model unseeded. + await expect( + runtime.serializeMainTerminalBuffer('pty-1', { scrollbackRows: 100 }) + ).resolves.toMatchObject({ cols: 80, rows: 24 }) + }) +}) diff --git a/src/main/runtime/orca-runtime.test.ts b/src/main/runtime/orca-runtime.test.ts index 9d223233ef2..05739f4fe39 100644 --- a/src/main/runtime/orca-runtime.test.ts +++ b/src/main/runtime/orca-runtime.test.ts @@ -34,6 +34,7 @@ await import('./orca-runtime-tests/terminal-side-effect-facts-part-03.spec') await import('./orca-runtime-tests/decorative-title-fact-throttle.spec') await import('./orca-runtime-tests/headless-snapshots.spec') await import('./orca-runtime-tests/headless-snapshots-part-02.spec') +await import('./orca-runtime-tests/reattach-headless-grid.spec') await import('./orca-runtime-tests/agent-status-and-waits.spec') await import('./orca-runtime-tests/agent-status-and-waits-part-02.spec') await import('./orca-runtime-tests/agent-status-and-waits-part-03.spec') diff --git a/src/main/runtime/orchestration/__snapshots__/preamble.test.ts.snap b/src/main/runtime/orchestration/__snapshots__/preamble.test.ts.snap index d92228517aa..8b7ffc115a7 100644 --- a/src/main/runtime/orchestration/__snapshots__/preamble.test.ts.snap +++ b/src/main/runtime/orchestration/__snapshots__/preamble.test.ts.snap @@ -10,6 +10,7 @@ Slack, GitHub comments, or any other channel to reach a human during the run. === CLI COMMANDS === +\`\`\`sh # Report the terminal task outcome (REQUIRED exactly once). # # RULE: --body must be a 3-sentence executive summary (what you did, @@ -71,6 +72,7 @@ Slack, GitHub comments, or any other channel to reach a human during the run. # Check for messages from the coordinator: orca orchestration check --terminal term_WORKER +\`\`\` === AFTER YOU SEND worker_done === diff --git a/src/main/runtime/orchestration/coordinator-task-dispatch.ts b/src/main/runtime/orchestration/coordinator-task-dispatch.ts index 685552cc2e9..e058d9cdc98 100644 --- a/src/main/runtime/orchestration/coordinator-task-dispatch.ts +++ b/src/main/runtime/orchestration/coordinator-task-dispatch.ts @@ -132,7 +132,7 @@ export async function dispatchTaskToWorker(params: { let gateContext = '' if (gates.length > 0) { const latest = gates.at(-1)! - gateContext = `\n\n--- DECISION GATE RESOLVED ---\nQuestion: ${latest.question}\nResolution: ${latest.resolution}\n---\n` + gateContext = `\n\n--- DECISION GATE RESOLVED ---\nQuestion: ${latest.question}\nResolution: ${latest.resolution}\n\n---\n` } try { diff --git a/src/main/runtime/orchestration/preamble.test.ts b/src/main/runtime/orchestration/preamble.test.ts index 79e06f5a55a..57b8b35f266 100644 --- a/src/main/runtime/orchestration/preamble.test.ts +++ b/src/main/runtime/orchestration/preamble.test.ts @@ -1,4 +1,6 @@ import { spawnSync } from 'node:child_process' +import remarkParse from 'remark-parse' +import { unified } from 'unified' import { describe, expect, it } from 'vitest' import { buildDispatchPreamble } from './preamble' @@ -23,6 +25,22 @@ function afterWorkerDoneSection(result: string) { return result.slice(sectionStart, sectionEnd) } +function cliFence(result: string): string { + const match = result.match(/=== CLI COMMANDS ===\n\n```sh\n([\s\S]*?)\n```/) + expect(match).not.toBeNull() + return match?.[1] ?? '' +} + +function markdownBlocks(result: string) { + const tree = unified().use(remarkParse).parse(result) + return { + headings: tree.children.filter((node) => node.type === 'heading'), + codeBlocks: tree.children.filter((node) => node.type === 'code') + } +} + +const driftParams = { base: 'origin/main', behind: 3, recentSubjects: ['fix: a', 'feat: b'] } + describe('buildDispatchPreamble', () => { it('substitutes template variables', () => { const result = buildDispatchPreamble(baseParams()) @@ -58,25 +76,43 @@ describe('buildDispatchPreamble', () => { { timeout: 15_000 }, () => { const result = buildDispatchPreamble(baseParams()) - // Why: feeding `bash -n` the full preamble falsely fails on apostrophes - // in the surrounding prose. Slice between the CLI markers and strip - // shell-style comment lines so we only syntax-check the commands. - const cliStart = result.indexOf('=== CLI COMMANDS ===') - const cliEnd = result.indexOf('=== AFTER YOU SEND worker_done ===') - expect(cliStart).toBeGreaterThan(-1) - expect(cliEnd).toBeGreaterThan(cliStart) - const block = result.slice(cliStart, cliEnd) - const stripped = block - .split('\n') - .filter((line) => !line.trim().startsWith('#')) - .filter((line) => !line.trim().startsWith('===')) - .join('\n') - - const check = spawnSync('bash', ['-n'], { input: stripped, encoding: 'utf8' }) + const check = spawnSync('bash', ['-n'], { input: cliFence(result), encoding: 'utf8' }) expect(check.status).toBe(0) } ) + it('fences shell comments so Markdown does not promote them to headings', () => { + const result = buildDispatchPreamble(baseParams()) + const { headings, codeBlocks } = markdownBlocks(result) + + expect(headings).toHaveLength(0) + expect(codeBlocks).toHaveLength(1) + expect(codeBlocks[0]).toMatchObject({ lang: 'sh', value: cliFence(result) }) + }) + + // Why: a `---` rule directly under a paragraph is a setext H2, so the optional + // sections' closing rules must not turn their last sentence into a heading. + it('renders no Markdown headings when the sub-dispatch and drift sections are present', () => { + const result = buildDispatchPreamble( + baseParams({ canDispatchSubWorkers: true, baseDrift: driftParams }) + ) + const { headings, codeBlocks } = markdownBlocks(result) + + expect(headings).toHaveLength(0) + expect(codeBlocks).toHaveLength(2) + expect(codeBlocks[1]).toMatchObject({ lang: 'sh' }) + expect(codeBlocks[1].value).toContain('orchestration worker-start --task ') + expect(result).toContain('able to dispatch further.\n\n---') + expect(result).toContain('before starting.\n\n---') + }) + + it('sub-dispatch fence passes bash -n', { timeout: 15_000 }, () => { + const result = buildDispatchPreamble(baseParams({ canDispatchSubWorkers: true })) + const { codeBlocks } = markdownBlocks(result) + const check = spawnSync('bash', ['-n'], { input: codeBlocks[1].value, encoding: 'utf8' }) + expect(check.status).toBe(0) + }) + it('includes heartbeat CLI block with taskId and dispatchId and 5-minute cadence', () => { const result = buildDispatchPreamble(baseParams()) expect(result).toContain('--type heartbeat') diff --git a/src/main/runtime/orchestration/preamble.ts b/src/main/runtime/orchestration/preamble.ts index 7d426184954..d4519f154b9 100644 --- a/src/main/runtime/orchestration/preamble.ts +++ b/src/main/runtime/orchestration/preamble.ts @@ -59,6 +59,7 @@ export function buildDispatchPreamble(params: PreambleParams): string { ? ` --dispatch-capability ${params.dispatchCapability}` : '' + // Why: fencing keeps shell comments executable to agents without turning them into Chat UI headings. const header = `You are working inside Orca, a multi-agent IDE. You are a dispatched worker. Your coordinator's terminal handle is: ${params.coordinatorHandle} Your task ID is: ${params.taskId} @@ -68,6 +69,7 @@ Slack, GitHub comments, or any other channel to reach a human during the run. === CLI COMMANDS === +\`\`\`sh # Report the terminal task outcome (REQUIRED exactly once). # # RULE: --body must be a 3-sentence executive summary (what you did, @@ -129,6 +131,7 @@ Slack, GitHub comments, or any other channel to reach a human during the run. # Check for messages from the coordinator: ${cli} orchestration check --terminal ${params.workerHandle} +\`\`\` ${postDoneInstructions}` @@ -193,6 +196,8 @@ work under the new Dispatch; ignore stale follow-ups from the settled task.` // Why the whole section is omitted rather than softened when nesting is off: a // worker told it "usually cannot" delegate still tries, then reports the refusal // as a blocker. +// Why fenced + blank line before the closing `---`: unfenced `` are stripped as raw +// HTML by the Chat UI, and a rule directly under a paragraph is a setext H2 (giant last sentence). function buildSubDispatchSection(cli: string): string { return ` @@ -200,13 +205,16 @@ function buildSubDispatchSection(cli: string): string { You may dispatch sub-workers for this task. Bind your own Run first, then create and start each one: +\`\`\`sh ${cli} orchestration run-create --objective "" --json ${cli} orchestration task-create --spec "" --json ${cli} orchestration worker-start --task --worktree current --agent --json +\`\`\` You own those sub-workers: wait for their worker_done, and do not report your own until they have settled. Nesting is capped, so a sub-worker of yours may not be able to dispatch further. + ---` } @@ -221,5 +229,6 @@ ${subjects} If any look relevant to your task, either pull them in (\`git pull --rebase ${drift.base}\` or equivalent) or escalate to the coordinator before starting. + ---` } diff --git a/src/main/runtime/rpc/e2ee-channel-v2.test.ts b/src/main/runtime/rpc/e2ee-channel-v2.test.ts index f9e602ced41..b26b057abaf 100644 --- a/src/main/runtime/rpc/e2ee-channel-v2.test.ts +++ b/src/main/runtime/rpc/e2ee-channel-v2.test.ts @@ -156,6 +156,25 @@ describe('E2EEChannel v2', () => { }) }) + it('forwards post-auth capability-shaped frames without mutating authenticated capabilities', () => { + const ctx = setup() + const { schedule } = startV2(ctx) + const onMessage = vi.fn() + ctx.channel.onMessage(onMessage) + authenticate(ctx, schedule) + + const capabilityFrame = JSON.stringify({ + type: 'e2ee_client_capabilities', + v: 1, + clientCapabilities: ['agent-session.structured.v1'] + }) + ctx.channel.handleRawMessage(clientText(capabilityFrame, schedule, 1n)) + + expect(ctx.channel.clientCapabilities).toEqual([]) + expect(onMessage).toHaveBeenCalledOnce() + expect(onMessage.mock.calls[0]?.[0]).toBe(capabilityFrame) + }) + it('rejects legacy downgrade and runtime-only capability metadata when mobile v2 is required', () => { const legacy = setup() legacy.channel.handleRawMessage( diff --git a/src/main/runtime/rpc/methods/clipboard.test.ts b/src/main/runtime/rpc/methods/clipboard.test.ts index 118b21c766a..c0d224b85c6 100644 --- a/src/main/runtime/rpc/methods/clipboard.test.ts +++ b/src/main/runtime/rpc/methods/clipboard.test.ts @@ -1,6 +1,6 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { RpcDispatcher } from '../dispatcher' -import type { RpcRequest } from '../core' +import type { RpcRequest, RpcResponse } from '../core' import type { OrcaRuntimeService } from '../../orca-runtime' import { CLIPBOARD_IMAGE_MAX_BASE64_CHARS, @@ -21,6 +21,10 @@ import { CLIPBOARD_METHODS, resetClipboardImageUploadsForTest } from './clipboard' +import { + hasMobileClipboardImagePath, + resetMobileClipboardImageProvenanceForTest +} from '../mobile-clipboard-image-provenance' function makeRequest(method: string, params?: unknown): RpcRequest { return { id: 'req-1', authToken: 'tok', method, params } @@ -31,15 +35,36 @@ function makeDispatcher(): RpcDispatcher { return new RpcDispatcher({ runtime, methods: CLIPBOARD_METHODS }) } +async function callMobile( + dispatcher: RpcDispatcher, + method: string, + params: unknown, + clientId = 'device-a' +): Promise { + const replies: RpcResponse[] = [] + await dispatcher.dispatchStreaming( + makeRequest(method, params), + (raw) => replies.push(JSON.parse(raw) as RpcResponse), + { clientKind: 'mobile', clientId } + ) + const response = replies[0] + if (!response) { + throw new Error(`no reply for ${method}`) + } + return response +} + describe('clipboard RPC methods', () => { beforeEach(() => { saveClipboardImageBufferAsTempFile.mockReset() resetClipboardImageUploadsForTest() + resetMobileClipboardImageProvenanceForTest() }) afterEach(() => { vi.useRealTimers() resetClipboardImageUploadsForTest() + resetMobileClipboardImageProvenanceForTest() }) it('saves browser-provided clipboard image bytes on the runtime host', async () => { @@ -64,6 +89,37 @@ describe('clipboard RPC methods', () => { }) }) + it('records a successful direct mobile upload for only the authenticated client', async () => { + const path = '/tmp/orca-paste-image.png' + saveClipboardImageBufferAsTempFile.mockResolvedValue(path) + const dispatcher = makeDispatcher() + + await expect( + callMobile(dispatcher, 'clipboard.saveImageAsTempFile', { + contentBase64: Buffer.from('png-bytes').toString('base64'), + connectionId: null + }) + ).resolves.toMatchObject({ ok: true, result: path }) + + expect(hasMobileClipboardImagePath('device-a', path)).toBe(true) + expect(hasMobileClipboardImagePath('device-b', path)).toBe(false) + }) + + it('does not authorize a remote-host clipboard path for local structured delivery', async () => { + const path = '/tmp/orca-paste-image.png' + saveClipboardImageBufferAsTempFile.mockResolvedValue(path) + const dispatcher = makeDispatcher() + + await expect( + callMobile(dispatcher, 'clipboard.saveImageAsTempFile', { + contentBase64: Buffer.from('png-bytes').toString('base64'), + connectionId: 'ssh-1' + }) + ).resolves.toMatchObject({ ok: true, result: path }) + + expect(hasMobileClipboardImagePath('device-a', path)).toBe(false) + }) + it('rejects non-base64 clipboard image payloads', async () => { const dispatcher = makeDispatcher() @@ -140,6 +196,48 @@ describe('clipboard RPC methods', () => { expect(saveClipboardImageBufferAsTempFile).toHaveBeenCalledWith(Buffer.from('png-bytes'), { connectionId: 'ssh-1' }) + expect(hasMobileClipboardImagePath('device-a', '/tmp/orca-paste-image.png')).toBe(false) + }) + + it('binds chunk mutation and provenance to the mobile client that started the upload', async () => { + saveClipboardImageBufferAsTempFile.mockResolvedValue('/tmp/orca-paste-image.png') + const dispatcher = makeDispatcher() + const contentBase64 = Buffer.from('png-bytes').toString('base64') + const start = await callMobile(dispatcher, 'clipboard.startImageUpload', { + expectedBase64Length: contentBase64.length, + connectionId: null + }) + const uploadId = (start.ok ? start.result : null) as { uploadId: string } + + for (const method of [ + 'clipboard.appendImageUploadChunk', + 'clipboard.commitImageUpload', + 'clipboard.abortImageUpload' + ]) { + const params = + method === 'clipboard.appendImageUploadChunk' + ? { uploadId: uploadId.uploadId, offset: 0, contentBase64 } + : { uploadId: uploadId.uploadId } + await expect(callMobile(dispatcher, method, params, 'device-b')).resolves.toMatchObject({ + ok: false + }) + } + + await expect( + callMobile(dispatcher, 'clipboard.appendImageUploadChunk', { + uploadId: uploadId.uploadId, + offset: 0, + contentBase64 + }) + ).resolves.toMatchObject({ + ok: true, + result: { receivedBase64Length: contentBase64.length } + }) + await expect( + callMobile(dispatcher, 'clipboard.commitImageUpload', { uploadId: uploadId.uploadId }) + ).resolves.toMatchObject({ ok: true, result: '/tmp/orca-paste-image.png' }) + expect(hasMobileClipboardImagePath('device-a', '/tmp/orca-paste-image.png')).toBe(true) + expect(hasMobileClipboardImagePath('device-b', '/tmp/orca-paste-image.png')).toBe(false) }) it('rejects out-of-order chunk offsets', async () => { diff --git a/src/main/runtime/rpc/methods/clipboard.ts b/src/main/runtime/rpc/methods/clipboard.ts index 3d5212c7a52..e6b487d7761 100644 --- a/src/main/runtime/rpc/methods/clipboard.ts +++ b/src/main/runtime/rpc/methods/clipboard.ts @@ -1,11 +1,12 @@ import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod, type RpcContext, type RpcMethod } from '../core' import { saveClipboardImageBufferAsTempFile } from '../../../window/clipboard-image-temp-file' import { randomUUID } from 'node:crypto' import { CLIPBOARD_IMAGE_MAX_BASE64_CHARS, CLIPBOARD_IMAGE_TOO_LARGE_ERROR } from '../../../../shared/clipboard-image' +import { recordMobileClipboardImagePath } from '../mobile-clipboard-image-provenance' const MAX_CLIPBOARD_IMAGE_BASE64_CHARS = CLIPBOARD_IMAGE_MAX_BASE64_CHARS export const CLIPBOARD_IMAGE_UPLOAD_CHUNK_BASE64_CHARS = 512 * 1024 @@ -16,6 +17,7 @@ const BASE64_PATTERN = /^[A-Za-z0-9+/]*={0,2}$/ type ClipboardImageUpload = { expectedBase64Length: number connectionId?: string | null + mobileClientId?: string chunks: string[] receivedBase64Length: number expiresAt: number @@ -69,6 +71,28 @@ function getUpload(uploadId: string): ClipboardImageUpload { return upload } +function mobileClientId(ctx: RpcContext): string | undefined { + if (ctx.clientKind !== 'mobile') { + return undefined + } + const clientId = ctx.clientId?.trim() + if (!clientId) { + throw new Error('Clipboard image upload requires an authenticated mobile client') + } + return clientId +} + +function assertMobileUploadOwner( + upload: ClipboardImageUpload, + ctx: RpcContext +): string | undefined { + const clientId = mobileClientId(ctx) + if (clientId && upload.mobileClientId !== clientId) { + throw new Error('Clipboard image upload was not found') + } + return clientId +} + function assertValidBase64Content(value: string): void { if (!isValidBase64(value)) { throw new Error('Clipboard image content must be base64') @@ -131,15 +155,24 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ defineMethod({ name: 'clipboard.saveImageAsTempFile', params: SaveImageAsTempFile, - handler: async (params) => - saveClipboardImageBufferAsTempFile(Buffer.from(params.contentBase64, 'base64'), { - connectionId: params.connectionId - }) + handler: async (params, ctx) => { + const clientId = mobileClientId(ctx) + const path = await saveClipboardImageBufferAsTempFile( + Buffer.from(params.contentBase64, 'base64'), + { + connectionId: params.connectionId + } + ) + if (clientId && !params.connectionId) { + recordMobileClipboardImagePath(clientId, path) + } + return path + } }), defineMethod({ name: 'clipboard.startImageUpload', params: StartImageUpload, - handler: (params) => { + handler: (params, ctx) => { pruneExpiredUploads() if (clipboardImageUploads.size >= CLIPBOARD_IMAGE_UPLOAD_MAX_CONCURRENT) { throw new Error('Too many clipboard image uploads are in progress') @@ -148,6 +181,7 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ clipboardImageUploads.set(uploadId, { expectedBase64Length: params.expectedBase64Length, connectionId: params.connectionId, + mobileClientId: mobileClientId(ctx), chunks: [], receivedBase64Length: 0, expiresAt: Date.now() + CLIPBOARD_IMAGE_UPLOAD_TTL_MS, @@ -159,8 +193,9 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ defineMethod({ name: 'clipboard.appendImageUploadChunk', params: AppendImageUploadChunk, - handler: (params) => { + handler: (params, ctx) => { const upload = getUpload(params.uploadId) + assertMobileUploadOwner(upload, ctx) if (params.offset !== upload.receivedBase64Length) { throw new Error('Clipboard image chunk offset is out of order') } @@ -177,17 +212,25 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ defineMethod({ name: 'clipboard.commitImageUpload', params: CommitImageUpload, - handler: async (params) => { + handler: async (params, ctx) => { const upload = getUpload(params.uploadId) + const clientId = assertMobileUploadOwner(upload, ctx) try { if (upload.receivedBase64Length !== upload.expectedBase64Length) { throw new Error('Clipboard image upload is incomplete') } const contentBase64 = upload.chunks.join('') assertValidBase64Content(contentBase64) - return await saveClipboardImageBufferAsTempFile(Buffer.from(contentBase64, 'base64'), { - connectionId: upload.connectionId - }) + const path = await saveClipboardImageBufferAsTempFile( + Buffer.from(contentBase64, 'base64'), + { + connectionId: upload.connectionId + } + ) + if (clientId && !upload.connectionId) { + recordMobileClipboardImagePath(clientId, path) + } + return path } finally { // Why: failed SSH or filesystem commits must not leave bounded upload // memory pinned until TTL cleanup. @@ -198,7 +241,12 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ defineMethod({ name: 'clipboard.abortImageUpload', params: AbortImageUpload, - handler: (params) => { + handler: (params, ctx) => { + pruneExpiredUploads() + const upload = clipboardImageUploads.get(params.uploadId) + if (upload) { + assertMobileUploadOwner(upload, ctx) + } deleteUpload(params.uploadId) return { aborted: true } } diff --git a/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts b/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts new file mode 100644 index 00000000000..dcba8b7b64e --- /dev/null +++ b/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts @@ -0,0 +1,22 @@ +import { defineMethod, type RpcAnyMethod } from '../core' +import { ActivateTab, SaveMarkdownTab } from './session-tabs-schemas' + +export const MOBILE_MARKDOWN_TAB_METHODS: RpcAnyMethod[] = [ + defineMethod({ + name: 'markdown.readTab', + params: ActivateTab, + handler: async (params, { runtime }) => + runtime.readMobileMarkdownTab(params.worktree, params.tabId) + }), + defineMethod({ + name: 'markdown.saveTab', + params: SaveMarkdownTab, + handler: async (params, { runtime }) => + runtime.saveMobileMarkdownTab( + params.worktree, + params.tabId, + params.baseVersion, + params.content + ) + }) +] diff --git a/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts b/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts index 226277f6ebd..183f981ccee 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it, vi } from 'vitest' import { + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, type RuntimeCapability } from '../../../../shared/protocol-version' @@ -49,6 +50,8 @@ const METHODS = [ } ] as const +const DESTRUCTIVE_METHOD_NAMES = new Set(['session.tabs.close', 'session.tabs.closeLifecycle']) + describe('session tab structured capability mutations', () => { for (const method of METHODS) { it(`rejects ${method.name} when the structured row is hidden`, async () => { @@ -67,13 +70,56 @@ describe('session tab structured capability mutations', () => { expect(fixture.calls[method.runtimeMethod]).toHaveBeenCalledOnce() }) - it(`rejects ${method.name} for a legacy Claude row`, async () => { + it(`rejects ${method.name} on a Claude row the client never negotiated`, async () => { const fixture = createFixture([STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]) const response = await fixture.dispatch(method.name, method.params('claude-session')) expect(response.ok).toBe(false) expect(fixture.calls[method.runtimeMethod]).not.toHaveBeenCalled() }) + + it(`allows ${method.name} for a client that negotiated Claude rows`, async () => { + const fixture = createFixture([ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ]) + const response = await fixture.dispatch(method.name, method.params('claude-session')) + + expect(response.ok).toBe(true) + expect(fixture.calls[method.runtimeMethod]).toHaveBeenCalledOnce() + }) + } + + for (const method of METHODS) { + const expectedToAllowPromptedRow = !DESTRUCTIVE_METHOD_NAMES.has(method.name) + + it(`${expectedToAllowPromptedRow ? 'allows' : 'rejects'} ${method.name} on a row an old mobile client was prompted to update`, async () => { + const { calls, dispatch } = createFixture([], { + clientKind: 'mobile', + structuredNativeChatEnabled: true + }) + + const response = await dispatch(method.name, method.params('codex-session')) + + expect(response.ok).toBe(expectedToAllowPromptedRow) + expect(calls[method.runtimeMethod as keyof typeof calls]).toHaveBeenCalledTimes( + expectedToAllowPromptedRow ? 1 : 0 + ) + }) + + it(`${expectedToAllowPromptedRow ? 'allows' : 'rejects'} ${method.name} on a prompted Claude row for a mobile client without the Claude capability`, async () => { + const { calls, dispatch } = createFixture([STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], { + clientKind: 'mobile', + structuredNativeChatEnabled: true + }) + + const response = await dispatch(method.name, method.params('claude-session')) + + expect(response.ok).toBe(expectedToAllowPromptedRow) + expect(calls[method.runtimeMethod as keyof typeof calls]).toHaveBeenCalledTimes( + expectedToAllowPromptedRow ? 1 : 0 + ) + }) } it.each(['session.tabs.close', 'session.tabs.closeLifecycle'] as const)( @@ -119,7 +165,10 @@ describe('session tab structured capability mutations', () => { ) }) -function createFixture(capabilities: RuntimeCapability[]) { +function createFixture( + capabilities: RuntimeCapability[], + options: { clientKind?: 'mobile' | 'runtime'; structuredNativeChatEnabled?: boolean } = {} +) { const snapshot = agentSnapshot() const calls = { closeMobileSessionTab: vi.fn().mockResolvedValue({ closed: true }), @@ -130,11 +179,14 @@ function createFixture(capabilities: RuntimeCapability[]) { const runtime = { getRuntimeId: () => 'test-runtime', listMobileSessionTabs: vi.fn().mockResolvedValue(snapshot), + getClientSettings: () => ({ + experimentalStructuredNativeChat: options.structuredNativeChatEnabled === true + }), ...calls } as unknown as OrcaRuntimeService const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) const context: RpcDispatchStreamingOptions = { - clientKind: 'runtime', + clientKind: options.clientKind ?? 'runtime', pairedDeviceId: 'paired-client', clientCapabilities: capabilities } diff --git a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts index 4a61a99bc20..cf68f7739c0 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts @@ -1,10 +1,15 @@ import { describe, expect, it } from 'vitest' import { AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import type { RuntimeMobileSessionTabsSnapshot } from '../../../../shared/runtime-types' -import { projectSessionTabAgentStatus } from './session-tab-agent-status-projection' +import { + CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE, + STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE, + projectSessionTabAgentStatus +} from './session-tab-agent-status-projection' function makeSnapshot(sessionBoundary: boolean): RuntimeMobileSessionTabsSnapshot { return { @@ -119,44 +124,178 @@ describe('projectSessionTabAgentStatus', () => { expect(capable).toBe(snapshot) }) - it('withholds legacy Claude rows from paired structured clients', () => { - const snapshot = { - ...makeSnapshot(false), - tabs: [ - { - type: 'agent-session', - id: 'agent-session:codex', - title: 'Codex Chat', - sessionId: 'codex', - agent: 'codex', - isActive: true - }, - { - type: 'agent-session', - id: 'agent-session:claude', - title: 'Claude Chat', - sessionId: 'claude', - agent: 'claude', - isActive: false - } - ], - activeTabId: 'agent-session:codex', - activeTabType: 'agent-session' + const claudeSnapshot = { + ...makeSnapshot(false), + tabs: [ + { + type: 'agent-session', + id: 'agent-session:codex', + title: 'Codex Chat', + sessionId: 'codex', + agent: 'codex', + isActive: true + }, + { + type: 'agent-session', + id: 'agent-session:claude', + title: 'Claude Chat', + sessionId: 'claude', + agent: 'claude', + isActive: false + } + ], + activeGroupId: 'group-a', + activeTabId: 'agent-session:codex', + activeTabType: 'agent-session', + tabGroups: [ + { id: 'group-a', activeTabId: 'agent-session:codex', tabOrder: ['agent-session:codex'] }, + { id: 'group-b', activeTabId: 'agent-session:claude', tabOrder: ['agent-session:claude'] } + ], + tabGroupLayout: { + type: 'split', + direction: 'horizontal', + first: { type: 'leaf', groupId: 'group-a' }, + second: { type: 'leaf', groupId: 'group-b' } + } + } as unknown as RuntimeMobileSessionTabsSnapshot + + const structuredMobile = [ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ] + + it('withholds Claude rows from a paired runtime client that never negotiated them', () => { + const projected = projectSessionTabAgentStatus( + claudeSnapshot, + 'runtime', + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + true + ) + + expect(projected.tabs.map((tab) => tab.id)).toEqual(['agent-session:codex']) + // A row pruned from `tabs` but left in the layout is its own dead tab. + expect(projected.tabGroups?.map((group) => group.id)).toEqual(['group-a']) + expect(projected.tabGroupLayout).toEqual({ type: 'leaf', groupId: 'group-a' }) + expect(projected.activeGroupId).toBe('group-a') + expect(projected.activeTabId).toBe('agent-session:codex') + expect(projected.activeTabType).toBe('agent-session') + }) + + it('uses a desktop fallback for an unsupported Claude row instead of withholding it', () => { + const projected = projectSessionTabAgentStatus( + claudeSnapshot, + 'mobile', + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + true + ) + + // The row survives so the chat the desktop shows is not simply absent on the phone. + expect(projected.tabs.map((tab) => tab.id)).toEqual([ + 'agent-session:codex', + 'agent-session:claude' + ]) + expect(projected.tabs.map((tab) => tab.title)).toEqual([ + 'Codex Chat', + CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE + ]) + // Nothing is removed, so the layout it belonged to is untouched. + expect(projected.tabGroups?.map((group) => group.id)).toEqual(['group-a', 'group-b']) + expect(projected.tabGroupLayout).toEqual(claudeSnapshot.tabGroupLayout) + expect(projected.activeTabId).toBe('agent-session:codex') + }) + + it('projects agent-specific fallback titles for a mobile client with no capabilities', () => { + const projected = projectSessionTabAgentStatus(claudeSnapshot, 'mobile', [], true) + + expect(projected.tabs.map((tab) => tab.title)).toEqual([ + STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE, + CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE + ]) + expect(projected.tabGroupLayout).toEqual(claudeSnapshot.tabGroupLayout) + }) + + it('does not treat the Claude capability as a substitute for the base structured capability', () => { + const projected = projectSessionTabAgentStatus( + claudeSnapshot, + 'mobile', + [CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + true + ) + + expect(projected.tabs.map((tab) => tab.title)).toEqual([ + STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE, + CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE + ]) + }) + + it('shows both real titles once mobile negotiates Claude', () => { + expect(projectSessionTabAgentStatus(claudeSnapshot, 'mobile', structuredMobile, true)).toBe( + claudeSnapshot + ) + }) + + // Why: updating cannot reveal a chat the desktop is not serving, so the prompt would lie. + it('withholds rather than prompts when the desktop experiment is off', () => { + for (const capabilities of [ + [], + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + structuredMobile + ]) { + const projected = projectSessionTabAgentStatus(claudeSnapshot, 'mobile', capabilities, false) + expect(projected.tabs).toEqual([]) + } + }) + + it('never emits an empty structured tab title', () => { + for (const capabilities of [[], [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]]) { + const projected = projectSessionTabAgentStatus(claudeSnapshot, 'mobile', capabilities, true) + for (const tab of projected.tabs) { + expect(tab.title.length).toBeGreaterThan(0) + } + } + }) + + it.each([ + ['mobile', 'mobile' as const, structuredMobile], + [ + 'runtime', + 'runtime' as const, + [ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ] + ] + ])( + 'publishes Claude rows to a paired %s client that negotiated them', + (_name, clientKind, capabilities) => { + const projected = projectSessionTabAgentStatus(claudeSnapshot, clientKind, capabilities, true) + + expect(projected).toBe(claudeSnapshot) + expect(projected.tabGroupLayout).toEqual(claudeSnapshot.tabGroupLayout) + } + ) + + it('keeps Claude rows on the local renderer, which negotiates nothing', () => { + expect(projectSessionTabAgentStatus(claudeSnapshot, undefined, undefined)).toBe(claudeSnapshot) + expect(projectSessionTabAgentStatus(claudeSnapshot, undefined, [])).toBe(claudeSnapshot) + }) + + it('leaves Codex rows untouched whether or not the Claude capability is present', () => { + const codexOnly = { + ...claudeSnapshot, + tabs: claudeSnapshot.tabs.filter((tab) => tab.id !== 'agent-session:claude'), + tabGroups: claudeSnapshot.tabGroups?.filter((group) => group.id !== 'group-b'), + tabGroupLayout: { type: 'leaf', groupId: 'group-a' } } as unknown as RuntimeMobileSessionTabsSnapshot - expect( - projectSessionTabAgentStatus(snapshot, 'runtime', [ - STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY - ]).tabs.map((tab) => tab.id) - ).toEqual(['agent-session:codex']) - expect( - projectSessionTabAgentStatus( - snapshot, - 'mobile', - [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], - true - ).tabs.map((tab) => tab.id) - ).toEqual(['agent-session:codex']) + for (const capabilities of [[STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], structuredMobile]) { + for (const clientKind of ['mobile', 'runtime'] as const) { + expect(projectSessionTabAgentStatus(codexOnly, clientKind, capabilities, true)).toBe( + codexOnly + ) + } + } + expect(projectSessionTabAgentStatus(codexOnly, undefined, undefined)).toBe(codexOnly) }) it('withholds session boundaries from legacy paired clients', () => { diff --git a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts index 375b3b499d5..4496fdc5435 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts @@ -1,5 +1,7 @@ import { AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, type RuntimeCapability } from '../../../../shared/protocol-version' import type { @@ -12,6 +14,43 @@ import { structuredNativeChatProjectionEnabled } from './structured-agent-sessio type SessionTabsPayload = RuntimeMobileSessionTabsResult | RuntimeMobileSessionTabsSnapshot +/** Capped at 128px / one line in every shipped mobile build, so ~15-18 characters render. */ +export const STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE = 'Update to view' +export const CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE = 'Open on desktop' + +function clientCanRenderStructuredAgentSessionTab( + tab: RuntimeMobileSessionAgentTab, + clientCapabilities: readonly RuntimeCapability[] | undefined +): boolean { + if (!clientCapabilities?.includes(STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY)) { + return false + } + return ( + tab.agent === 'codex' || + clientCapabilities.includes(CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) + ) +} + +function resolveMobileStructuredChatFallbackTitle( + tab: RuntimeMobileSessionAgentTab, + args: { + clientKind: 'mobile' | 'runtime' | undefined + clientCapabilities: readonly RuntimeCapability[] | undefined + structuredNativeChatEnabled?: boolean + } +): string | null { + if ( + args.clientKind !== 'mobile' || + args.structuredNativeChatEnabled !== true || + clientCanRenderStructuredAgentSessionTab(tab, args.clientCapabilities) + ) { + return null + } + return tab.agent === 'claude' + ? CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE + : STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE +} + export function projectSessionTabAgentStatus( payload: TPayload, clientKind: 'mobile' | 'runtime' | undefined, @@ -23,9 +62,27 @@ export function projectSessionTabAgentStatus true) - if (structuredVisible && clientKind !== undefined) { - projected = projectAgentSessionTabsOut(projected, (tab) => tab.agent !== 'codex') + let projected: TPayload + if (clientKind === 'mobile' && structuredNativeChatEnabled === true) { + // Why: deleting the row left the user hunting for a chat the desktop says exists; the row + // survives with a title naming the fix. Nothing is removed, so no group/layout repair applies. + projected = projectUnsupportedAgentSessionTabTitles(payload, { + clientKind, + clientCapabilities, + structuredNativeChatEnabled + }) + } else { + projected = structuredVisible ? payload : projectAgentSessionTabsOut(payload, () => true) + // Why: a paired client renders only codex structured tabs unless it says otherwise + // (mobile's resolveMobileNativeChat returns null for every other agent), so an + // ungated row would list and select into a pane that shows neither chat nor terminal. + if ( + structuredVisible && + clientKind !== undefined && + !clientCapabilities?.includes(CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) + ) { + projected = projectAgentSessionTabsOut(projected, (tab) => tab.agent !== 'codex') + } } // Why: only paired runtimes have legacy `done` completion side effects; mobile must keep its row without changing the exact v2 auth shape. if ( @@ -47,6 +104,47 @@ export function projectSessionTabAgentStatus( + payload: TPayload, + args: { + clientKind: 'mobile' + clientCapabilities: readonly RuntimeCapability[] | undefined + structuredNativeChatEnabled: true + } +): TPayload { + let changed = false + const tabs = payload.tabs.map((tab) => { + if (tab.type !== 'agent-session') { + return tab + } + const title = resolveMobileStructuredChatFallbackTitle(tab, args) + if (title === null) { + return tab + } + changed = true + return { ...tab, title } + }) + return changed ? ({ ...payload, tabs } as TPayload) : payload +} + +export function assertAgentSessionTabDestructiveMutationSupported( + payload: SessionTabsPayload, + tabId: string, + clientKind: 'mobile' | 'runtime' | undefined, + clientCapabilities: readonly RuntimeCapability[] | undefined +): void { + if (clientKind === undefined) { + return + } + const tab = payload.tabs.find((candidate) => candidate.id === tabId) + if ( + tab?.type === 'agent-session' && + !clientCanRenderStructuredAgentSessionTab(tab, clientCapabilities) + ) { + throw new Error('structured_agent_session_unsupported') + } +} + function projectAgentSessionTabsOut( payload: TPayload, shouldHide: (tab: RuntimeMobileSessionAgentTab) => boolean diff --git a/src/main/runtime/rpc/methods/session-tab-close-methods.ts b/src/main/runtime/rpc/methods/session-tab-close-methods.ts index bd60ecd6ddf..361ba8e4c51 100644 --- a/src/main/runtime/rpc/methods/session-tab-close-methods.ts +++ b/src/main/runtime/rpc/methods/session-tab-close-methods.ts @@ -3,6 +3,7 @@ import { SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY } from '../../../../shared/ import { defineMethod, type RpcAnyMethod } from '../core' import { CloseLifecycleTab, CloseTab } from './session-tabs-schemas' import { assertProjectedSessionTabVisible } from './session-tab-browser-placement-projection' +import { assertAgentSessionTabDestructiveMutationSupported } from './session-tab-agent-status-projection' import { projectSessionTabsForClient } from './session-tabs-inventory' import { isStructuredNativeChatEnabled } from './structured-agent-session-policy' @@ -12,8 +13,12 @@ export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ params: CloseTab, handler: async (params, context) => { if (context.clientKind) { + const raw = await context.runtime.listMobileSessionTabs( + params.worktree, + context.pairedDeviceId + ) const visible = projectSessionTabsForClient( - await context.runtime.listMobileSessionTabs(params.worktree, context.pairedDeviceId), + raw, context.clientKind, context.clientCapabilities, context.clientKind === 'mobile' @@ -21,6 +26,12 @@ export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ : undefined ) assertProjectedSessionTabVisible(visible, params.tabId) + assertAgentSessionTabDestructiveMutationSupported( + raw, + params.tabId, + context.clientKind, + context.clientCapabilities + ) } const requiresIntent = context.clientKind === undefined || @@ -81,8 +92,12 @@ export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ params: CloseLifecycleTab, handler: async (params, context) => { if (context.clientKind) { + const raw = await context.runtime.listMobileSessionTabs( + params.worktree, + context.pairedDeviceId + ) const visible = projectSessionTabsForClient( - await context.runtime.listMobileSessionTabs(params.worktree, context.pairedDeviceId), + raw, context.clientKind, context.clientCapabilities, context.clientKind === 'mobile' @@ -90,6 +105,12 @@ export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ : undefined ) assertProjectedSessionTabVisible(visible, params.tabId) + assertAgentSessionTabDestructiveMutationSupported( + raw, + params.tabId, + context.clientKind, + context.clientCapabilities + ) } return withSpan( 'runtime.session-tabs.close-lifecycle', diff --git a/src/main/runtime/rpc/methods/session-tabs-structured-restore.test.ts b/src/main/runtime/rpc/methods/session-tabs-structured-restore.test.ts new file mode 100644 index 00000000000..083f334285e --- /dev/null +++ b/src/main/runtime/rpc/methods/session-tabs-structured-restore.test.ts @@ -0,0 +1,98 @@ +import { describe, expect, it, vi } from 'vitest' +import { RpcDispatcher } from '../dispatcher' +import type { RpcRequest } from '../core' +import type { OrcaRuntimeService } from '../../orca-runtime' +import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' +import { SESSION_TAB_METHODS } from './session-tabs' + +function makeRequest(method: string, params?: unknown): RpcRequest { + return { id: 'req-1', authToken: 'tok', method, params } +} + +describe('session tab structured restore gating', () => { + it('does not restore structured tabs for mobile while the host setting is off', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: false })), + restoreStructuredAgentSessionTabs: vi.fn(), + listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) + + const response = await dispatcher.dispatch( + makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), + { + clientKind: 'mobile', + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + } + ) + + expect(response.ok).toBe(true) + expect(runtime.restoreStructuredAgentSessionTabs).not.toHaveBeenCalled() + }) + + // Why: an old build has no capability to advertise, and skipping the restore left it with + // nothing to project after a desktop restart — neither the chat nor its fallback row. + it('restores structured tabs for a mobile client that advertises no capability', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: true })), + restoreStructuredAgentSessionTabs: vi.fn(), + listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) + + const response = await dispatcher.dispatch( + makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), + { clientKind: 'mobile', clientCapabilities: [] } + ) + + expect(response.ok).toBe(true) + expect(runtime.restoreStructuredAgentSessionTabs).toHaveBeenCalledTimes(1) + }) + + it('restores structured tabs for mobile once the setting is present', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: true })), + restoreStructuredAgentSessionTabs: vi.fn(), + listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) + + const response = await dispatcher.dispatch( + makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), + { + clientKind: 'mobile', + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + } + ) + + expect(response.ok).toBe(true) + expect(runtime.restoreStructuredAgentSessionTabs).toHaveBeenCalledTimes(1) + }) +}) + +function visibleSnapshot() { + return { + worktree: 'wt-1', + publicationEpoch: 'epoch-1', + snapshotVersion: 1, + activeGroupId: 'group-1', + activeTabId: 'tab-1::leaf-1', + activeTabType: 'terminal' as const, + tabGroups: [{ id: 'group-1', activeTabId: 'tab-1', tabOrder: ['tab-1'] }], + tabs: [ + { + type: 'terminal' as const, + id: 'tab-1::leaf-1', + parentTabId: 'tab-1', + leafId: 'leaf-1', + title: 'Terminal', + status: 'ready' as const, + terminal: 'pty-1', + isActive: true + } + ] + } +} diff --git a/src/main/runtime/rpc/methods/session-tabs.test.ts b/src/main/runtime/rpc/methods/session-tabs.test.ts index 29131869fa0..be61fc55edf 100644 --- a/src/main/runtime/rpc/methods/session-tabs.test.ts +++ b/src/main/runtime/rpc/methods/session-tabs.test.ts @@ -2,10 +2,7 @@ import { describe, expect, it, vi } from 'vitest' import { RpcDispatcher } from '../dispatcher' import type { RpcRequest } from '../core' import type { OrcaRuntimeService } from '../../orca-runtime' -import { - SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY, - STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY -} from '../../../../shared/protocol-version' +import { SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import { SESSION_TAB_METHODS } from './session-tabs' function makeRequest(method: string, params?: unknown): RpcRequest { @@ -13,48 +10,6 @@ function makeRequest(method: string, params?: unknown): RpcRequest { } describe('session tab RPC methods', () => { - it('does not restore structured tabs for mobile while the host setting is off', async () => { - const runtime = { - getRuntimeId: () => 'test-runtime', - getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: false })), - restoreStructuredAgentSessionTabs: vi.fn(), - listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) - } as unknown as OrcaRuntimeService - const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) - - const response = await dispatcher.dispatch( - makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), - { - clientKind: 'mobile', - clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] - } - ) - - expect(response.ok).toBe(true) - expect(runtime.restoreStructuredAgentSessionTabs).not.toHaveBeenCalled() - }) - - it('restores structured tabs for mobile only after capability and setting are present', async () => { - const runtime = { - getRuntimeId: () => 'test-runtime', - getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: true })), - restoreStructuredAgentSessionTabs: vi.fn(), - listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) - } as unknown as OrcaRuntimeService - const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) - - const response = await dispatcher.dispatch( - makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), - { - clientKind: 'mobile', - clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] - } - ) - - expect(response.ok).toBe(true) - expect(runtime.restoreStructuredAgentSessionTabs).toHaveBeenCalledTimes(1) - }) - it('routes mobile-only activation without notifying desktop clients', async () => { const runtime = { getRuntimeId: () => 'test-runtime', diff --git a/src/main/runtime/rpc/methods/structured-agent-session-gate.ts b/src/main/runtime/rpc/methods/structured-agent-session-gate.ts index 33018c21f4d..e15544b2d24 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-gate.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-gate.ts @@ -2,8 +2,11 @@ // // Shared by every structured method file so one gate governs the whole surface: a client that does // not advertise `agent-session.structured.v1` is told the surface does not exist rather than being -// handed a session it cannot render or drive — and, just as importantly, cannot make the host EXIST -// by calling into it, which is an observable side effect. +// handed the session journal or mutation surface. +// +// This gate no longer implies such a client cannot make the host exist: session-tab restore runs +// for old mobile clients while structured chat is enabled so they receive a fallback row, and that +// path constructs the host. `agentSession.*` stays refused either way, which is what this gate is for. import { getStructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-registry' import type { StructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-host' diff --git a/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.test.ts b/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.test.ts new file mode 100644 index 00000000000..1a62045c85b --- /dev/null +++ b/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.test.ts @@ -0,0 +1,239 @@ +// The create route's pre-commit boundary: a failure before `attach` must reach the client as a +// refusal it can classify, and a failure at or after `attach` must not. + +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-host' +import { setStructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-registry' +import { computeAgentSessionPayloadFingerprint } from '../../../../shared/agent-session-mutation-envelope' +import { isDefinitiveAgentSessionCreateRefusal } from '../../../../shared/agent-session-definitive-refusal' +import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' +import type { OrcaRuntimeService } from '../../orca-runtime' +import type { RpcResponse } from '../core' +import { RpcDispatcher } from '../dispatcher' +import { STRUCTURED_AGENT_SESSION_METHODS } from './structured-agent-session' + +const SESSION = 'session-alpha' +const OPERATION = '1800000000000-00000000000000000000000000000001' +const WORKTREE = 'id:workspace-1' + +const STRUCTURED_CLIENT = { + clientKind: 'runtime' as const, + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] +} + +function createParams(overrides: Record = {}) { + return { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION, + expectedRuntimeFence: null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: SESSION, + fields: { worktree: WORKTREE, agent: 'codex' } + }), + ...(overrides.envelope as Record | undefined) + }, + worktree: WORKTREE, + agent: 'codex' + } +} + +let attach: ReturnType + +function hostStub(): StructuredAgentSessionHost { + attach = vi.fn(async () => ({ + ok: true, + replayed: false, + fence: 1, + cursor: { epoch: 'epoch-a', sequence: 0 }, + value: { sessionId: SESSION, fence: 1, page: {}, unconfirmedClientMessageIds: [] } + })) + return { attach } as unknown as StructuredAgentSessionHost +} + +const resolvedIntent = { + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }, + provider: 'codex', + agent: 'codex', + accountHome: { variable: 'CODEX_HOME', path: '/host/.codex' }, + runtimeKind: 'native' +} + +async function create( + runtimeOverrides: Record = {}, + params: unknown = createParams() +): Promise { + const runtime = { + getRuntimeId: () => 'runtime-1', + registerSubscriptionCleanup: vi.fn(), + cleanupSubscription: vi.fn(), + cleanupSubscriptionsByPrefix: vi.fn(), + ensureStructuredAgentSessionHost: vi.fn(async () => undefined), + resolveStructuredAgentSessionCreateIntent: vi.fn(async (input: { envelope: unknown }) => ({ + envelope: input.envelope, + ...resolvedIntent + })), + publishStructuredAgentSessionTab: vi.fn(async () => undefined), + ...runtimeOverrides + } + const replies: RpcResponse[] = [] + await new RpcDispatcher({ + runtime: runtime as unknown as OrcaRuntimeService, + methods: STRUCTURED_AGENT_SESSION_METHODS + }).dispatchStreaming( + { id: 'request-1', authToken: 'token', method: 'agentSession.create', params }, + (raw) => replies.push(JSON.parse(raw) as RpcResponse), + STRUCTURED_CLIENT + ) + const first = replies[0] + if (!first) { + throw new Error('no reply for agentSession.create') + } + return first +} + +/** The refusal a client can act on, or null when the reply was not one. */ +function refusalOf(response: RpcResponse): { code: string; message: string } | null { + if (!response.ok) { + return null + } + const result = response.result as { ok: boolean; refusal?: { code: string; message: string } } + return result.ok ? null : (result.refusal ?? null) +} + +beforeEach(() => { + setStructuredAgentSessionHost(hostStub()) + vi.spyOn(console, 'warn').mockImplementation(() => undefined) +}) + +afterEach(() => { + setStructuredAgentSessionHost(null) + vi.restoreAllMocks() +}) + +describe('a create refused before it commits', () => { + it('answers a code-carrying refusal as a definitive envelope', async () => { + const response = await create({ + resolveStructuredAgentSessionCreateIntent: vi.fn(async () => { + throw new Error('structured_agent_session_unsupported') + }) + }) + + const refusal = refusalOf(response) + expect(refusal?.code).toBe('structured_agent_session_unsupported') + expect(refusal?.message).toContain('structured agent chat') + expect(refusal?.message).not.toContain('Codex') + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(true) + expect(attach).not.toHaveBeenCalled() + }) + + it('answers a code-less failure as a definitive envelope too, keeping the cause in the message', async () => { + // The class no per-site conversion catches: an unresolvable worktree throws prose, not a code. + const response = await create({ + resolveStructuredAgentSessionCreateIntent: vi.fn(async () => { + throw new Error('No worktree matches id:workspace-1') + }) + }) + + const refusal = refusalOf(response) + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(true) + expect(refusal?.message).toContain('structured agent chat') + expect(refusal?.message).not.toContain('Codex') + expect(refusal?.message).toContain('No worktree matches id:workspace-1') + expect(attach).not.toHaveBeenCalled() + }) + + it('answers a host that will not install as a definitive envelope', async () => { + setStructuredAgentSessionHost(null) + + const response = await create({ + ensureStructuredAgentSessionHost: vi.fn(async () => { + throw new Error('EACCES: could not open the session store') + }) + }) + + const refusal = refusalOf(response) + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(true) + expect(refusal?.message).toContain('could not open the session store') + }) + + it('answers a missing host as a definitive envelope rather than a thrown code', async () => { + setStructuredAgentSessionHost(null) + + const response = await create() + + const refusal = refusalOf(response) + expect(refusal?.code).toBe('structured_agent_session_unsupported') + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(true) + }) + + it('still refuses a fingerprint conflict with its own code, not a pre-commit one', async () => { + const response = await create( + {}, + createParams({ envelope: { payloadFingerprint: 'a'.repeat(64) } }) + ) + + expect(refusalOf(response)?.code).toBe('agent_session_operation_conflict') + expect(attach).not.toHaveBeenCalled() + }) +}) + +describe('the boundary the envelope stops at', () => { + it('leaves a failure at attach unknown, because it may have committed', async () => { + attach.mockRejectedValueOnce(new Error('attach exploded')) + + const response = await create() + + expect(response).toMatchObject({ ok: false, error: { code: 'runtime_error' } }) + expect(refusalOf(response)).toBeNull() + }) + + it('leaves a committed create whose tab could not be published unknown', async () => { + const response = await create({ + publishStructuredAgentSessionTab: vi.fn(async () => { + throw new Error('publish failed') + }) + }) + + const refusal = refusalOf(response) + expect(refusal?.code).toBe('agent_session_operation_unknown') + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(false) + }) + + it('keeps hiding the surface from a client that never advertised it', async () => { + const replies: RpcResponse[] = [] + await new RpcDispatcher({ + runtime: { getRuntimeId: () => 'runtime-1' } as unknown as OrcaRuntimeService, + methods: STRUCTURED_AGENT_SESSION_METHODS + }).dispatchStreaming( + { + id: 'request-1', + authToken: 'token', + method: 'agentSession.create', + params: createParams() + }, + (raw) => replies.push(JSON.parse(raw) as RpcResponse), + { clientKind: 'runtime', clientCapabilities: [] } + ) + + expect(replies[0]).toMatchObject({ + ok: false, + error: { message: expect.stringContaining('structured_agent_session_unsupported') } + }) + }) + + it('keeps a client-declared fence a programming error, not a refusal', async () => { + const response = await create({}, createParams({ envelope: { expectedRuntimeFence: 1 } })) + + expect(response).toMatchObject({ + ok: false, + error: { code: 'agent_session_operation_invalid' } + }) + }) +}) diff --git a/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.ts b/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.ts new file mode 100644 index 00000000000..24d9fa2aec4 --- /dev/null +++ b/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.ts @@ -0,0 +1,71 @@ +// Nothing before `attach` commits a session, so every failure in that span definitively created +// nothing. Thrown, it reaches a remote client as a generic transport error, indistinguishable from +// an answer that was lost on the way back — and a client that cannot tell those apart either +// strands the user with no chat and no terminal, or spawns a sibling beside a session that may +// already exist. So the whole span answers with a refusal envelope carrying a code, whatever it +// failed on. +// +// Converting the span rather than each throw site is deliberate: alongside the throws that carry a +// code there is a code-less class — an unresolvable worktree, a store that will not open, a host +// that will not install — that no per-site list catches, and it is exactly the class that reaches +// the user as nothing at all. + +import { + AGENT_SESSION_WIRE_REFUSAL_CODES, + type AgentSessionWireRefusal, + type AgentSessionWireRefusalCode +} from '../../../../shared/agent-session-wire' + +export type StructuredCreateRefused = { refusal: AgentSessionWireRefusal } + +/** A pre-commit failure with no code of its own still proves the host could not serve a structured + * session for this request and did not create one, which is what `unsupported` says on the wire. + * A new code would say it more precisely, but only to clients new enough to know it. */ +const UNCODED_PRECOMMIT_REFUSAL_CODE: AgentSessionWireRefusalCode = + 'structured_agent_session_unsupported' + +function wireRefusalCode(error: unknown): AgentSessionWireRefusalCode | null { + const candidates = [ + error instanceof Error && 'code' in error ? (error as { code: unknown }).code : undefined, + error instanceof Error ? error.message : String(error) + ] + for (const candidate of candidates) { + if ( + typeof candidate === 'string' && + (AGENT_SESSION_WIRE_REFUSAL_CODES as readonly string[]).includes(candidate) + ) { + return candidate as AgentSessionWireRefusalCode + } + } + return null +} + +function precommitRefusal(error: unknown): AgentSessionWireRefusal { + const code = wireRefusalCode(error) + if (code) { + return { code, message: 'Orca cannot open a structured agent chat for this workspace.' } + } + const message = error instanceof Error ? error.message : String(error) + // A code-less failure here is often a defect, not a policy answer; the refusal keeps the user + // moving, the log keeps the cause findable. + console.warn('[agent-session] create refused before it committed anything', error) + return { + code: UNCODED_PRECOMMIT_REFUSAL_CODE, + message: `Orca could not prepare a structured agent chat for this workspace: ${message}` + } +} + +/** + * Runs the pre-commit half of a create. Anything it throws becomes a refusal; a refusal it returns + * itself passes through. Must not wrap `attach` or anything after it — past that point a failure no + * longer proves the session does not exist. + */ +export async function resolveUncommittedStructuredCreate( + prepare: () => Promise +): Promise { + try { + return await prepare() + } catch (error) { + return { refusal: precommitRefusal(error) } + } +} diff --git a/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts b/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts index 6a9372ed2c1..da66923c731 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts @@ -98,7 +98,7 @@ export const CreateIntentParams = z .object({ envelope: MutationEnvelope, worktree: Identifier('Invalid worktree selector'), - agent: z.literal('codex') + agent: z.enum(['claude', 'codex']) }) .strict() @@ -107,7 +107,7 @@ export const CreateParams = z.union([AttachParams, CreateIntentParams]) export const CreateSupportParams = z .object({ worktree: Identifier('Invalid worktree selector'), - agent: z.literal('codex') + agent: z.enum(['claude', 'codex']) }) .strict() @@ -149,7 +149,11 @@ export const SendParams = z .strict() export const CancelParams = z - .object({ envelope: MutationEnvelope, turnId: Identifier('Invalid turn id') }) + .object({ + envelope: MutationEnvelope, + turnId: Identifier('Invalid turn id'), + scope: z.literal('background-tasks').optional() + }) .strict() export const RespondParams = z @@ -170,6 +174,15 @@ export const SetOptionParams = z }) .strict() +export const HandoffParams = z + .object({ + envelope: MutationEnvelope, + direction: z.enum(['to-tui', 'to-native']), + mode: z.enum(['now', 'after-turn', 'stop-turn']), + action: z.enum(['start', 'cancel-queued', 'retry', 'recover']).optional() + }) + .strict() + export const OptionsParams = z.object({ sessionId: SessionId }).strict() /** One surface's claim on one session. The id names the surface, not the client: two chat views diff --git a/src/main/runtime/rpc/methods/structured-agent-session.test.ts b/src/main/runtime/rpc/methods/structured-agent-session.test.ts index b65e6eff825..e4888e4a9df 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.test.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.test.ts @@ -99,6 +99,22 @@ function hostStub(): StructuredAgentSessionHost { setSessionTabVisibility: vi.fn(async () => undefined), respondToPrompt: vi.fn(async () => ({ ok: true, replayed: false })), setOption: vi.fn(async () => ({ ok: true, replayed: false })), + requestHandoff: vi.fn(async () => ({ + ok: true, + replayed: false, + fence: 1, + cursor: { epoch: 'epoch-a', sequence: 0 }, + value: { + status: { + owner: 'native', + direction: null, + phase: 'idle', + stage: null, + operationId: null + } + } + })), + supportsCreate: vi.fn(() => true), handoffStatus: vi.fn(async () => ({ owner: 'native' })), readOptions: vi.fn(async () => ({ models: [{ id: 'gpt-live', label: 'GPT Live', isDefault: true, efforts: [] }], @@ -122,9 +138,12 @@ function dispatcher(runtimeOverrides: Record = {}): RpcDispatch workspaceId: 'workspace-1', workspaceKind: 'git-worktree' }, - provider: 'codex', - agent: 'codex', - accountHome: { variable: 'CODEX_HOME', path: '/host/.codex' }, + provider: params.agent, + agent: params.agent, + accountHome: { + variable: params.agent === 'claude' ? 'CLAUDE_CONFIG_DIR' : 'CODEX_HOME', + path: params.agent === 'claude' ? '/host/.claude' : '/host/.codex' + }, runtimeKind: 'native' })), publishStructuredAgentSessionTab: vi.fn() @@ -221,7 +240,7 @@ describe('capability gating', () => { } // Bump deliberately: the whole agentSession.* surface is behind the structured capability, // so an additive method is invisible to old clients and needs no protocol bump. - expect(STRUCTURED_AGENT_SESSION_METHODS).toHaveLength(16) + expect(STRUCTURED_AGENT_SESSION_METHODS).toHaveLength(17) }) it('hides the surface from a declared client that did not advertise it', async () => { @@ -329,6 +348,49 @@ describe('method routing', () => { ) }) + it('routes Claude create support and create through the provider-aware runtime', async () => { + const worktree = 'id:workspace-1' + const support = await call( + 'agentSession.createSupport', + { worktree, agent: 'claude' }, + STRUCTURED_CLIENT + ) + expect(support).toMatchObject({ ok: true, result: { supported: true } }) + expect(runtimeCalls.getStructuredAgentSessionCreateSupport).toHaveBeenCalledWith( + worktree, + 'claude' + ) + + const params = { + envelope: envelope({ + expectedRuntimeFence: null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: SESSION, + fields: { worktree, agent: 'claude' } + }) + }), + worktree, + agent: 'claude' + } + const created = await call('agentSession.create', params, STRUCTURED_CLIENT) + expect(created).toMatchObject({ ok: true, result: { ok: true } }) + expect(runtimeCalls.resolveStructuredAgentSessionCreateIntent).toHaveBeenCalledWith(params) + expect(hostCalls.attach).toHaveBeenCalledWith( + expect.anything(), + expect.objectContaining({ + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/host/.claude' } + }) + ) + expect(runtimeCalls.publishStructuredAgentSessionTab).toHaveBeenCalledWith( + expect.objectContaining({ + sessionId: SESSION, + activate: true, + agent: 'claude' + }) + ) + }) + it('reports an unknown create outcome when attach commits before tab publication fails', async () => { const worktree = 'id:workspace-1' const params = { @@ -371,6 +433,38 @@ describe('method routing', () => { expect(ensured).toMatchObject({ ok: true }) }) + /** A client-supplied location skips the worktree-resolving support check, so both attach-shaped + * entries must ask the executing host directly or a host that cannot fence a provider child + * would create one anyway. */ + it('returns a refusal envelope when create cannot support a client-supplied location', async () => { + hostCalls.supportsCreate.mockReturnValue(false) + + const refused = await call('agentSession.create', attachParams()) + + expect(refused).toMatchObject({ + ok: true, + result: { + ok: false, + refusal: { code: 'structured_agent_session_unsupported' } + } + }) + expect(hostCalls.attach).not.toHaveBeenCalled() + expect(hostCalls.supportsCreate).toHaveBeenCalledWith(attachParams().location, 'codex') + }) + + it('keeps ensure failures as top-level errors for an unsupported client location', async () => { + hostCalls.supportsCreate.mockReturnValue(false) + + const refused = await call('agentSession.ensure', attachParams()) + + expect(refused).toMatchObject({ + ok: false, + error: { message: expect.stringContaining('structured_agent_session_unsupported') } + }) + expect(hostCalls.attach).not.toHaveBeenCalled() + expect(hostCalls.supportsCreate).toHaveBeenCalledWith(attachParams().location, 'codex') + }) + it('tags the prompt kind from the method name, not from the client', async () => { const params = { envelope: envelope(), @@ -386,7 +480,7 @@ describe('method routing', () => { ]) }) - it('does not register the structured handoff mutation', async () => { + it('routes the structured handoff mutation through the host', async () => { const response = await call('agentSession.requestHandoff', { envelope: envelope(), direction: 'to-tui', @@ -394,7 +488,11 @@ describe('method routing', () => { action: 'start' }) - expect(response).toMatchObject({ ok: false, error: { code: 'method_not_found' } }) + expect(response).toMatchObject({ ok: true }) + expect(hostCalls.requestHandoff).toHaveBeenCalledWith( + expect.anything(), + expect.objectContaining({ direction: 'to-tui', mode: 'now', action: 'start' }) + ) }) }) @@ -434,25 +532,6 @@ describe('parameter validation', () => { ) }) - it('rejects Claude structured create shapes', async () => { - await rejects('agentSession.createSupport', { - worktree: 'id:workspace-1', - agent: 'claude' - }) - const fields = { worktree: 'id:workspace-1', agent: 'claude' } - await rejects('agentSession.create', { - envelope: envelope({ - expectedRuntimeFence: null, - payloadFingerprint: computeAgentSessionPayloadFingerprint({ - method: 'agentSession.create', - sessionId: SESSION, - fields - }) - }), - ...fields - }) - }) - it('requires a sha256 fingerprint and a positive fence', async () => { await rejects( 'agentSession.send', diff --git a/src/main/runtime/rpc/methods/structured-agent-session.ts b/src/main/runtime/rpc/methods/structured-agent-session.ts index ffd23499a3e..b69ff6fd628 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.ts @@ -2,14 +2,14 @@ // // Every method here is gated on the client advertising // `agent-session.structured.v1`. A client that does not is told the surface does -// not exist rather than being handed a session it cannot render or drive; that -// is the whole visibility rule, because nothing else on the runtime publishes a -// structured session. +// not exist rather than receiving the journal or mutation surface. Session-tab +// inventory may expose only a metadata placeholder for an incapable mobile client. import { agentSessionFingerprintConflict, computeAgentSessionPayloadFingerprint } from '../../../../shared/agent-session-mutation-envelope' +import type { z } from 'zod' import { defineMethod, defineStreamingMethod, type RpcAnyMethod, type RpcContext } from '../core' import { ensureStructuredHostInstalled as ensureHostInstalled, @@ -18,13 +18,16 @@ import { structuredCallerFor as callerFor, supportsStructuredSessions } from './structured-agent-session-gate' +import type { AgentSessionAttachParams } from '../../../native-chat/agent-session-wire/structured-agent-session-attach' import { STRUCTURED_AGENT_SESSION_HOLD_METHODS } from './structured-agent-session-hold' +import { resolveUncommittedStructuredCreate } from './structured-agent-session-precommit-refusal' import { AttachParams, CancelParams, CreateParams, CreateSupportParams, HistoryParams, + HandoffParams, HandoffStatusParams, OptionsParams, RespondParams, @@ -43,6 +46,35 @@ function subscriptionIdFor(ctx: RpcContext, sessionId: string): string { return ctx.requestId ? `${base}:${ctx.requestId}` : base } +/** + * The attach-shaped entries take the location from the client instead of resolving it from a + * worktree, so they never reach the worktree-resolving create-support check. Ask the executing + * host the same question directly: the answer includes host-measured facts the client cannot see + * or forge, such as whether this machine can read a provider child's process start time. + */ +async function resolveClientSuppliedAttach(params: z.infer, ctx: RpcContext) { + await ensureHostInstalled(ctx) + const host = requireHost(ctx) + if (!host.supportsCreate(params.location, params.agent)) { + throw new Error('structured_agent_session_unsupported') + } + const { agent: _attachAgent, provider: _attachProvider, ...attachWithoutAgent } = params + const attachParams = { + ...attachWithoutAgent, + provider: params.provider as 'claude' | 'codex', + agent: params.agent as 'claude' | 'codex' + } as AgentSessionAttachParams + return { host, attachParams } +} + +async function attachClientSuppliedLocation( + params: z.infer, + ctx: RpcContext +): Promise { + const { host, attachParams } = await resolveClientSuppliedAttach(params, ctx) + return host.attach(callerFor(ctx), attachParams) +} + export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ defineMethod({ name: 'agentSession.createSupport', @@ -62,66 +94,82 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ if (params.envelope.expectedRuntimeFence !== null) { throw new Error('agent_session_operation_invalid') } - if ('worktree' in params) { - const intentFingerprint = computeAgentSessionPayloadFingerprint({ - method: 'agentSession.create', - sessionId: params.envelope.sessionId, - fields: { worktree: params.worktree, agent: params.agent } - }) - const conflict = agentSessionFingerprintConflict(params.envelope, intentFingerprint) - if (conflict) { - return { ok: false, refusal: conflict } - } - const resolved = await ctx.runtime.resolveStructuredAgentSessionCreateIntent(params) - const hostFingerprint = computeAgentSessionPayloadFingerprint({ - method: 'agentSession.attach', - sessionId: params.envelope.sessionId, - fields: { - location: resolved.location, - provider: resolved.provider, - agent: resolved.agent, - accountHome: resolved.accountHome, - runtimeKind: resolved.runtimeKind, - expectedRuntimeFence: null + // Everything up to `attach` is pre-commit, and answers with a refusal rather than a throw so + // a client can tell "nothing was created" from "the outcome is unknown". + const prepared = await resolveUncommittedStructuredCreate(async () => { + if ('worktree' in params) { + const intentFingerprint = computeAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: params.envelope.sessionId, + fields: { worktree: params.worktree, agent: params.agent } + }) + const conflict = agentSessionFingerprintConflict(params.envelope, intentFingerprint) + if (conflict) { + return { refusal: conflict } } - }) - await ensureHostInstalled(ctx) - const result = await requireHost(ctx).attach(callerFor(ctx), { - ...resolved, - envelope: { ...params.envelope, payloadFingerprint: hostFingerprint } - }) - if (result.ok && resolved.agent === 'codex') { - try { - await ctx.runtime.publishStructuredAgentSessionTab({ + const resolved = await ctx.runtime.resolveStructuredAgentSessionCreateIntent(params) + const hostFingerprint = computeAgentSessionPayloadFingerprint({ + method: 'agentSession.attach', + sessionId: params.envelope.sessionId, + fields: { + location: resolved.location, + provider: resolved.provider, + agent: resolved.agent, + accountHome: resolved.accountHome, + runtimeKind: resolved.runtimeKind, + expectedRuntimeFence: null + } + }) + await ensureHostInstalled(ctx) + const { agent: _resolvedAgent, provider: _resolvedProvider, ...resolvedAttach } = resolved + const attachParams: AgentSessionAttachParams = { + ...resolvedAttach, + provider: resolved.provider as 'claude' | 'codex', + agent: resolved.agent as 'claude' | 'codex', + envelope: { ...params.envelope, payloadFingerprint: hostFingerprint } + } + return { + host: requireHost(ctx), + attachParams, + tab: { workspaceId: resolved.location.workspaceId, - sessionId: result.value.sessionId, - agent: 'codex', - activate: true - }) - } catch (error) { - console.warn('[agent-session] create committed before tab publication failed', error) - return { - ok: false, - refusal: { - code: 'agent_session_operation_unknown', - message: 'The Codex chat may have been created, but its tab could not be confirmed.' - } + agent: resolved.agent as 'claude' | 'codex' } } } - return result + const { host, attachParams } = await resolveClientSuppliedAttach(params, ctx) + return { host, attachParams, tab: null } + }) + if ('refusal' in prepared) { + return { ok: false, refusal: prepared.refusal } } - await ensureHostInstalled(ctx) - return requireHost(ctx).attach(callerFor(ctx), params) + const result = await prepared.host.attach(callerFor(ctx), prepared.attachParams) + if (result.ok && prepared.tab) { + try { + await ctx.runtime.publishStructuredAgentSessionTab({ + workspaceId: prepared.tab.workspaceId, + sessionId: result.value.sessionId, + agent: prepared.tab.agent, + activate: true + }) + } catch (error) { + console.warn('[agent-session] create committed before tab publication failed', error) + return { + ok: false, + refusal: { + code: 'agent_session_operation_unknown', + message: 'The chat may have been created, but its tab could not be confirmed.' + } + } + } + } + return result } }), defineMethod({ name: 'agentSession.ensure', params: AttachParams, - handler: async (params, ctx) => { - await ensureHostInstalled(ctx) - return requireHost(ctx).attach(callerFor(ctx), params) - } + handler: async (params, ctx) => attachClientSuppliedLocation(params, ctx) }), defineMethod({ name: 'agentSession.send', @@ -165,6 +213,11 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ params: SetOptionParams, handler: async (params, ctx) => requireHost(ctx).setOption(callerFor(ctx), params) }), + defineMethod({ + name: 'agentSession.requestHandoff', + params: HandoffParams, + handler: async (params, ctx) => requireHost(ctx).requestHandoff(callerFor(ctx), params) + }), defineMethod({ name: 'agentSession.handoffStatus', params: HandoffStatusParams, diff --git a/src/main/runtime/rpc/methods/structured-session-tab-restore.ts b/src/main/runtime/rpc/methods/structured-session-tab-restore.ts index 4333713a445..f9efa46d16d 100644 --- a/src/main/runtime/rpc/methods/structured-session-tab-restore.ts +++ b/src/main/runtime/rpc/methods/structured-session-tab-restore.ts @@ -1,13 +1,24 @@ import type { RpcContext } from '../core' -import { supportsStructuredAgentSessions } from './structured-agent-session-policy' +import { + isStructuredNativeChatEnabled, + supportsStructuredAgentSessions +} from './structured-agent-session-policy' +/** Republishes structured tabs into the host's own snapshot map. + * + * Mobile is gated on the host setting alone, NOT on the client's capability: an old build is + * shown a fallback prompt in place of each chat, and gating on capability left it with nothing to + * project after a desktop restart — no chat and no prompt. The setting still gates it, because + * with structured chat off there is nothing for any mobile client to reach. Restoring spawns no + * provider child for a cleanly closed session. */ export async function restoreStructuredTabsIfSupported( context: Pick ): Promise { - if ( - supportsStructuredAgentSessions(context) && - typeof context.runtime.restoreStructuredAgentSessionTabs === 'function' - ) { + const shouldRestore = + context.clientKind === 'mobile' + ? isStructuredNativeChatEnabled(context.runtime) + : supportsStructuredAgentSessions(context) + if (shouldRestore && typeof context.runtime.restoreStructuredAgentSessionTabs === 'function') { await context.runtime.restoreStructuredAgentSessionTabs() } } diff --git a/src/main/runtime/rpc/mobile-clipboard-image-provenance.test.ts b/src/main/runtime/rpc/mobile-clipboard-image-provenance.test.ts new file mode 100644 index 00000000000..aa1709fcecb --- /dev/null +++ b/src/main/runtime/rpc/mobile-clipboard-image-provenance.test.ts @@ -0,0 +1,53 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { + hasMobileClipboardImagePath, + MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES, + MOBILE_CLIPBOARD_IMAGE_PROVENANCE_TTL_MS, + mobileClipboardImageProvenanceSizeForTest, + recordMobileClipboardImagePath, + resetMobileClipboardImageProvenanceForTest +} from './mobile-clipboard-image-provenance' + +describe('mobile clipboard image provenance', () => { + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(new Date('2026-01-01T00:00:00Z')) + resetMobileClipboardImageProvenanceForTest() + }) + + afterEach(() => { + resetMobileClipboardImageProvenanceForTest() + vi.useRealTimers() + }) + + it('expires records without consuming them on repeated checks', () => { + recordMobileClipboardImagePath('device-a', '/tmp/image.png') + + expect(hasMobileClipboardImagePath('device-a', '/tmp/image.png')).toBe(true) + expect(hasMobileClipboardImagePath('device-a', '/tmp/image.png')).toBe(true) + vi.advanceTimersByTime(MOBILE_CLIPBOARD_IMAGE_PROVENANCE_TTL_MS + 1) + expect(hasMobileClipboardImagePath('device-a', '/tmp/image.png')).toBe(false) + expect(mobileClipboardImageProvenanceSizeForTest()).toBe(0) + }) + + it('evicts the oldest record at the global bound and supports test cleanup', () => { + for (let index = 0; index <= MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES; index++) { + recordMobileClipboardImagePath(`device-${index}`, `/tmp/image-${index}.png`) + vi.advanceTimersByTime(1) + } + + expect(mobileClipboardImageProvenanceSizeForTest()).toBe( + MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES + ) + expect(hasMobileClipboardImagePath('device-0', '/tmp/image-0.png')).toBe(false) + expect( + hasMobileClipboardImagePath( + `device-${MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES}`, + `/tmp/image-${MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES}.png` + ) + ).toBe(true) + + resetMobileClipboardImageProvenanceForTest() + expect(mobileClipboardImageProvenanceSizeForTest()).toBe(0) + }) +}) diff --git a/src/main/runtime/rpc/mobile-clipboard-image-provenance.ts b/src/main/runtime/rpc/mobile-clipboard-image-provenance.ts new file mode 100644 index 00000000000..117455593c9 --- /dev/null +++ b/src/main/runtime/rpc/mobile-clipboard-image-provenance.ts @@ -0,0 +1,90 @@ +export const MOBILE_CLIPBOARD_IMAGE_PROVENANCE_TTL_MS = 60 * 60 * 1000 +export const MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES = 256 +const MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_PER_CLIENT = 64 + +const pathsByClient = new Map>() +let entryCount = 0 + +function deletePath(clientId: string, path: string): void { + const paths = pathsByClient.get(clientId) + if (!paths?.delete(path)) { + return + } + entryCount-- + if (paths.size === 0) { + pathsByClient.delete(clientId) + } +} + +function pruneExpired(now: number): void { + for (const [clientId, paths] of pathsByClient) { + for (const [path, expiresAt] of paths) { + if (expiresAt <= now) { + deletePath(clientId, path) + } + } + } +} + +function deleteOldestEntry(): void { + let oldest: { clientId: string; path: string; expiresAt: number } | null = null + for (const [clientId, paths] of pathsByClient) { + for (const [path, expiresAt] of paths) { + if (!oldest || expiresAt < oldest.expiresAt) { + oldest = { clientId, path, expiresAt } + } + } + } + if (oldest) { + deletePath(oldest.clientId, oldest.path) + } +} + +export function recordMobileClipboardImagePath(clientId: string | undefined, path: string): void { + const owner = clientId?.trim() + if (!owner) { + return + } + const now = Date.now() + pruneExpired(now) + let paths = pathsByClient.get(owner) + if (!paths) { + paths = new Map() + pathsByClient.set(owner, paths) + } + if (paths.delete(path)) { + entryCount-- + } + while (paths.size >= MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_PER_CLIENT) { + const oldestPath = paths.keys().next().value + if (typeof oldestPath !== 'string') { + break + } + deletePath(owner, oldestPath) + } + while (entryCount >= MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES) { + deleteOldestEntry() + } + paths = pathsByClient.get(owner) ?? new Map() + pathsByClient.set(owner, paths) + paths.set(path, now + MOBILE_CLIPBOARD_IMAGE_PROVENANCE_TTL_MS) + entryCount++ +} + +export function hasMobileClipboardImagePath(clientId: string | undefined, path: string): boolean { + const owner = clientId?.trim() + if (!owner) { + return false + } + pruneExpired(Date.now()) + return pathsByClient.get(owner)?.has(path) ?? false +} + +export function resetMobileClipboardImageProvenanceForTest(): void { + pathsByClient.clear() + entryCount = 0 +} + +export function mobileClipboardImageProvenanceSizeForTest(): number { + return entryCount +} diff --git a/src/main/runtime/rpc/mobile-e2ee-v2-client-capabilities.ts b/src/main/runtime/rpc/mobile-e2ee-v2-client-capabilities.ts new file mode 100644 index 00000000000..3834a417ddb --- /dev/null +++ b/src/main/runtime/rpc/mobile-e2ee-v2-client-capabilities.ts @@ -0,0 +1,21 @@ +import type { RuntimeCapability } from '../../../shared/protocol-version' +import { parseRemoteRuntimeJsonText } from '../../../shared/remote-runtime-request-frames' +import { parseRuntimeClientCapabilities } from './runtime-client-capabilities' + +export function parseMobileE2EEV2ClientCapabilities( + plaintext: string +): readonly RuntimeCapability[] | null { + try { + const message = parseRemoteRuntimeJsonText(plaintext) as Record + if ( + Object.keys(message).sort().join(',') !== 'clientCapabilities,type,v' || + message.type !== 'e2ee_client_capabilities' || + message.v !== 1 + ) { + return null + } + return parseRuntimeClientCapabilities(message.clientCapabilities) + } catch { + return null + } +} diff --git a/src/main/runtime/rpc/mobile-socket-wiring.test.ts b/src/main/runtime/rpc/mobile-socket-wiring.test.ts index 505cf63feb6..5ce3325083c 100644 --- a/src/main/runtime/rpc/mobile-socket-wiring.test.ts +++ b/src/main/runtime/rpc/mobile-socket-wiring.test.ts @@ -190,6 +190,58 @@ describe('MobileSocketWiring', () => { expect(wiring.connectionCount).toBe(0) }) + it('lets the capability RPC write capabilities back onto the socket', () => { + // `runtime.clientCapabilities.update` stores the advertised set by assigning + // `authenticatedSocket.clientCapabilities`. A read-only socket makes that a + // TypeError, the RPC answers `runtime_error`, and every structured + // agent-session tab is then projected away from a capable phone. + const desktop = generateKeyPair() + const phone = generateKeyPair() + const ws = new FakeSocket() + const transport = new FakeTransport() + const onText = vi.fn() + const wiring = new MobileSocketWiring({ + deviceRegistry: registryFor('device-1', 'valid-token', 'mobile'), + e2eeKeypair: { + publicKey: desktop.publicKey, + secretKey: desktop.secretKey, + publicKeyB64: Buffer.from(desktop.publicKey).toString('base64') + }, + onText, + onBinary: vi.fn(), + onClose: vi.fn() + }) + wiring.attachTransport(transport) + + transport.receive( + ws, + JSON.stringify({ + type: 'e2ee_hello', + publicKeyB64: Buffer.from(phone.publicKey).toString('base64') + }) + ) + const sharedKey = deriveSharedKey(phone.secretKey, desktop.publicKey) + transport.receive( + ws, + encrypt(JSON.stringify({ type: 'e2ee_auth', deviceToken: 'valid-token' }), sharedKey) + ) + transport.receive(ws, encrypt('{"id":"rpc-1","method":"status.get"}', sharedKey)) + + const socket = onText.mock.calls[0]?.[0] + expect(socket).toBeDefined() + expect(socket.clientCapabilities).toEqual([]) + + expect(() => { + socket.clientCapabilities = ['agent-session.structured.v1'] + }).not.toThrow() + expect(socket.clientCapabilities).toEqual(['agent-session.structured.v1']) + + // Later requests on the same connection must see the updated set, so the + // channel is the single source of truth rather than a detached copy. + transport.receive(ws, encrypt('{"id":"rpc-2","method":"status.get"}', sharedKey)) + expect(onText.mock.calls[1]?.[0].clientCapabilities).toEqual(['agent-session.structured.v1']) + }) + it('closes an unknown-token socket even when reporting the failure throws', () => { const desktop = generateKeyPair() const phone = generateKeyPair() @@ -348,4 +400,86 @@ describe('MobileSocketWiring', () => { expect(transport.setClientId).not.toHaveBeenCalled() expect(ws.close).toHaveBeenCalledWith(4001, 'Unauthorized') }) + + it('keeps post-auth v2 capability-shaped frames on the RPC path', () => { + const desktop = nacl.box.keyPair.fromSecretKey(new Uint8Array(32).fill(1)) + const phone = nacl.box.keyPair.fromSecretKey(new Uint8Array(32).fill(2)) + const ws = new FakeSocket() + const transport = new FakeTransport() + const onText = vi.fn() + const metadata: MobileSocketTransportMetadata = { + transport: 'relay', + relayHostId: 'AbCdEf0123_-xyZ9', + relayDeviceId: 'device-1', + basisConnId: 'connection-1', + credentialKind: 'resume' + } + const wiring = new MobileSocketWiring({ + deviceRegistry: registryFor('device-1', 'valid-token'), + e2eeKeypair: { + publicKey: desktop.publicKey, + secretKey: desktop.secretKey, + publicKeyB64: Buffer.from(desktop.publicKey).toString('base64') + }, + onText, + onBinary: vi.fn(), + onClose: vi.fn() + }) + wiring.attachTransport(transport, () => metadata) + const hello: MobileE2EEV2Hello = { + type: 'e2ee_hello', + v: 2, + clientPublicKeyB64: Buffer.from(phone.publicKey).toString('base64'), + clientNonceB64: Buffer.from(new Uint8Array(32).fill(3)).toString('base64'), + capabilities: { framing: [2], payloadKinds: ['text', 'binary'] }, + context: { + protocol: 'orca-mobile-e2ee', + initiator: 'mobile', + responder: 'desktop', + transport: 'relay', + relayHostId: metadata.relayHostId + } + } + transport.receive(ws, JSON.stringify(hello)) + const ready = JSON.parse(ws.sent[0]!.toString()) as MobileE2EEV2Ready + const handshake = validateMobileE2EEV2Handshake(hello, ready)! + const schedule = deriveMobileE2EEV2KeySchedule({ + sharedSecret: deriveSharedKey(phone.secretKey, desktop.publicKey), + transcript: encodeMobileE2EEV2Transcript(handshake), + clientNonce: handshake.clientNonce, + desktopNonce: handshake.desktopNonce + }) + const send = (value: unknown, counter: bigint): void => { + const frame = sealMobileE2EEV2Frame({ + payload: new TextEncoder().encode(JSON.stringify(value)), + key: schedule.mobileToDesktopKey, + sessionId: schedule.sessionId, + direction: 'mobile-to-desktop', + payloadKind: 'text', + counter + }) + transport.receive(ws, Buffer.from(frame).toString('base64')) + } + send( + { + type: 'e2ee_auth', + v: 2, + transcriptHashB64: Buffer.from(schedule.transcriptHash).toString('base64'), + deviceToken: 'valid-token' + }, + 0n + ) + const capabilityFrame = { + type: 'e2ee_client_capabilities', + v: 1, + clientCapabilities: ['agent-session.structured.v1'] + } + send(capabilityFrame, 1n) + send({ id: 'rpc-1', method: 'agentSession.history', params: {} }, 2n) + + expect(onText).toHaveBeenCalledTimes(2) + expect(onText.mock.calls[0]?.[0].clientCapabilities).toEqual([]) + expect(JSON.parse(onText.mock.calls[0]?.[1] ?? '')).toEqual(capabilityFrame) + expect(onText.mock.calls[1]?.[0].clientCapabilities).toEqual([]) + }) }) diff --git a/src/main/runtime/rpc/mobile-socket-wiring.ts b/src/main/runtime/rpc/mobile-socket-wiring.ts index 43004be4582..2d536b15038 100644 --- a/src/main/runtime/rpc/mobile-socket-wiring.ts +++ b/src/main/runtime/rpc/mobile-socket-wiring.ts @@ -165,7 +165,16 @@ export class MobileSocketWiring { ws, connectionId, device, - clientCapabilities: channel.clientCapabilities, + // Why: the channel owns the set for the whole connection, so this reads + // through rather than snapshotting. It must also WRITE through — the + // capability RPC updates the socket, and a getter-only property makes + // that a TypeError, which strands a capable phone with no capabilities. + get clientCapabilities() { + return channel.clientCapabilities + }, + set clientCapabilities(next: readonly RuntimeCapability[]) { + channel.clientCapabilities = next + }, transport: metadata } this.authenticatedSockets.set(ws, socket) diff --git a/src/main/runtime/structured-agent-session-integration-replay.test.ts b/src/main/runtime/structured-agent-session-integration-replay.test.ts index 4edec30499b..e5baa032341 100644 --- a/src/main/runtime/structured-agent-session-integration-replay.test.ts +++ b/src/main/runtime/structured-agent-session-integration-replay.test.ts @@ -259,6 +259,7 @@ beforeEach(async () => { claimKeyId: 'key-1', resolveWorkspacePath: async (workspaceId) => `/repos/${workspaceId}`, resolveCodexCommand: () => '/usr/local/bin/codex', + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveEnvironment: async () => { bootEnvironmentReads += 1 return { @@ -320,6 +321,7 @@ describe('a structured codex session over agentSession.*', () => { claimKeyId: 'key-1', resolveWorkspacePath: async (workspaceId) => `/repos/${workspaceId}`, resolveCodexCommand: () => '/usr/local/bin/codex', + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), openCodexConnection: codex.openConnection, readProcessStartTime: async () => 1_700_000_000_000 }) diff --git a/src/main/runtime/structured-agent-session-integration.test.ts b/src/main/runtime/structured-agent-session-integration.test.ts index aa5f819e1fb..2982a6530b2 100644 --- a/src/main/runtime/structured-agent-session-integration.test.ts +++ b/src/main/runtime/structured-agent-session-integration.test.ts @@ -307,6 +307,7 @@ beforeEach(async () => { claimKeyId: 'key-1', resolveWorkspacePath: async (workspaceId) => `/repos/${workspaceId}`, resolveCodexCommand: () => '/usr/local/bin/codex', + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveEnvironment: async () => { bootEnvironmentReads += 1 return { diff --git a/src/main/runtime/structured-agent-session-owner-probe.ts b/src/main/runtime/structured-agent-session-owner-probe.ts new file mode 100644 index 00000000000..47f92a08ca6 --- /dev/null +++ b/src/main/runtime/structured-agent-session-owner-probe.ts @@ -0,0 +1,108 @@ +import type { AgentSessionOwnerProbe } from '../../shared/agent-session-lease-adjudication' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { + probeAgentSessionProcessIdentities, + probeAgentSessionProcessIdentity, + probeAgentSessionReservation +} from './agent-session-process-identity-probe' +import { findAgentSessionSpawnTokenProcesses } from './agent-session-spawn-token-process-scan' +import { readEchoedAgentSessionSpawnToken } from './agent-session-spawn-token-readback' + +/** + * The lease's only source of truth about a previous owner. Everything it cannot + * answer PID-reuse-safely reports `indeterminate`. An exact owner stays fenced in `recovering`; + * an ownerless, unattributable reservation enters `manual-recovery`. + */ +export function createStructuredAgentSessionOwnerProbe( + hostId: string, + probe = probeAgentSessionProcessIdentity, + findSpawnTokenProcesses = findAgentSessionSpawnTokenProcesses +): (record: AgentSessionRecord) => Promise { + return async (record) => { + const owner = record.lease.ownerProcess + if (!owner) { + if (record.lease.processlessAt !== undefined && record.lease.processlessAt !== null) { + return { outcome: 'reservation-unused' } + } + const spawnToken = record.lease.reservedSpawnToken + if (spawnToken === null) { + if (record.lease.claimStatus === 'reserved') { + return { + outcome: 'indeterminate', + reason: 'reservation recorded no spawn token to scan for' + } + } + // The token is minted before the child and is the only thing a child could be carrying. + // No owner and no token means nothing on any host can be holding this lease — answering + // `indeterminate` here is what latches an already-free record into recovery forever. + return { outcome: 'reservation-unused' } + } + // Freeing a reservation needs positive proof that nothing spawned under its token. The scan + // answers null where the platform cannot read another process's environment. + return probeAgentSessionReservation({ + spawnToken, + findProcessesWithSpawnToken: (token) => findSpawnTokenProcesses(token), + hasProviderActivitySinceReservation: async () => + agentSessionReservationTouchedProvider(record) + }) + } + if (owner.hostId !== hostId) { + // Checking a remote host's pid against this machine's process table is + // exactly how a live owner gets declared dead. + return { + outcome: 'indeterminate', + reason: `owner runs on ${owner.hostId}, which this host cannot probe` + } + } + // The env read-back answers on hosts that expose it and null elsewhere, giving the + // probe a PID-reuse-safe element even when no start time was recorded. + return probe({ + identity: owner, + deps: { readEchoedSpawnToken: readEchoedAgentSessionSpawnToken } + }) + } +} + +export function createStructuredAgentSessionOwnerProbes( + hostId: string, + probeMany: typeof probeAgentSessionProcessIdentities = probeAgentSessionProcessIdentities, + probeOne = createStructuredAgentSessionOwnerProbe(hostId) +): (records: readonly AgentSessionRecord[]) => Promise> { + return async (records) => { + const results = new Map() + const localOwners: { + record: AgentSessionRecord + owner: NonNullable + }[] = [] + for (const record of records) { + const owner = record.lease.ownerProcess + if (owner?.hostId === hostId) { + localOwners.push({ record, owner }) + } else { + results.set(record.sessionId, await probeOne(record)) + } + } + const probes = await probeMany({ + identities: localOwners.map(({ owner }) => owner), + deps: { readEchoedSpawnToken: readEchoedAgentSessionSpawnToken } + }) + for (const [index, { record }] of localOwners.entries()) { + results.set( + record.sessionId, + probes[index] ?? { outcome: 'indeterminate', reason: 'owner probe returned no result' } + ) + } + return results + } +} + +/** + * The only provider-side trace a reservation can leave in its own record: a handle link minted at + * this fence. `proveAgentSessionOwner` refuses to append one before an identity is committed, so a + * link at the reservation's fence means a child got far enough to resume the provider thread. It + * cannot see activity the child produced without proving a handle, which is why it is paired with + * the token scan rather than trusted alone. + */ +function agentSessionReservationTouchedProvider(record: AgentSessionRecord): boolean { + return record.providerHandleChain.at(-1)?.mintedAtFence === record.lease.runtimeFence +} diff --git a/src/main/runtime/structured-agent-session-runtime-exit.test.ts b/src/main/runtime/structured-agent-session-runtime-exit.test.ts index a8419176357..5c6e43c2bc0 100644 --- a/src/main/runtime/structured-agent-session-runtime-exit.test.ts +++ b/src/main/runtime/structured-agent-session-runtime-exit.test.ts @@ -84,6 +84,7 @@ describe('structured session runtime provider-exit wiring', () => { hostId: 'local', claimKeyId: 'key-1', resolveWorkspacePath: async () => root!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveCodexCommand: () => 'codex', resolveEnvironment: async () => ({ PATH: process.env.PATH }), openCodexConnection: openConnection, @@ -180,6 +181,7 @@ describe('structured session runtime provider-exit wiring', () => { hostId: 'local', claimKeyId: 'key-1', resolveWorkspacePath: async () => root!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveCodexCommand: () => 'codex', resolveEnvironment: async () => ({ PATH: process.env.PATH }), openCodexConnection: openConnection, @@ -260,6 +262,7 @@ describe('structured session runtime provider-exit wiring', () => { hostId: 'local', claimKeyId: 'key-1', resolveWorkspacePath: async () => root!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveCodexCommand: () => 'codex', resolveEnvironment: async () => ({ PATH: process.env.PATH }), openCodexConnection: openConnection, diff --git a/src/main/runtime/structured-agent-session-runtime.test.ts b/src/main/runtime/structured-agent-session-runtime.test.ts index b03d17cdb3f..3b69a0a4be3 100644 --- a/src/main/runtime/structured-agent-session-runtime.test.ts +++ b/src/main/runtime/structured-agent-session-runtime.test.ts @@ -13,7 +13,9 @@ import type { } from '../../shared/agent-session-record' import { createStructuredAgentSessionOwnerProbe, - createStructuredAgentSessionOwnerProbes, + createStructuredAgentSessionOwnerProbes +} from './structured-agent-session-owner-probe' +import { ensureStructuredAgentSessionHost, hasPersistedStructuredAgentSessionStore, stopStructuredAgentSessionRuntime @@ -226,6 +228,7 @@ describe('structured agent-session runtime install', () => { hostId: HOST_ID, claimKeyId: 'key-1', resolveWorkspacePath: async () => stateDirectory!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveEnvironment: async () => ({}), reapOrphanChildren, onError @@ -252,6 +255,7 @@ describe('structured agent-session runtime install', () => { hostId: HOST_ID, claimKeyId: 'key-1', resolveWorkspacePath: async () => stateDirectory!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveEnvironment: async () => ({}), reapOrphanChildren: async () => { throw failure @@ -301,7 +305,8 @@ describe('a teardown that fails is retried by the next stop', () => { claimKeyId: 'key-1', resolveWorkspacePath: async () => directory!, resolveEnvironment: async () => ({}), - reapOrphanChildren: async () => [] + reapOrphanChildren: async () => [], + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }) }) const journalDir = join(directory, 'stubborn-journal') diff --git a/src/main/runtime/structured-agent-session-runtime.ts b/src/main/runtime/structured-agent-session-runtime.ts index 52a3bd81818..d670c16f47d 100644 --- a/src/main/runtime/structured-agent-session-runtime.ts +++ b/src/main/runtime/structured-agent-session-runtime.ts @@ -9,29 +9,33 @@ import { existsSync } from 'node:fs' import { join } from 'node:path' -import type { AgentSessionOwnerProbe } from '../../shared/agent-session-lease-adjudication' import type { AgentSessionRecord } from '../../shared/agent-session-record' import { createCodexStructuredLaunchResolver } from '../codex/codex-structured-launch-resolution' import { CodexStructuredSessionAdapter, type CodexStructuredSessionAdapterDeps } from '../codex/codex-structured-session-adapter' +import type { ClaudeStructuredSessionAdapterDeps } from '../claude/claude-structured-session-adapter' import { StructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-host' +import { StructuredAgentSessionAdapterRouter } from '../native-chat/agent-session-wire/structured-agent-session-adapter-router' import type { StructuredAgentSessionHandoffTransport } from '../native-chat/agent-session-wire/structured-agent-session-handoff-types' import { setStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' +import { + readClaudeManagedAccountGateSettings, + type ClaudeManagedAccountGateSettings +} from '../native-chat/claude-structured-managed-account-support' import { AgentSessionRecordStore } from './agent-session-record-store' import { agentSessionStorePath } from './agent-session-record-store-file' import { stopOrphanAgentSessionChildren } from './agent-session-orphan-child-reaper' import { - probeAgentSessionProcessIdentities, - probeAgentSessionProcessIdentity, - probeAgentSessionReservation -} from './agent-session-process-identity-probe' -import { findAgentSessionSpawnTokenProcesses } from './agent-session-spawn-token-process-scan' -import { readEchoedAgentSessionSpawnToken } from './agent-session-spawn-token-readback' + createStructuredAgentSessionOwnerProbe, + createStructuredAgentSessionOwnerProbes +} from './structured-agent-session-owner-probe' import { agentSessionPtyWriteGate } from './agent-session-pty-write-gate' import { resolveLoginShellEnvironment } from '../startup/login-shell-environment' import { recordAgentSessionProviderHandle } from './agent-session-provider-handle-transition' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { createStructuredClaudeRuntimeAdapter } from './structured-claude-runtime-adapter' /** Sibling of the journal tree rather than inside it: one file adjudicates every * session's lease, while a journal is per session. */ @@ -55,13 +59,20 @@ export type StructuredAgentSessionRuntimeDeps = { claimKeyId: string resolveWorkspacePath: (workspaceId: string) => Promise resolveCodexCommand?: (options?: { pathEnv?: string | null; homePath?: string }) => string + resolveClaudeCommand?: () => string /** Provider transports are overridden only to drive the runtime against scripted children. */ openCodexConnection?: CodexStructuredSessionAdapterDeps['openConnection'] + openClaudeConnection?: ClaudeStructuredSessionAdapterDeps['openConnection'] /** Scripted app-servers carry fake pids the real start-time read cannot answer for. */ readProcessStartTime?: CodexStructuredSessionAdapterDeps['readProcessStartTime'] resolveLaunchArgs?: (provider: AgentSessionRecord['provider']) => Promise | string[] resolveLaunchEnv?: () => Promise resolveLaunchEnvOverlay?: () => Promise> | Record + resolveClaudeLaunchEnv?: () => Promise> | Record + /** Required, and asserted at install time — an absent policy must not degrade to a guess. */ + resolveClaudeAuthPolicy: () => Promise | ClaudeStructuredAuthPolicy + /** Raw settings getter; the reader that fails closed around it is built here, in checked code. */ + getClaudeManagedAccountGateSettings?: () => ClaudeManagedAccountGateSettings resolveEnvironment?: () => Promise resolveCodexOverrides?: () => NodeJS.ProcessEnv onError?: (input: { scope: string; error: unknown }) => void @@ -71,13 +82,18 @@ export type StructuredAgentSessionRuntimeDeps = { type InstalledRuntime = { host: StructuredAgentSessionHost - adapter: CodexStructuredSessionAdapter - /** Resolves after every adapter-exit recovery callback has settled. */ + adapter: { closeAll(): Promise } + /** Resolves after every observed adapter exit has published, and every + * recovery callback it raised has settled. */ waitForRecovery: () => Promise } let installing: Promise | null = null +/** Thrown when the host is installed without a Claude auth policy resolver. */ +export const CLAUDE_STRUCTURED_AUTH_POLICY_REQUIRED = + 'structured agent-session host requires a Claude auth policy resolver' + /** * Runtimes whose teardown did not finish. `installing` is cleared regardless so * nothing new attaches, but dropping the runtime as well would strand every @@ -98,6 +114,17 @@ export function ensureStructuredAgentSessionHost( return installing.then((installed) => installed.host) } +/** Resolves once every provider exit observed so far has been published by its + * adapter and reconciled by the host. Nothing is installed, nothing to wait on. + * + * This is the only handle onto that barrier: reconciliation is driven by exit + * callbacks, so a caller that needs the settled lease — rather than the one the + * exit is still being reconciled out of — has no other way to know it landed. */ +export async function waitForStructuredAgentSessionRecovery(): Promise { + const installed = await installing?.catch(() => null) + await installed?.waitForRecovery() +} + /** Drops the host and reaps every Codex child under it. Runtime teardown and * test isolation take the same path, so neither can leave a live app-server. * @@ -147,8 +174,14 @@ async function tearDownRuntime(installed: InstalledRuntime): Promise { } async function install(deps: StructuredAgentSessionRuntimeDeps): Promise { + // Why thrown rather than defaulted: the caller is `@ts-nocheck`, so a dropped + // field arrives here as `undefined`. Refusing to install is loud; guessing a + // policy is the silent under-strip this assertion exists to prevent. + if (typeof deps.resolveClaudeAuthPolicy !== 'function') { + throw new Error(CLAUDE_STRUCTURED_AUTH_POLICY_REQUIRED) + } const bootEnvironment = (deps.resolveEnvironment ?? resolveLoginShellEnvironment)() - const resolveEnvironment = async (): Promise => ({ + const resolveCodexEnvironment = async (): Promise => ({ ...(await bootEnvironment), ...(await deps.resolveLaunchEnv?.()), ...(await deps.resolveLaunchEnvOverlay?.()), @@ -181,7 +214,7 @@ async function install(deps: StructuredAgentSessionRuntimeDeps): Promise + readClaudeManagedAccountGateSettings(deps.getClaudeManagedAccountGateSettings!) + } + : {}), + onUnexpectedExit: (event) => { + recoveryChain = recoveryChain.then(async () => { + try { + await host?.handleAdapterEvent(event) + } catch (error) { + deps.onError?.({ scope: `structured-agent-session-exit:${event.sessionId}`, error }) + } + }) + }, + onBackgroundTasksChanged: (sessionId, state) => + host?.publishBackgroundTaskState(sessionId, state), + ...(deps.openClaudeConnection ? { openClaudeConnection: deps.openClaudeConnection } : {}), + ...(deps.readProcessStartTime ? { readProcessStartTime: deps.readProcessStartTime } : {}) + }) + const adapter = new StructuredAgentSessionAdapterRouter({ codex, claude }, async () => { + await Promise.all([codex.closeAll(), claude.closeAll()]) + }) host = new StructuredAgentSessionHost({ store, adapter, @@ -233,6 +296,10 @@ async function install(deps: StructuredAgentSessionRuntimeDeps): Promise Promise { - return async (record) => { - const owner = record.lease.ownerProcess - if (!owner) { - if (record.lease.processlessAt !== undefined && record.lease.processlessAt !== null) { - return { outcome: 'reservation-unused' } - } - const spawnToken = record.lease.reservedSpawnToken - if (spawnToken === null) { - if (record.lease.claimStatus === 'reserved') { - return { - outcome: 'indeterminate', - reason: 'reservation recorded no spawn token to scan for' - } - } - // The token is minted before the child and is the only thing a child could be carrying. - // No owner and no token means nothing on any host can be holding this lease — answering - // `indeterminate` here is what latches an already-free record into recovery forever. - return { outcome: 'reservation-unused' } - } - // Freeing a reservation needs positive proof that nothing spawned under its token. The scan - // answers null where the platform cannot read another process's environment. - return probeAgentSessionReservation({ - spawnToken, - findProcessesWithSpawnToken: (token) => findSpawnTokenProcesses(token), - hasProviderActivitySinceReservation: async () => - agentSessionReservationTouchedProvider(record) - }) - } - if (owner.hostId !== hostId) { - // Checking a remote host's pid against this machine's process table is - // exactly how a live owner gets declared dead. - return { - outcome: 'indeterminate', - reason: `owner runs on ${owner.hostId}, which this host cannot probe` - } - } - // The env read-back answers on hosts that expose it and null elsewhere, giving the - // probe a PID-reuse-safe element even when no start time was recorded. - return probe({ - identity: owner, - deps: { readEchoedSpawnToken: readEchoedAgentSessionSpawnToken } - }) - } -} - -export function createStructuredAgentSessionOwnerProbes( - hostId: string, - probeMany: typeof probeAgentSessionProcessIdentities = probeAgentSessionProcessIdentities, - probeOne = createStructuredAgentSessionOwnerProbe(hostId) -): (records: readonly AgentSessionRecord[]) => Promise> { - return async (records) => { - const results = new Map() - const localOwners: { - record: AgentSessionRecord - owner: NonNullable - }[] = [] - for (const record of records) { - const owner = record.lease.ownerProcess - if (owner?.hostId === hostId) { - localOwners.push({ record, owner }) - } else { - results.set(record.sessionId, await probeOne(record)) - } - } - const probes = await probeMany({ - identities: localOwners.map(({ owner }) => owner), - deps: { readEchoedSpawnToken: readEchoedAgentSessionSpawnToken } - }) - for (const [index, { record }] of localOwners.entries()) { - results.set( - record.sessionId, - probes[index] ?? { outcome: 'indeterminate', reason: 'owner probe returned no result' } - ) - } - return results - } -} - -/** - * The only provider-side trace a reservation can leave in its own record: a handle link minted at - * this fence. `proveAgentSessionOwner` refuses to append one before an identity is committed, so a - * link at the reservation's fence means a child got far enough to resume the provider thread. It - * cannot see activity the child produced without proving a handle, which is why it is paired with - * the token scan rather than trusted alone. - */ -function agentSessionReservationTouchedProvider(record: AgentSessionRecord): boolean { - return record.providerHandleChain.at(-1)?.mintedAtFence === record.lease.runtimeFence -} diff --git a/src/main/runtime/structured-agent-session-support-probe.test.ts b/src/main/runtime/structured-agent-session-support-probe.test.ts new file mode 100644 index 00000000000..e393e41f3a4 --- /dev/null +++ b/src/main/runtime/structured-agent-session-support-probe.test.ts @@ -0,0 +1,174 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { OrcaRuntimeService } from './orca-runtime' +import { + getStructuredAgentSessionHost, + setStructuredAgentSessionHost +} from '../native-chat/agent-session-wire/structured-agent-session-registry' +import { agentSessionPtyWriteGate } from './agent-session-pty-write-gate' + +type InstallEffects = { + storeOpened: boolean + writeGateAttached: boolean + reaperStarted: boolean +} + +/** Stands in for `install()` by performing the three effects it performs, so a probe that + * reinstalls the host is caught by what the install *does*, not by a call count alone. */ +function stubStructuredHostInstall(runtime: OrcaRuntimeService): { + effects: InstallEffects + ensure: ReturnType +} { + const effects: InstallEffects = { + storeOpened: false, + writeGateAttached: false, + reaperStarted: false + } + // `supportsCreate` answers as the real Codex adapter would, so a probe that reinstalls the host + // still returns the right answer and fails on the install effects alone. + const host = { + reconcileRestartLeases: vi.fn(async () => {}), + supportsCreate: (location: { executionHostId: string; wslDistro: string | null }) => + location.executionHostId === 'local' && location.wslDistro === null + } + const ensure = vi.fn(async () => { + effects.storeOpened = true + effects.reaperStarted = true + agentSessionPtyWriteGate.attachRecordLookup(() => null) + effects.writeGateAttached = true + setStructuredAgentSessionHost(host as never) + }) + vi.spyOn(runtime, 'ensureStructuredAgentSessionHost').mockImplementation(ensure) + return { effects, ensure } +} + +type TestLocation = { + executionHostId: string + wslDistro: string | null + workspaceKind?: 'folder' | 'git-worktree' +} + +type SupportResult = { + supported: boolean + reason?: 'agent' | 'remote' | 'wsl' +} + +function createRuntime(location: TestLocation): OrcaRuntimeService { + const runtime = new OrcaRuntimeService({ getSettings: () => ({}) } as never) + const internal = runtime as unknown as { + resolveStructuredAgentSessionLocation: () => Promise + } + internal.resolveStructuredAgentSessionLocation = vi.fn(async () => ({ + executionHostId: location.executionHostId, + wslDistro: location.wslDistro, + workspaceId: 'workspace-1', + workspaceKind: location.workspaceKind ?? 'git-worktree' + })) + return runtime +} + +async function expectSupportWithoutInstall(input: { + agent: 'claude' | 'codex' + location: TestLocation + expected: SupportResult + repetitions?: number +}): Promise { + const runtime = createRuntime(input.location) + const { effects, ensure } = stubStructuredHostInstall(runtime) + + const answers: SupportResult[] = [] + for (let index = 0; index < (input.repetitions ?? 1); index += 1) { + answers.push( + await runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', input.agent) + ) + } + + expect(answers).toEqual(Array(input.repetitions ?? 1).fill(input.expected)) + expect(ensure).not.toHaveBeenCalled() + expect(effects).toEqual({ + storeOpened: false, + writeGateAttached: false, + reaperStarted: false + }) + expect(getStructuredAgentSessionHost()).toBeNull() +} + +describe('structured agent-session create-support probe', () => { + afterEach(() => { + setStructuredAgentSessionHost(null) + agentSessionPtyWriteGate.detachRecordLookup() + vi.restoreAllMocks() + }) + + it.each(['codex', 'claude'] as const)( + 'answers %s support repeatedly without installing the host', + async (agent) => { + await expectSupportWithoutInstall({ + agent, + location: { executionHostId: 'local', wslDistro: null }, + expected: { supported: true }, + repetitions: 3 + }) + } + ) + + it.each(['codex', 'claude'] as const)( + 'still reports an unsupported remote %s location without installing the host', + async (agent) => { + await expectSupportWithoutInstall({ + agent, + location: { executionHostId: 'ssh-host-1', wslDistro: null }, + expected: { supported: false, reason: 'remote' } + }) + } + ) + + it.each(['codex', 'claude'] as const)( + 'still reports an unsupported WSL %s location without installing the host', + async (agent) => { + await expectSupportWithoutInstall({ + agent, + location: { executionHostId: 'local', wslDistro: 'Ubuntu' }, + expected: { supported: false, reason: 'wsl' } + }) + } + ) + + it.each(['codex', 'claude'] as const)( + 'supports a local folder workspace for %s without installing the host', + async (agent) => { + await expectSupportWithoutInstall({ + agent, + location: { + executionHostId: 'local', + wslDistro: null, + workspaceKind: 'folder' + }, + expected: { supported: true } + }) + } + ) + + it('still installs and reconciles on startup when a store is already persisted', async () => { + const runtime = createRuntime({ executionHostId: 'local', wslDistro: null }) + const { effects, ensure } = stubStructuredHostInstall(runtime) + const internal = runtime as unknown as { + hasPersistedStructuredAgentSessionStore: () => boolean + refreshMobileSessionPtyRecords: () => Promise + } + internal.hasPersistedStructuredAgentSessionStore = () => true + internal.refreshMobileSessionPtyRecords = vi.fn(async () => {}) + + await runtime.prepareStructuredAgentSessionStartupRestoration() + + expect(ensure).toHaveBeenCalledTimes(1) + expect(effects).toEqual({ + storeOpened: true, + writeGateAttached: true, + reaperStarted: true + }) + expect( + (getStructuredAgentSessionHost() as unknown as { reconcileRestartLeases: () => void }) + .reconcileRestartLeases + ).toHaveBeenCalledTimes(1) + }) +}) diff --git a/src/main/runtime/structured-claude-auth-policy-wiring.test.ts b/src/main/runtime/structured-claude-auth-policy-wiring.test.ts new file mode 100644 index 00000000000..f018dfd30da --- /dev/null +++ b/src/main/runtime/structured-claude-auth-policy-wiring.test.ts @@ -0,0 +1,59 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { afterEach, describe, expect, it } from 'vitest' +import { + CLAUDE_STRUCTURED_AUTH_POLICY_REQUIRED, + ensureStructuredAgentSessionHost, + stopStructuredAgentSessionRuntime +} from './structured-agent-session-runtime' + +/** + * The structured host's Claude auth policy has exactly one production wiring, and it + * lives in `orca-runtime-get-worktree-ps.ts` — a `@ts-nocheck` file, so neither the + * compiler nor a type test can see the field disappear. Deleting that wiring used to + * leave ~1000 tests green while every `ANTHROPIC_*` variable in the shell reached the + * child, because `stripAuthEnv` silently fell back to `false`. + * + * Two independent guards replace that silence, and this file pins both. + */ +describe('structured Claude auth policy wiring', () => { + // The behavioural version of this assertion — importing the runtime class and + // capturing the installed deps — costs 35s of module transform for the whole + // OrcaRuntime chain (measured), so the wiring itself is pinned by source and the + // policy's meaning by claude-structured-auth-policy.test.ts. + it('passes a settings-derived Claude auth policy to the host installer', () => { + const source = readFileSync(join(__dirname, 'orca-runtime-get-worktree-ps.ts'), 'utf8') + + expect(source).toContain('claudeStructuredAuthPolicyForSettings') + expect(source).toMatch( + /resolveClaudeAuthPolicy:\s*\(\)\s*=>\s*\n?\s*claudeStructuredAuthPolicyForSettings\(/ + ) + }) + + describe('installing without one', () => { + let stateDirectory: string | null = null + + afterEach(async () => { + await stopStructuredAgentSessionRuntime() + if (stateDirectory) { + await rm(stateDirectory, { recursive: true, force: true }) + stateDirectory = null + } + }) + + it('refuses loudly rather than defaulting to a guess', async () => { + stateDirectory = await mkdtemp(join(tmpdir(), 'orca-auth-policy-wiring-')) + + await expect( + ensureStructuredAgentSessionHost({ + stateDirectory, + hostId: 'local', + claimKeyId: 'key-1', + resolveWorkspacePath: async () => stateDirectory as string + } as unknown as Parameters[0]) + ).rejects.toThrow(CLAUDE_STRUCTURED_AUTH_POLICY_REQUIRED) + }) + }) +}) diff --git a/src/main/runtime/structured-claude-runtime-adapter.ts b/src/main/runtime/structured-claude-runtime-adapter.ts new file mode 100644 index 00000000000..95151f1dae9 --- /dev/null +++ b/src/main/runtime/structured-claude-runtime-adapter.ts @@ -0,0 +1,106 @@ +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire' +import { join } from 'node:path' +import { resolveClaudeCommand } from '../codex-cli/command' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { createClaudeStructuredLaunchResolver } from '../claude/claude-structured-launch-resolution' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionAdapterDeps +} from '../claude/claude-structured-session-adapter' +import { claudeProviderHandleLink } from '../claude/claude-structured-owner-identity' +import type { StructuredAgentSessionLifecycleEvent } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { + readClaudeTranscriptLeafUuid, + resolveSessionFilePath +} from '../native-chat/session-file-resolver' +import { recordAgentSessionProviderHandle } from './agent-session-provider-handle-transition' +import type { ClaudeManagedAccountGateSettings } from '../native-chat/claude-structured-managed-account-support' +import type { AgentSessionRecordStore } from './agent-session-record-store' + +export type StructuredClaudeRuntimeAdapterDeps = { + store: AgentSessionRecordStore + resolveWorkspacePath: (workspaceId: string) => Promise + resolveClaudeCommand?: () => string + resolveClaudeLaunchEnv?: () => Promise> | Record + /** Managed-account auth state for a Claude launch, mirroring the terminal preflight. + * Required: an absent policy is what silently under-strips. */ + resolveClaudeAuthPolicy: () => Promise | ClaudeStructuredAuthPolicy + readClaudeManagedAccountGate?: () => ClaudeManagedAccountGateSettings | null + openClaudeConnection?: ClaudeStructuredSessionAdapterDeps['openConnection'] + readProcessStartTime?: ClaudeStructuredSessionAdapterDeps['readProcessStartTime'] + onUnexpectedExit: (event: StructuredAgentSessionLifecycleEvent) => void + onBackgroundTasksChanged?: ( + sessionId: string, + state: AgentSessionBackgroundTaskState | null + ) => void +} + +export function createStructuredClaudeRuntimeAdapter( + deps: StructuredClaudeRuntimeAdapterDeps +): ClaudeStructuredSessionAdapter { + const { store } = deps + return new ClaudeStructuredSessionAdapter({ + resolveLaunch: createClaudeStructuredLaunchResolver({ + store, + resolveWorkspacePath: deps.resolveWorkspacePath, + resolveCommand: deps.resolveClaudeCommand ?? resolveClaudeCommand, + ...(deps.resolveClaudeLaunchEnv ? { resolveEnv: deps.resolveClaudeLaunchEnv } : {}), + resolveAuthPolicy: deps.resolveClaudeAuthPolicy, + ...(deps.readClaudeManagedAccountGate + ? { readManagedAccountGate: deps.readClaudeManagedAccountGate } + : {}) + }), + persistHandle: async ({ sessionId, providerSessionId, leafUuid, fence }) => { + const currentFence = store.getRecord(sessionId)?.lease.runtimeFence ?? fence + const observedAt = Date.now() + await store.transitionHandoff(sessionId, (record: AgentSessionRecord) => + recordAgentSessionProviderHandle({ + record, + fence: currentFence, + link: claudeProviderHandleLink({ + sessionId: providerSessionId, + leafUuid, + resumed: true, + fence: currentFence, + observedAt + }), + now: observedAt + }) + ) + }, + readTranscriptLeaf: async ({ providerSessionId, previousLeafUuid, claudeConfigDir }) => { + const transcriptPath = await resolveSessionFilePath('claude', providerSessionId, { + claudeProjectsDir: join(claudeConfigDir, 'projects') + }) + return transcriptPath + ? await readClaudeTranscriptLeafUuid(transcriptPath, providerSessionId, previousLeafUuid) + : null + }, + onEvent: (event) => { + if ( + event.type === 'ended' && + event.cause === 'unexpected-exit' && + event.fence !== undefined && + event.acquisitionGeneration + ) { + deps.onUnexpectedExit({ + type: 'ended', + sessionId: event.sessionId, + reason: event.reason, + cause: event.cause, + fence: event.fence, + acquisitionGeneration: event.acquisitionGeneration, + ...(event.settlementRetryRequired + ? { settlementRetryRequired: event.settlementRetryRequired } + : {}) + }) + } + }, + ...(deps.onBackgroundTasksChanged + ? { onBackgroundTasksChanged: deps.onBackgroundTasksChanged } + : {}), + ...(deps.openClaudeConnection ? { openConnection: deps.openClaudeConnection } : {}), + ...(deps.readProcessStartTime ? { readProcessStartTime: deps.readProcessStartTime } : {}) + }) +} diff --git a/src/main/runtime/structured-tui-process-identity.test.ts b/src/main/runtime/structured-tui-process-identity.test.ts index 4324f8c5a26..fb5919824e5 100644 --- a/src/main/runtime/structured-tui-process-identity.test.ts +++ b/src/main/runtime/structured-tui-process-identity.test.ts @@ -238,6 +238,41 @@ describe('structured TUI process identity', () => { } }) + it('does not call a child absent after a single look that outlasted the budget', async () => { + // Measured on a 2,085-process host under load: one whole-machine `ps` took 6.2s while the + // shell-delivered child landed at ~3.5s. `ps` reads the table when it STARTS, so that one + // capture reported a t=0 machine and returned with the 5s budget already spent -- the loop + // answered "no exact child" without ever looking again. + let clockMs = 0 + let captures = 0 + await expect( + readStructuredTuiProcessIdentity({ + hostId: 'local', + rootPid: 100, + spawnToken: 'spawn-slow-ps', + agent: 'claude', + platform: 'darwin', + readPosixRows: async () => { + captures += 1 + const observedAtMs = clockMs + clockMs += 6_200 + return [ + { pid: 100, ppid: 1, stat: 'Ss', command: '/bin/zsh' }, + ...(observedAtMs >= 3_500 + ? [{ pid: 101, ppid: 100, stat: 'S+', command: 'claude --resume session-1' }] + : []) + ] + }, + readStartTime: async () => 1_700_000_000_000, + now: () => clockMs, + sleep: async (delayMs) => { + clockMs += delayMs + } + }) + ).resolves.toMatchObject({ pid: 101, spawnToken: 'spawn-slow-ps' }) + expect(captures).toBe(2) + }) + it('fails closed when the process snapshot omitted the PTY root', async () => { await expect( readStructuredTuiProcessIdentity({ diff --git a/src/main/runtime/structured-tui-process-identity.ts b/src/main/runtime/structured-tui-process-identity.ts index 0014351d781..f5ee019882b 100644 --- a/src/main/runtime/structured-tui-process-identity.ts +++ b/src/main/runtime/structured-tui-process-identity.ts @@ -20,6 +20,12 @@ const STRUCTURED_TUI_PROCESS_POLL_MS = 50 // window the added latency is bounded by one interval. const STRUCTURED_TUI_PROCESS_FAST_POLL_WINDOW_MS = 1_000 const STRUCTURED_TUI_PROCESS_MAX_POLL_MS = 500 +// Why a floor and not just the deadline: the first capture races the spawn it is looking for, +// so a null from it is absence of the child's arrival, not evidence the child is missing. The +// budget above assumes a look is nearly free, but one whole-machine `ps` measured 6.2s on a +// 2,085-process host under load -- long enough to spend the entire budget before the child +// (observed landing at ~3.5s) could exist, and answer "no exact child" after a single look. +const STRUCTURED_TUI_PROCESS_MIN_CAPTURES = 2 function descendants(rows: ProcessRow[], rootPid: number): (ProcessRow & { depth: number })[] { const children = new Map() @@ -175,6 +181,7 @@ export async function readStructuredTuiProcessIdentity(input: { const startedAtMs = now() const deadline = startedAtMs + (input.timeoutMs ?? STRUCTURED_TUI_PROCESS_WAIT_MS) let pollDelayMs = input.pollIntervalMs ?? STRUCTURED_TUI_PROCESS_POLL_MS + let captures = 0 while (true) { const rows: ProcessRow[] = @@ -186,6 +193,7 @@ export async function readStructuredTuiProcessIdentity(input: { foreground: false })) : posixRows(await (input.readPosixRows ?? getFreshProcessTableSnapshot)()) + captures += 1 let rootPresent = false for (const row of rows) { if (row.pid === input.rootPid) { @@ -218,11 +226,11 @@ export async function readStructuredTuiProcessIdentity(input: { } } const remainingMs = deadline - now() - if (remainingMs <= 0) { + if (remainingMs <= 0 && captures >= STRUCTURED_TUI_PROCESS_MIN_CAPTURES) { const label = input.agent === 'codex' ? 'Codex' : 'Claude' throw new Error(`The resumed terminal did not expose one exact ${label} child process.`) } - await sleep(Math.min(pollDelayMs, remainingMs)) + await sleep(Math.max(0, Math.min(pollDelayMs, remainingMs))) if (now() - startedAtMs >= STRUCTURED_TUI_PROCESS_FAST_POLL_WINDOW_MS) { // Never below the caller's interval, so an explicitly slow poll stays slow. pollDelayMs = Math.max( diff --git a/src/main/shell-prompt-readiness-probe.test.ts b/src/main/shell-prompt-readiness-probe.test.ts index c2c783953ac..56d4b35a06c 100644 --- a/src/main/shell-prompt-readiness-probe.test.ts +++ b/src/main/shell-prompt-readiness-probe.test.ts @@ -3,12 +3,16 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const lineEditorProbe = vi.hoisted(() => vi.fn()) const processReadinessProbe = vi.hoisted(() => vi.fn()) const resolveExecutablePath = vi.hoisted(() => vi.fn((value: string) => Promise.resolve(value))) +const resolveInstalledExecutablePaths = vi.hoisted(() => + vi.fn((): Promise => Promise.resolve([])) +) vi.mock('../shared/pty-slave-line-discipline-echo', () => ({ createPtySlaveLineEditorProbe: () => lineEditorProbe })) vi.mock('../shared/shell-process-readiness', () => ({ readShellProcessReadiness: processReadinessProbe, - resolveShellExecutablePath: resolveExecutablePath + resolveShellExecutablePath: resolveExecutablePath, + resolveInstalledShellExecutablePaths: resolveInstalledExecutablePaths })) import { createShellPromptReadinessProbe } from './shell-prompt-readiness-probe' @@ -19,6 +23,8 @@ describe('shell prompt readiness probe', () => { lineEditorProbe.mockReset() processReadinessProbe.mockReset() resolveExecutablePath.mockClear() + resolveInstalledExecutablePaths.mockClear() + resolveInstalledExecutablePaths.mockResolvedValue([]) }) afterEach(() => { @@ -95,6 +101,99 @@ describe('shell prompt readiness probe', () => { } }) + it('accepts a second installation of the same shell that the pane PATH resolves', async () => { + lineEditorProbe.mockResolvedValue('line-editor') + processReadinessProbe.mockResolvedValue({ + executablePath: '/opt/homebrew/bin/bash', + foreground: true + }) + resolveInstalledExecutablePaths.mockResolvedValue(['/bin/bash', '/opt/homebrew/bin/bash']) + const onPromptReady = vi.fn() + const probe = createShellPromptReadinessProbe({ + slavePath: '/dev/ttys048', + shellPath: '/bin/bash', + shellCwd: '/work', + shellPathEnv: '/opt/homebrew/bin:/usr/bin:/bin', + getShellPid: () => 42, + onPromptReady, + settleMs: 10 + }) + + probe?.notifyOutput('\x1b[?2004h') + await vi.advanceTimersByTimeAsync(10) + + expect(resolveInstalledExecutablePaths).toHaveBeenCalledWith( + 'bash', + '/work', + '/opt/homebrew/bin:/usr/bin:/bin' + ) + expect(onPromptReady).toHaveBeenCalledOnce() + }) + + it('rejects a replacement with the shell basename that the pane PATH cannot reach', async () => { + lineEditorProbe.mockResolvedValue('line-editor') + processReadinessProbe.mockResolvedValue({ executablePath: '/tmp/bash', foreground: true }) + resolveInstalledExecutablePaths.mockResolvedValue(['/bin/bash', '/opt/homebrew/bin/bash']) + const onPromptReady = vi.fn() + const probe = createShellPromptReadinessProbe({ + slavePath: '/dev/ttys048', + shellPath: '/bin/bash', + getShellPid: () => 42, + onPromptReady, + settleMs: 10 + }) + + probe?.notifyOutput('\x1b[?2004h') + await vi.advanceTimersByTimeAsync(10) + + expect(onPromptReady).not.toHaveBeenCalled() + }) + + it('does not widen identity when the launched shell path resolves exactly', async () => { + lineEditorProbe.mockResolvedValue('line-editor') + processReadinessProbe.mockResolvedValue({ executablePath: '/bin/zsh', foreground: true }) + const probe = createShellPromptReadinessProbe({ + slavePath: '/dev/ttys048', + shellPath: '/bin/zsh', + getShellPid: () => 42, + onPromptReady: vi.fn(), + settleMs: 10 + }) + + probe?.notifyOutput('\x1b[?2004h') + await vi.advanceTimersByTimeAsync(10) + + expect(resolveInstalledExecutablePaths).not.toHaveBeenCalled() + }) + + it('invalidates an alternate-installation result that resolves after disposal', async () => { + const pending: { resolve?: (value: string[]) => void } = {} + lineEditorProbe.mockResolvedValue('line-editor') + processReadinessProbe.mockResolvedValue({ + executablePath: '/opt/homebrew/bin/bash', + foreground: true + }) + resolveInstalledExecutablePaths.mockImplementation( + () => new Promise((resolve) => (pending.resolve = resolve)) + ) + const onPromptReady = vi.fn() + const probe = createShellPromptReadinessProbe({ + slavePath: '/dev/ttys048', + shellPath: '/bin/bash', + getShellPid: () => 42, + onPromptReady, + settleMs: 10 + }) + + probe?.notifyOutput('\x1b[?2004h') + await vi.advanceTimersByTimeAsync(10) + probe?.dispose() + pending.resolve?.(['/opt/homebrew/bin/bash']) + await vi.advanceTimersByTimeAsync(0) + + expect(onPromptReady).not.toHaveBeenCalled() + }) + it('does no external work when the ready marker cancels the settle window', async () => { const probe = createShellPromptReadinessProbe({ slavePath: '/dev/ttys048', diff --git a/src/main/shell-prompt-readiness-probe.ts b/src/main/shell-prompt-readiness-probe.ts index 11583fd2de7..3309ac91108 100644 --- a/src/main/shell-prompt-readiness-probe.ts +++ b/src/main/shell-prompt-readiness-probe.ts @@ -1,6 +1,7 @@ import { createPtySlaveLineEditorProbe } from '../shared/pty-slave-line-discipline-echo' import { readShellProcessReadiness, + resolveInstalledShellExecutablePaths, resolveShellExecutablePath } from '../shared/shell-process-readiness' import { @@ -32,6 +33,7 @@ export function createShellPromptReadinessProbe(options: { } const settleMs = options.settleMs ?? SHELL_PROMPT_PROBE_SETTLE_MS const expectedShellName = options.shellPath ? basename(options.shellPath).toLowerCase() : null + const shellCwd = options.shellCwd ?? process.cwd() const outputScanState = createLineEditorReadyOutputScanState() let disposed = false let timer: ReturnType | null = null @@ -52,11 +54,7 @@ export function createShellPromptReadinessProbe(options: { const [shell, expectedPath] = await Promise.all([ readShellProcessReadiness(shellPid), options.shellPath - ? resolveShellExecutablePath( - options.shellPath, - options.shellCwd ?? process.cwd(), - options.shellPathEnv - ) + ? resolveShellExecutablePath(options.shellPath, shellCwd, options.shellPathEnv) : Promise.resolve(null) ]) if (disposed || scheduledGeneration !== generation) { @@ -66,11 +64,28 @@ export function createShellPromptReadinessProbe(options: { !shell?.foreground || !expectedShellName || !expectedPath || - basename(shell.executablePath).toLowerCase() !== expectedShellName || - shell.executablePath !== expectedPath + basename(shell.executablePath).toLowerCase() !== expectedShellName ) { return } + if (shell.executablePath !== expectedPath) { + // Why widen past the launched path: a startup profile that `exec`s a second + // install of the same shell (Homebrew Bash over /bin/bash) keeps the pid but + // loses the wrapper's marker. Only installs this pane's own PATH resolves + // count, so a binary merely *named* bash/zsh outside it stays rejected. + const installedPaths = await resolveInstalledShellExecutablePaths( + expectedShellName, + shellCwd, + options.shellPathEnv + ) + if ( + disposed || + scheduledGeneration !== generation || + !installedPaths.includes(shell.executablePath) + ) { + return + } + } disposed = true options.onPromptReady() } diff --git a/src/main/shell-wrapper-generated-file-snapshot.test.ts b/src/main/shell-wrapper-generated-file-snapshot.test.ts index ddbf1537984..36cd837e4fd 100644 --- a/src/main/shell-wrapper-generated-file-snapshot.test.ts +++ b/src/main/shell-wrapper-generated-file-snapshot.test.ts @@ -73,6 +73,7 @@ const CONTRACT_GLOBALS = new Set([ 'OPENCODE_CONFIG_DIR', 'PATH', 'PROMPT_COMMAND', + 'PS1', // Bash appends its non-printing Readline readiness marker. 'CURSOR', 'ZDOTDIR', 'precmd_functions', diff --git a/src/main/ssh/build-toolchain-diagnosis.ts b/src/main/ssh/build-toolchain-diagnosis.ts index c64a41ead51..77d20ce475a 100644 --- a/src/main/ssh/build-toolchain-diagnosis.ts +++ b/src/main/ssh/build-toolchain-diagnosis.ts @@ -163,3 +163,66 @@ export function formatMissingToolchainError( ] return lines.join('\n') } + +const NODE_HEADERS_TARBALL_RE = /node-v[0-9.]+-headers\.tar\.gz/i + +/** + * Whether a native-deps failure is node-gyp failing to download Node headers from nodejs.org. + * + * Why it needs naming: the raw output is forty lines of `gyp http` and stack frames around one + * `ECONNREFUSED`, and it reads as a broken host or a broken Orca. Which of two things it is + * depends on what the local-headers export found first, so the formatter takes that answer. + */ +export function isNodeHeadersDownloadFailure(message: string): boolean { + // Why `configure error` is required: node-gyp's fetch client logs `attempt N failed with ` + // on retries it then recovers from, so a network token alone also matches a build that got its + // headers and died later for an unrelated reason. Only the configure step downloads headers. + return ( + /gyp ERR! configure error/i.test(message) && + NODE_HEADERS_TARBALL_RE.test(message) && + /\b(ECONNREFUSED|ENOTFOUND|ETIMEDOUT|EHOSTUNREACH|ENETUNREACH|EAI_AGAIN|ECONNRESET)\b/.test( + message + ) + ) +} + +const NODE_HEADERS_CONTEXT = + 'node-pty has no prebuilt binary for Linux, so it must be compiled on the remote host, and ' + + 'node-gyp fetches the Node.js headers from nodejs.org unless the Node install provides them ' + + 'at /include/node.' + +/** + * @param localHeadersDir what the local-headers export found: a dir it exported, `null` when + * the host's Node ships no matching headers, `undefined` when the answer never came back. + * + * Why the exported-dir case is its own message: the export is the fix, so node-gyp downloading + * anyway means its `nodedir` env keys were not honoured (a future npm dropping the passthrough, + * a wrapper scrubbing the env). That is an Orca defect, not a host problem, and must not be + * reported as one -- it names the dir so the report is checkable. + */ +export function formatNodeHeadersDownloadError( + underlyingError: string, + localHeadersDir: string | null | undefined +): string { + const lines = localHeadersDir + ? [ + `The remote host could not download the Node.js headers needed to compile node-pty, even ` + + `though its Node install ships matching headers at ${localHeadersDir}/include/node and ` + + `Orca pointed node-gyp at them. node-gyp ignored that setting; this is an Orca defect, ` + + `please report it with the log below.`, + '', + 'Workaround on the remote host until then: allow outbound HTTPS to nodejs.org, or point ' + + 'npm at a mirror: npm config set disturl https:///dist' + ] + : [ + 'The remote host could not download the Node.js headers needed to compile node-pty, and ' + + `its Node install has no local headers matching its own version. ${NODE_HEADERS_CONTEXT}`, + '', + 'Fix one of the following on the remote host, then reconnect:', + ' - Install Node.js from an official build or a version manager (nvm, fnm, volta, n), ' + + 'which ship headers for exactly the Node they run; or', + ' - Allow outbound HTTPS to nodejs.org, or point npm at a mirror: ' + + 'npm config set disturl https:///dist' + ] + return [...lines, '', `Underlying install error: ${underlyingError}`].join('\n') +} diff --git a/src/main/ssh/ssh-relay-build-toolchain.test.ts b/src/main/ssh/ssh-relay-build-toolchain.test.ts index b216b599f5a..6fcc4f5321d 100644 --- a/src/main/ssh/ssh-relay-build-toolchain.test.ts +++ b/src/main/ssh/ssh-relay-build-toolchain.test.ts @@ -3,7 +3,9 @@ import { buildToolchainProbeCommand, parseBuildToolchainProbe, formatMissingToolchainError, + formatNodeHeadersDownloadError, formatSkippedNodePtyWarning, + isNodeHeadersDownloadFailure, shouldProbeBuildToolchainAfterNativeDepsFailure } from './ssh-relay-build-toolchain' @@ -125,3 +127,71 @@ describe('formatSkippedNodePtyWarning', () => { expect(warning).toContain('install a C/C++ toolchain') }) }) + +// Verbatim shape of the STA-6674 failure: node-gyp on a host whose nodejs.org is refused. +const HEADERS_REFUSED = + 'npm error gyp http GET https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz\n' + + 'npm error gyp http fetch GET https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz attempt 1 failed with ECONNREFUSED\n' + + 'npm error gyp ERR! configure error\n' + + 'npm error gyp ERR! stack FetchError: request to https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz failed, reason: connect ECONNREFUSED 127.0.0.1:443' + +describe('isNodeHeadersDownloadFailure', () => { + it('matches node-gyp failing to fetch the Node headers tarball', () => { + expect(isNodeHeadersDownloadFailure(HEADERS_REFUSED)).toBe(true) + expect( + isNodeHeadersDownloadFailure( + 'gyp http fetch GET https://nodejs.org/download/release/v20.19.0/node-v20.19.0-headers.tar.gz attempt 1 failed with ENOTFOUND\ngyp ERR! configure error' + ) + ).toBe(true) + }) + + it('is not the toolchain diagnosis, and does not fire on other network failures', () => { + expect(shouldProbeBuildToolchainAfterNativeDepsFailure(HEADERS_REFUSED)).toBe(false) + // The registry, not nodejs.org: a different remedy. + expect( + isNodeHeadersDownloadFailure( + 'npm error network request to https://registry.npmjs.org/node-pty failed, reason: connect ECONNREFUSED' + ) + ).toBe(false) + // Headers named but the build failed for another reason. + expect( + isNodeHeadersDownloadFailure( + 'gyp info using node-v24.12.0-headers.tar.gz\ngyp ERR! build error make failed with exit code: 2' + ) + ).toBe(false) + // A retried attempt that recovered, then a compile failure: not a download failure. + expect( + isNodeHeadersDownloadFailure( + 'gyp http fetch GET https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz attempt 1 failed with ECONNRESET\n' + + 'gyp http 200 https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz\n' + + 'gyp ERR! build error\ngyp ERR! stack Error: `make` failed with exit code: 2' + ) + ).toBe(false) + // A mirror answering non-2xx is a FetchError without a network code: a different remedy. + expect( + isNodeHeadersDownloadFailure( + 'gyp ERR! configure error\ngyp ERR! stack FetchError: 404 Not Found https://mirror/dist/v24.12.0/node-v24.12.0-headers.tar.gz' + ) + ).toBe(false) + }) +}) + +describe('formatNodeHeadersDownloadError', () => { + it('names both host remedies when the host ships no headers', () => { + const msg = formatNodeHeadersDownloadError(HEADERS_REFUSED, null) + expect(msg).toContain('no local headers matching its own version') + expect(msg).toContain('/include/node') + expect(msg).toContain('nvm, fnm, volta, n') + expect(msg).toContain('disturl') + expect(msg).toContain('ECONNREFUSED') + }) + + it('reports an Orca defect, not a host problem, when headers were exported and ignored', () => { + const msg = formatNodeHeadersDownloadError(HEADERS_REFUSED, '/usr/local') + expect(msg).toContain('/usr/local/include/node') + expect(msg).toContain('Orca defect') + expect(msg).not.toContain('no local headers matching its own version') + expect(msg).not.toContain('nvm, fnm, volta, n') + expect(msg).toContain('ECONNREFUSED') + }) +}) diff --git a/src/main/ssh/ssh-relay-build-toolchain.ts b/src/main/ssh/ssh-relay-build-toolchain.ts index bcb64d3bdf9..db7b6353697 100644 --- a/src/main/ssh/ssh-relay-build-toolchain.ts +++ b/src/main/ssh/ssh-relay-build-toolchain.ts @@ -16,7 +16,9 @@ export { shouldProbeBuildToolchainAfterNativeDepsFailure, toolchainInstallHintLines, formatSkippedNodePtyWarning, - formatMissingToolchainError + formatMissingToolchainError, + formatNodeHeadersDownloadError, + isNodeHeadersDownloadFailure } from './build-toolchain-diagnosis' export type { BuildToolchainStatus } from './build-toolchain-diagnosis' diff --git a/src/main/ssh/ssh-relay-deploy.ts b/src/main/ssh/ssh-relay-deploy.ts index e7450478d92..5d8101361c6 100644 --- a/src/main/ssh/ssh-relay-deploy.ts +++ b/src/main/ssh/ssh-relay-deploy.ts @@ -53,11 +53,14 @@ import { } from './ssh-relay-deploy-timing' import { createSshOperationAbortError, shellEscape } from './ssh-connection-utils' import { isWindowsRelayPlatform } from '../../shared/relay-artifacts' +import { exportLocalNodeHeadersPrefix, localNodeHeadersFromOutput } from './ssh-relay-node-headers' import { probeBuildToolchain, formatMissingToolchainError, formatSkippedNodePtyWarning, - shouldProbeBuildToolchainAfterNativeDepsFailure + shouldProbeBuildToolchainAfterNativeDepsFailure, + formatNodeHeadersDownloadError, + isNodeHeadersDownloadFailure } from './ssh-relay-build-toolchain' import { commandWithNodePath, @@ -1174,7 +1177,7 @@ async function installNativeDeps( hostPlatform, nodePath, remoteDir, - `${resetPrefix}npm install --ignore-scripts=false --omit=dev --no-audit --no-fund ${installArgs} 2>&1` + `${exportLocalNodeHeadersPrefix(nodePath)}${resetPrefix}npm install --ignore-scripts=false --omit=dev --no-audit --no-fund ${installArgs} 2>&1` ) await execHostCommand(conn, hostPlatform, command, { timeoutMs: NATIVE_DEPS_COMMAND_TIMEOUT_MS, @@ -1236,6 +1239,14 @@ async function installNativeDeps( return } } + // Why: either the local-headers export found nothing (a host both header-less and offline) or + // it did and node-gyp downloaded anyway (the export is broken) -- name which, or the log reads + // as a broken relay either way. + if (platform.startsWith('linux') && isNodeHeadersDownloadFailure(msg)) { + throw new Error(formatNodeHeadersDownloadError(msg, localNodeHeadersFromOutput(msg)), { + cause: err + }) + } throw err } @@ -1254,8 +1265,15 @@ async function installNativeDeps( throw err } signal?.throwIfAborted() + // Same diagnosis as the install catch: this fallback is non-fatal, so the log is the only + // place the offline-headers cause can reach anyone. + const rebuildMsg = (err as Error).message console.warn( - `[ssh-relay][NATIVE-DEPS-REBUILD-FAIL] npm rebuild native deps failed at ${remoteDir} (${platform}): ${(err as Error).message}` + `[ssh-relay][NATIVE-DEPS-REBUILD-FAIL] npm rebuild native deps failed at ${remoteDir} (${platform}): ${ + platform.startsWith('linux') && isNodeHeadersDownloadFailure(rebuildMsg) + ? formatNodeHeadersDownloadError(rebuildMsg, localNodeHeadersFromOutput(rebuildMsg)) + : rebuildMsg + }` ) } signal?.throwIfAborted() @@ -1347,7 +1365,7 @@ async function applyNodePtyMasterCloexecPatch( hostPlatform, nodePath, remoteDir, - `${shellEscape(nodePath)} ${shellEscape(NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME)} 2>&1` + `${exportLocalNodeHeadersPrefix(nodePath)}${shellEscape(nodePath)} ${shellEscape(NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME)} 2>&1` ) const output = await execHostCommand(conn, hostPlatform, command, { timeoutMs: NATIVE_DEPS_COMMAND_TIMEOUT_MS, @@ -1529,7 +1547,7 @@ async function rebuildNativeDeps( hostPlatform, nodePath, remoteDir, - `npm rebuild --ignore-scripts=false ${depNames.map(shellEscape).join(' ')} 2>&1` + `${exportLocalNodeHeadersPrefix(nodePath)}npm rebuild --ignore-scripts=false ${depNames.map(shellEscape).join(' ')} 2>&1` ) await execHostCommand(conn, hostPlatform, command, { timeoutMs: NATIVE_DEPS_COMMAND_TIMEOUT_MS, diff --git a/src/main/ssh/ssh-relay-native-deps-install-staged-upload.test.ts b/src/main/ssh/ssh-relay-native-deps-install-staged-upload.test.ts index 4d90a7c8289..e8616e7c230 100644 --- a/src/main/ssh/ssh-relay-native-deps-install-staged-upload.test.ts +++ b/src/main/ssh/ssh-relay-native-deps-install-staged-upload.test.ts @@ -154,6 +154,69 @@ describe('installNativeDeps staged uploads', () => { expect(writeObservedAt).toBeLessThanOrEqual(npmInstallIdx) }) + it('exports the host Node headers dir to node-gyp on every command that can compile node-pty (STA-6674)', async () => { + const conn = makeMockConnection(sftpCapture) + // Install succeeds, the probe fails, the rebuild repairs it, then the cloexec patch rebuilds again. + feed(makeExecResponses({ npmInstall: 'ok', probe: 'missing', repairProbe: 'ok' })) + + await deployAndLaunchRelay(conn) + + const commands = vi.mocked(execCommand).mock.calls.map(([, command]) => command) + const compiling = ['npm install', 'npm rebuild', 'node-pty-1.1.0-master-cloexec-patch.cjs'] + for (const compileStep of compiling) { + const command = commands.find((candidate) => candidate.includes(compileStep)) + expect(command, compileStep).toBeDefined() + // Both spellings: node-gyp 10 (Node 20) reads only npm_config_, node-gyp >= 11.4 prefers the other. + expect(command).toContain('export npm_config_nodedir=') + expect(command).toContain('npm_package_config_node_gyp_nodedir=') + // The export precedes the compile on the same command line, and only when the probe found headers. + expect(command!.indexOf('npm_config_nodedir')).toBeLessThan(command!.indexOf(compileStep)) + expect(command).toContain('node_version.h') + // The marker lands in the captured output, so a failure after it can say what was exported. + expect(command).toContain('echo "ORCA-NODE-HEADERS:${ORCA_NODE_HEADERS_DIR:-none}"') + } + }) + + // What execCommand actually rejects with: the whole command line (marker echo included) quoted + // ahead of the host's output. A fixture that omits the command hides the marker-parsing bug. + function rejectNpmInstallLikeExecCommand(hostOutput: string): void { + vi.mocked(execCommand).mockImplementationOnce(async (_conn, command) => { + throw new Error(`Command "${command}" failed (exit 1): ${hostOutput}`) + }) + } + const HEADERS_REFUSED = + 'npm error gyp http fetch GET https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz attempt 1 failed with ECONNREFUSED\nnpm error gyp ERR! configure error' + + it('names the fix when node-gyp cannot download headers and the host ships none (STA-6674)', async () => { + const conn = makeMockConnection(sftpCapture) + feed(makeStagedFirstInstallExecPrefix()) + rejectNpmInstallLikeExecCommand(`ORCA-NODE-HEADERS:none\n${HEADERS_REFUSED}`) + feed(['']) // clean stage root + + const error = await deployAndLaunchRelay(conn).catch((e: Error) => e) + expect((error as Error).message).toContain('could not download the Node.js headers') + expect((error as Error).message).toContain('no local headers matching its own version') + expect((error as Error).message).not.toContain('Orca defect') + expect((error as Error).message).toContain('ECONNREFUSED') + // A full toolchain: the toolchain probe must not run, and this is not a "build tools" error. + expect((error as Error).message).not.toContain('build tools') + const commands = vi.mocked(execCommand).mock.calls.map(([, command]) => command) + expect(commands.some((command) => command.includes('command -v "$t"'))).toBe(false) + }) + + it('reports an Orca defect when headers were exported but node-gyp downloaded anyway', async () => { + // The marker says the export happened; a download after it means node-gyp never read the env. + const conn = makeMockConnection(sftpCapture) + feed(makeStagedFirstInstallExecPrefix()) + rejectNpmInstallLikeExecCommand(`ORCA-NODE-HEADERS:/usr/local\n${HEADERS_REFUSED}`) + feed(['']) // clean stage root + + const error = await deployAndLaunchRelay(conn).catch((e: Error) => e) + expect((error as Error).message).toContain('/usr/local/include/node') + expect((error as Error).message).toContain('Orca defect') + expect((error as Error).message).not.toContain('no local headers matching its own version') + }) + it('promotes only after the first-install lock is acquired', async () => { const conn = makeMockConnection(sftpCapture) feed(makeExecResponses({ npmInstall: 'ok', probe: 'ok' })) diff --git a/src/main/ssh/ssh-relay-node-headers.test.ts b/src/main/ssh/ssh-relay-node-headers.test.ts new file mode 100644 index 00000000000..84e9016738b --- /dev/null +++ b/src/main/ssh/ssh-relay-node-headers.test.ts @@ -0,0 +1,164 @@ +import { spawnSync } from 'node:child_process' +import { + chmodSync, + copyFileSync, + mkdtempSync, + mkdirSync, + rmSync, + symlinkSync, + writeFileSync +} from 'node:fs' +import { tmpdir } from 'node:os' +import { dirname, join } from 'node:path' +import process from 'node:process' +import { afterEach, describe, expect, it } from 'vitest' +import { exportLocalNodeHeadersPrefix, localNodeHeadersFromOutput } from './ssh-relay-node-headers' + +const POSIX = process.platform !== 'win32' + +/** Runs the prefix under /bin/sh exactly as the relay does, then prints what node-gyp would see. */ +function runPrefix(nodePath: string): { + nodedir: string + pkgNodedir: string + marker: string | null | undefined +} { + const script = `${exportLocalNodeHeadersPrefix(nodePath)}printf '%s\\n%s\\n' "$npm_config_nodedir" "$npm_package_config_node_gyp_nodedir"` + const result = spawnSync('/bin/sh', ['-c', script], { encoding: 'utf8' }) + expect(result.status).toBe(0) + const marker = localNodeHeadersFromOutput(result.stdout) + const [nodedir = '', pkgNodedir = ''] = result.stdout + .split('\n') + .filter((line) => !line.startsWith('ORCA-NODE-HEADERS:')) + return { nodedir, pkgNodedir, marker } +} + +/** A fake `/bin/node` whose `include/node/node_version.h` claims `version`. */ +function fakeNodePrefix(root: string, version: string): string { + const prefix = join(root, 'prefix') + mkdirSync(join(prefix, 'bin'), { recursive: true }) + mkdirSync(join(prefix, 'include', 'node'), { recursive: true }) + const [major, minor, patch] = version.split('.') + writeFileSync( + join(prefix, 'include', 'node', 'node_version.h'), + `#define NODE_MAJOR_VERSION ${major}\n#define NODE_MINOR_VERSION ${minor}\n#define NODE_PATCH_VERSION ${patch}\n` + ) + // Why a symlink to the real binary: the probe reads process.execPath, which Node resolves + // through symlinks -- so this stands in for `/usr/bin/node -> /opt/node/bin/node` shims too. + symlinkSync(process.execPath, join(prefix, 'bin', 'node')) + return join(prefix, 'bin', 'node') +} + +describe.skipIf(!POSIX)('exportLocalNodeHeadersPrefix', () => { + const roots: string[] = [] + afterEach(() => { + for (const root of roots.splice(0)) { + rmSync(root, { recursive: true, force: true }) + } + }) + + it('exports nodedir when the running Node ships headers for its own version', () => { + // The test runner's Node is an official build, so its prefix has include/node. + const prefix = dirname(dirname(process.execPath)) + const { nodedir, pkgNodedir, marker } = runPrefix(process.execPath) + expect(nodedir).toBe(prefix) + expect(pkgNodedir).toBe(prefix) + expect(marker).toBe(prefix) + }) + + it('leaves nodedir unset when the shipped headers are for another Node version', () => { + // A symlinked node resolves execPath to the real binary, whose prefix is the real one; so + // to stage a mismatch the probe must run a node whose execPath lands in the fake prefix. + // A copy does that. + const root = mkdtempSync(join(tmpdir(), 'orca-node-headers-')) + roots.push(root) + const prefix = join(root, 'prefix') + mkdirSync(join(prefix, 'bin'), { recursive: true }) + mkdirSync(join(prefix, 'include', 'node'), { recursive: true }) + writeFileSync( + join(prefix, 'include', 'node', 'node_version.h'), + '#define NODE_MAJOR_VERSION 1\n#define NODE_MINOR_VERSION 0\n#define NODE_PATCH_VERSION 0\n' + ) + const copied = join(prefix, 'bin', 'node') + copyFileSync(process.execPath, copied) + chmodSync(copied, 0o755) + const { nodedir, pkgNodedir, marker } = runPrefix(copied) + expect(nodedir).toBe('') + expect(pkgNodedir).toBe('') + expect(marker).toBeNull() + }) + + it('leaves nodedir unset when the prefix has no headers at all', () => { + const root = mkdtempSync(join(tmpdir(), 'orca-node-headers-')) + roots.push(root) + const copied = join(root, 'bin', 'node') + mkdirSync(dirname(copied), { recursive: true }) + copyFileSync(process.execPath, copied) + chmodSync(copied, 0o755) + const { nodedir } = runPrefix(copied) + expect(nodedir).toBe('') + }) + + it('follows a symlinked node to the install that owns the headers', () => { + const root = mkdtempSync(join(tmpdir(), 'orca-node-headers-')) + roots.push(root) + const shim = fakeNodePrefix(root, '0.0.0') + // The shim's own fake headers are ignored: execPath resolves to the real binary, and the + // real prefix's headers are the ones that match. + const { nodedir } = runPrefix(shim) + expect(nodedir).toBe(dirname(dirname(process.execPath))) + }) + + it('clears an inherited nodedir when the probe finds no matching headers', () => { + // A remote profile's stale nodedir must not survive past the version check. + const root = mkdtempSync(join(tmpdir(), 'orca-node-headers-')) + roots.push(root) + const copied = join(root, 'bin', 'node') + mkdirSync(dirname(copied), { recursive: true }) + copyFileSync(process.execPath, copied) + chmodSync(copied, 0o755) + const script = `${exportLocalNodeHeadersPrefix(copied)}printf '%s|%s|%s' "$npm_config_nodedir" "$NPM_CONFIG_NODEDIR" "$npm_package_config_node_gyp_nodedir"` + const result = spawnSync('/bin/sh', ['-c', script], { + encoding: 'utf8', + env: { + ...process.env, + npm_config_nodedir: '/usr/stale-headers', + NPM_CONFIG_NODEDIR: '/usr/stale-headers', + npm_package_config_node_gyp_nodedir: '/usr/stale-headers' + } + }) + expect(result.status).toBe(0) + expect(result.stdout.split('\n').at(-1)).toBe('||') + }) + + it('does not fail the command line when node itself cannot run', () => { + const script = `${exportLocalNodeHeadersPrefix('/nonexistent/node')}echo "after:$npm_config_nodedir"` + const result = spawnSync('/bin/sh', ['-c', script], { encoding: 'utf8' }) + expect(result.status).toBe(0) + expect(result.stdout.trim()).toBe('ORCA-NODE-HEADERS:none\nafter:') + }) +}) + +describe('localNodeHeadersFromOutput', () => { + it('reads the host answer, not the copy of the marker echo quoted in an exec-failure head', () => { + // The real shape: execCommand quotes the whole command line, prefix included, before the output. + const command = `export PATH='/usr/local/bin':$PATH && cd '/root/.orca-remote/relay-x' && ${exportLocalNodeHeadersPrefix('/usr/local/bin/node')}npm install node-pty 2>&1` + const failed = (hostOutput: string): string => + `Command "${command}" failed (exit 1): ${hostOutput}` + expect( + localNodeHeadersFromOutput(failed('ORCA-NODE-HEADERS:none\ngyp ERR! configure error')) + ).toBeNull() + expect( + localNodeHeadersFromOutput(failed('ORCA-NODE-HEADERS:/usr/local\ngyp ERR! configure error')) + ).toBe('/usr/local') + // No host output at all after the head: the command copy alone must not count as a marker. + expect(localNodeHeadersFromOutput(failed(''))).toBeUndefined() + }) + + it('distinguishes an exported dir, an explicit none, and no marker at all', () => { + expect(localNodeHeadersFromOutput('x\nORCA-NODE-HEADERS:/usr/local\ngyp ERR!')).toBe( + '/usr/local' + ) + expect(localNodeHeadersFromOutput('ORCA-NODE-HEADERS:none\ngyp ERR!')).toBeNull() + expect(localNodeHeadersFromOutput('gyp ERR! only')).toBeUndefined() + }) +}) diff --git a/src/main/ssh/ssh-relay-node-headers.ts b/src/main/ssh/ssh-relay-node-headers.ts new file mode 100644 index 00000000000..a590bd40fca --- /dev/null +++ b/src/main/ssh/ssh-relay-node-headers.ts @@ -0,0 +1,97 @@ +/** + * Point node-gyp at the headers the host's Node install already ships, so compiling node-pty + * needs nothing from nodejs.org. + * + * Why: node-pty has no Linux prebuild, so every Linux relay compiles it, and node-gyp's default + * is to download `node-v-headers.tar.gz` before configuring. Every official Node build, and + * every version manager that unpacks one (nvm, fnm, volta, mise, n), already has those exact + * headers at `/include/node`. The download was the only step that needed the internet, + * so a firewalled host failed with ECONNREFUSED on work that never had to happen (STA-6674). + * + * Why both variables: node-gyp >= 11.4 prefers `npm_package_config_node_gyp_` and npm 11+ + * warns that arbitrary `npm_config_` is deprecated, but node-gyp 10 (bundled with Node 20) + * reads only `npm_config_`. Both together cover every Node the relay runs on. + * + * Why the version check: node-gyp trusts `nodedir` blindly, so a distro `/usr/include/node` left + * by an older headers package would be compiled against as-is. Whether that binding then misbehaves + * is not established (one measured run loaded a node-20-header build under node 24); refusing is + * the conservative default. A mismatch leaves the variables unset, which is today's path. + */ +import { shellEscape } from './ssh-connection-utils' + +/** Shell variable the probe answers into; namespaced so it cannot collide with npm's own. */ +const NODEDIR_SHELL_VAR = 'ORCA_NODE_HEADERS_DIR' + +/** + * Prints the running Node's install prefix when `/include/node/node_version.h` matches + * `process.versions.node`, and nothing otherwise. `process.execPath` is symlink-resolved, so a + * `/usr/bin/node` -> `/opt/node/bin/node` shim still finds `/opt/node/include`. + */ +export const LOCAL_NODE_HEADERS_PROBE_JS = [ + 'const p=require("path"),f=require("fs");', + 'const d=p.dirname(p.dirname(process.execPath));', + 'try{', + 'const h=f.readFileSync(p.join(d,"include","node","node_version.h"),"utf8");', + 'const v=["MAJOR","MINOR","PATCH"].map(k=>(h.match(new RegExp("#define NODE_"+k+"_VERSION ([0-9]+)"))||[])[1]).join(".");', + 'if(v===process.versions.node)process.stdout.write(d)', + '}catch{}' +].join('') + +/** + * Stdout marker naming what the probe found, printed before the compile so the answer is in the + * captured output of any failure that follows. `none` means no matching local headers. + */ +export const LOCAL_NODE_HEADERS_MARKER_PREFIX = 'ORCA-NODE-HEADERS:' + +/** + * POSIX-sh prefix (`...; `) that exports node-gyp's `nodedir` for the rest of the command line + * when the host's Node ships matching headers. Prepend to any command that may compile node-pty: + * `npm install`, `npm rebuild`, and the cloexec patch (its `npm rebuild` inherits the env). + */ +export function exportLocalNodeHeadersPrefix(nodePath: string): string { + const probe = `${shellEscape(nodePath)} -e ${shellEscape(LOCAL_NODE_HEADERS_PROBE_JS)} 2>/dev/null` + // Why the unset: a remote profile can already export a nodedir (a stale distro header dir), in + // either case npm accepts. Left alone it would bypass the version check above and compile + // against those headers. Deliberately env only: a `nodedir=` in ~/.npmrc is not reachable from here + // -- npm ignores an empty env override, and a CLI `--nodedir=` would also override the good + // export -- so an npmrc setting stays the operator's, as it was before this prefix existed. + return ( + `${NODEDIR_SHELL_VAR}=$(${probe}); ` + + `unset npm_config_nodedir NPM_CONFIG_NODEDIR npm_package_config_node_gyp_nodedir; ` + + `if [ -n "$${NODEDIR_SHELL_VAR}" ]; then ` + + `export npm_config_nodedir="$${NODEDIR_SHELL_VAR}" npm_package_config_node_gyp_nodedir="$${NODEDIR_SHELL_VAR}"; ` + + `fi; ` + + `echo "${LOCAL_NODE_HEADERS_MARKER_PREFIX}\${${NODEDIR_SHELL_VAR}:-none}"; ` + ) +} + +/** + * The headers dir the prefix exported, `null` when it found none, or `undefined` when the + * marker is absent (output truncated, or the command never reached the prefix). + */ +export function localNodeHeadersFromOutput(output: string): string | null | undefined { + // Why the head is stripped first: a failed exec's message is `Command "" failed + // (exit N): `, and quotes this prefix verbatim -- including the marker's + // `echo`. Scanning from the start would match that copy and return `${ORCA_NODE_HEADERS_DIR:- + // none}"...` as a "dir". Only what follows the head is the host's answer. + const head = output.match(EXEC_FAILURE_HEAD_RE) + const hostOutput = head ? output.slice(head[0].length) : output + // First match, not last: the host's own line comes first, and later lines are npm/gyp output + // that must not be able to spoof it. + for (const line of hostOutput.split(/\r?\n/)) { + const at = line.indexOf(LOCAL_NODE_HEADERS_MARKER_PREFIX) + if (at === -1) { + continue + } + const dir = line.slice(at + LOCAL_NODE_HEADERS_MARKER_PREFIX.length).trim() + return dir === 'none' || dir === '' ? null : dir + } + return undefined +} + +/** + * `Command "" failed (exit N): ` -- see ssh-relay-exec-command.ts. + * Lazy `[\s\S]*?` is safe: it stops at the first `" failed (exit N): `, and no command this + * module builds contains that literal, so the match cannot end early inside the command. + */ +const EXEC_FAILURE_HEAD_RE = /^Command "[\s\S]*?" failed \(exit -?\d+\): / diff --git a/src/main/ssh/ssh-relay-offline-node-headers.docker.test.ts b/src/main/ssh/ssh-relay-offline-node-headers.docker.test.ts new file mode 100644 index 00000000000..c5700c991c1 --- /dev/null +++ b/src/main/ssh/ssh-relay-offline-node-headers.docker.test.ts @@ -0,0 +1,204 @@ +// Why this exists (STA-6674): a Linux host whose only unreachable endpoint is nodejs.org could +// not run a relay. node-pty ships no Linux prebuild, so npm hands it to node-gyp, and node-gyp +// downloads `node-v-headers.tar.gz` unless told the host already has the headers -- which +// every official Node install does, at `/include/node`. This drives the real deploy at a +// Docker sshd whose nodejs.org resolves to 127.0.0.1 (ECONNREFUSED, exactly what the user saw). +// +// Run: ORCA_REVIEW_SSH_OFFLINE_HEADERS=1 pnpm test src/main/ssh/ssh-relay-offline-node-headers.docker.test.ts +// Needs Docker and `pnpm build:relay`. ORCA_REVIEW_SSH_NODE_IMAGE picks the Node image +// (default node:24.12.0-bookworm, the user's version); ORCA_REVIEW_SSH_TARGET_HOST overrides +// the address the app connects to (default 127.0.0.1). +import { execFileSync, spawnSync } from 'node:child_process' +import { randomUUID } from 'node:crypto' +import { mkdtempSync, readFileSync, rmSync } from 'node:fs' +import { connect } from 'node:net' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ app: { getAppPath: () => process.cwd() } })) + +import { SshConnection } from './ssh-connection' +import { deployAndLaunchRelay } from './ssh-relay-deploy' +import type { SshTarget } from '../../shared/ssh-types' + +const RUN_REVIEW_ORACLE = process.env.ORCA_REVIEW_SSH_OFFLINE_HEADERS === '1' +const NODE_IMAGE = process.env.ORCA_REVIEW_SSH_NODE_IMAGE ?? 'node:24.12.0-bookworm' +const TARGET_HOST = process.env.ORCA_REVIEW_SSH_TARGET_HOST ?? '127.0.0.1' + +type TargetFixture = { + containerName: string + identityFile: string + port: number + tempDir: string +} + +function run(command: string, args: string[], timeout = 30_000, input?: string): string { + return execFileSync(command, args, { + encoding: 'utf8', + stdio: [input === undefined ? 'ignore' : 'pipe', 'pipe', 'pipe'], + timeout, + input + }).trim() +} + +function dockerExec(fixture: TargetFixture, command: string): string { + return run('docker', ['exec', fixture.containerName, 'bash', '-lc', command], 60_000) +} + +async function startTarget(): Promise { + const image = `orca-review-offline-headers:${NODE_IMAGE.replace(/[^A-Za-z0-9_.-]/g, '-')}` + run( + 'docker', + ['build', '-q', '-t', image, '-'], + 600_000, + [ + `FROM ${NODE_IMAGE}`, + 'RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends openssh-server git && rm -rf /var/lib/apt/lists/* && mkdir -p /run/sshd /root/.ssh && chmod 700 /root/.ssh', + '' + ].join('\n') + ) + const tempDir = mkdtempSync(join(tmpdir(), 'orca-offline-headers-ssh-')) + const identityFile = join(tempDir, 'id_ed25519') + run('ssh-keygen', ['-t', 'ed25519', '-N', '', '-f', identityFile, '-q']) + const publicKey = readFileSync(`${identityFile}.pub`, 'utf8').trim() + const containerName = `orca-offline-headers-${randomUUID().slice(0, 12)}` + // Why a refused connection and not a dropped one: a timeout takes node-gyp's retry path and + // burns the deploy budget; the user's host refused, and that is the path under test. + run( + 'docker', + [ + 'run', + '-d', + '--name', + containerName, + '--add-host', + 'nodejs.org:127.0.0.1', + '-p', + '0.0.0.0::22', + '-e', + `AUTHORIZED_KEY=${publicKey}`, + image, + 'bash', + '-lc', + 'printf "%s\\n" "$AUTHORIZED_KEY" > /root/.ssh/authorized_keys && chmod 600 /root/.ssh/authorized_keys && exec /usr/sbin/sshd -D -e' + ], + 120_000 + ) + const port = Number(run('docker', ['port', containerName, '22/tcp']).split(':').at(-1)) + // `docker run -d` returns before sshd binds; connect() against a closed port is a flake. + await waitForSshBanner(port) + return { containerName, identityFile, port, tempDir } +} + +/** Resolves once sshd answers with its banner on the mapped port, or throws after the deadline. */ +async function waitForSshBanner(port: number, deadlineMs = 60_000): Promise { + const deadline = Date.now() + deadlineMs + for (;;) { + const gotBanner = await new Promise((resolve) => { + const socket = connect({ host: TARGET_HOST, port }) + const done = (value: boolean): void => { + socket.destroy() + resolve(value) + } + socket.setTimeout(2_000, () => done(false)) + socket.once('data', (chunk) => done(chunk.toString('utf8').startsWith('SSH-'))) + socket.once('error', () => done(false)) + }) + if (gotBanner) { + return + } + if (Date.now() > deadline) { + throw new Error(`sshd on port ${port} did not answer within ${deadlineMs / 1000}s`) + } + await new Promise((resolve) => setTimeout(resolve, 500)) + } +} + +function stopTarget(fixture: TargetFixture | null): void { + if (!fixture) { + return + } + spawnSync('docker', ['rm', '-f', fixture.containerName], { stdio: 'ignore', timeout: 30_000 }) + rmSync(fixture.tempDir, { recursive: true, force: true }) +} + +function createConnection(fixture: TargetFixture): SshConnection { + const target: SshTarget = { + id: `offline-headers-${randomUUID()}`, + label: 'Offline node headers Docker SSH target', + source: 'manual', + host: TARGET_HOST, + port: fixture.port, + username: 'root', + identityFile: fixture.identityFile, + identitiesOnly: true + } + return new SshConnection(target, { onStateChange: vi.fn() }) +} + +describe.skipIf(!RUN_REVIEW_ORACLE)( + 'SSH relay deploy on a host that cannot reach nodejs.org', + () => { + let fixture: TargetFixture | null = null + + beforeAll(async () => { + fixture = await startTarget() + }, 900_000) + + afterAll(() => { + stopTarget(fixture) + }) + + it('compiles node-pty from the host Node install headers instead of downloading them', async () => { + const activeFixture = fixture as TargetFixture + expect(dockerExec(activeFixture, 'getent hosts nodejs.org')).toContain('127.0.0.1') + const connection = createConnection(activeFixture) + await connection.connect() + try { + const result = await deployAndLaunchRelay(connection, undefined, 60) + expect(result.remoteRelayDir).toBeTruthy() + + const evidence = dockerExec( + activeFixture, + [ + `cd '${result.remoteRelayDir}'`, + 'test -f node_modules/node-pty/build/Release/pty.node && echo PTY_NODE=built', + 'test -d /root/.cache/node-gyp && echo HEADERS=downloaded || echo HEADERS=local', + `node -e "require('node-pty'); require('@parcel/watcher'); console.log('NATIVE=loadable')"` + ].join('; ') + ) + console.log(`[offline-node-headers] ${NODE_IMAGE}: ${evidence.replace(/\n/g, ' ')}`) + expect(evidence).toContain('PTY_NODE=built') + expect(evidence).toContain('HEADERS=local') + expect(evidence).toContain('NATIVE=loadable') + } finally { + await connection.disconnect() + } + }, 600_000) + + it('names the missing-local-headers cause, not an Orca defect, when the host ships no headers', async () => { + // Same offline host, headers removed and the relay uninstalled so the deploy compiles again. + // This is the shape a review found misreported: the exec-failure message quotes the whole + // command (marker echo included) ahead of the output, and the parser must not read that copy. + const activeFixture = fixture as TargetFixture + dockerExec( + activeFixture, + 'rm -rf /usr/local/include/node /root/.orca-remote /root/.cache/node-gyp' + ) + const connection = createConnection(activeFixture) + await connection.connect() + try { + const error = await deployAndLaunchRelay(connection, undefined, 60).catch((e: Error) => e) + expect(error).toBeInstanceOf(Error) + const message = (error as Error).message + console.log(`[offline-node-headers] ${NODE_IMAGE} no-headers: ${message.split('\n')[0]}`) + expect(message).toContain('no local headers matching its own version') + expect(message).not.toContain('Orca defect') + expect(message).toContain('ECONNREFUSED') + } finally { + await connection.disconnect() + } + }, 600_000) + } +) diff --git a/src/main/startup/main-process-preflight.ts b/src/main/startup/main-process-preflight.ts index 269eb2212e7..177a2357441 100644 --- a/src/main/startup/main-process-preflight.ts +++ b/src/main/startup/main-process-preflight.ts @@ -48,7 +48,7 @@ import { } from './single-instance-lock' import { setAppEnvironment } from '../../shared/app-environment' import { ElectronAppEnvironment } from '../host/electron-app-environment' -import { installProcessTreeKillBreadcrumbObserver } from '../crash-reporting/self-initiated-tree-kill-log' +import { installMainProcessTreeKillGate } from '../own-chromium-tree-kill-guard' import { setSecretStore } from '../../shared/secret-store' import { ElectronSecretStore } from '../host/electron-secret-store' import { setPtyHostBindings } from '../ipc/pty-host-bindings' @@ -163,8 +163,8 @@ export function runMainProcessPreflight(options: MainProcessPreflightOptions): b }) } // Why before any spawn: `signalProcessTree` is shared with the CLI and relay, so - // it can only reach the main-process breadcrumb store through a registered observer. - installProcessTreeKillBreadcrumbObserver() + // it can only reach the main-process guard and breadcrumb store once this is registered. + installMainProcessTreeKillGate() const isDev = is.dev configureDevUserDataPath(isDev) configureOrcaUserDataPathEnv() diff --git a/src/main/startup/main-window-actions.ts b/src/main/startup/main-window-actions.ts index 585fa0a754e..0acc8d1a074 100644 --- a/src/main/startup/main-window-actions.ts +++ b/src/main/startup/main-window-actions.ts @@ -16,7 +16,10 @@ import { describeInstallDirAclPoison, isBlockingInstallDirAclRepairInFlight } from './windows-install-dir-acl-recovery' -import { presentRendererRecoveryPrompt } from '../window/renderer-recovery-prompt' +import { + presentRendererRecoveryPrompt, + type RendererRecoveryPromptFailure +} from '../window/renderer-recovery-prompt' // The window module injects this callback to avoid a cycle between actions and lifecycle code. let openWindow: (options?: { revealOnDidFinishLoad?: boolean }) => BrowserWindow @@ -147,9 +150,14 @@ export function sendOpenCrashReport(targetWindow?: BrowserWindow | null): void { } // Why: on renderer crash-loop the breaker stops auto-reloading and the window goes blank, so a main-process dialog is the only retry/quit surface. -export async function showRendererRecoveryPrompt(recentRecoveryCount: number): Promise { +export async function showRendererRecoveryPrompt( + recentRecoveryCount: number, + failure?: RendererRecoveryPromptFailure, + retry?: () => void +): Promise { await presentRendererRecoveryPrompt({ recentRecoveryCount, + ...(failure ? { failure } : {}), isQuitting: () => state.isQuitting, diagnose: describeInstallDirAclPoison, showMessageBox: (options) => { @@ -164,6 +172,12 @@ export async function showRendererRecoveryPrompt(recentRecoveryCount: number): P } recordDurableCrashBreadcrumb('renderer_recovery_manual_retry') // Why: leave the breaker open so a re-crash re-raises this prompt instead of resuming the auto-reload loop. + // Why watched: Reload is the dialog's default button, and an unwatched retry that stalls returns the user to + // the same silent hang with no further prompt — the watchdog re-raises this dialog instead. + if (retry) { + retry() + return + } loadMainWindow(state.mainWindow) }, quit: () => { diff --git a/src/main/startup/main-window-controller.ts b/src/main/startup/main-window-controller.ts index 63d84df763c..8ea245fc256 100644 --- a/src/main/startup/main-window-controller.ts +++ b/src/main/startup/main-window-controller.ts @@ -113,13 +113,19 @@ export function openMainWindow(options: { revealOnDidFinishLoad?: boolean } = {} reason: details.reason, expectedTeardown: getExpectedTeardownScope(webContentsId, false) }), - onRendererRecoveryExhausted: ({ details, recentRecoveryCount }) => { - recordDurableCrashBreadcrumb('renderer_recovery_circuit_breaker_open', { - reason: details.reason, - exitCode: details.exitCode ?? null, - recentRecoveryCount - }) - void showRendererRecoveryPrompt(recentRecoveryCount) + onRendererRecoveryExhausted: ({ details, recentRecoveryCount, cause, retry }) => { + // Why two names: a stalled reload never opened the breaker, and a bundle that says it did misreads the failure. + recordDurableCrashBreadcrumb( + cause === 'reload-stalled' + ? 'renderer_recovery_reload_exhausted' + : 'renderer_recovery_circuit_breaker_open', + { + reason: details.reason, + exitCode: details.exitCode ?? null, + recentRecoveryCount + } + ) + void showRendererRecoveryPrompt(recentRecoveryCount, cause, retry) }, deferLoad: true, ...(options.revealOnDidFinishLoad === true ? { revealOnDidFinishLoad: true } : {}), @@ -131,9 +137,16 @@ export function openMainWindow(options: { revealOnDidFinishLoad?: boolean } = {} } recordCrashBreadcrumb('manual_reload_requested', { ignoreCache }) }, - onBeforeRecoveryReload: (webContentsId) => { + // Manual retries also preserve PTYs, but have their own intent breadcrumb. + onBeforeRecoveryReload: (webContentsId, trigger) => { markRecoveryReloadInFlight(webContentsId) - recordDurableCrashBreadcrumb('renderer_recovery_reload') + if (trigger === 'automatic') { + recordDurableCrashBreadcrumb('renderer_recovery_reload') + } + }, + // Pair the intent breadcrumb with its path-free outcome. + onRecoveryReloadOutcome: ({ status, ...outcome }) => { + recordDurableCrashBreadcrumb(`renderer_recovery_reload_${status}`, outcome) } }) recordCrashBreadcrumb('main_window_created') diff --git a/src/main/text-generation/commit-message-text-generation-cancellation.test.ts b/src/main/text-generation/commit-message-text-generation-cancellation.test.ts index 9cfde774384..820af710fb9 100644 --- a/src/main/text-generation/commit-message-text-generation-cancellation.test.ts +++ b/src/main/text-generation/commit-message-text-generation-cancellation.test.ts @@ -101,7 +101,7 @@ describe('generateCommitMessageFromContext', () => { cancelGenerateCommitMessageLocal('/repo') - expectChildTerminated(children[0]!) + await expectChildTerminated(children[0]!) expect(children[1]?.kill).not.toHaveBeenCalled() children[0]?.listeners.get('close')?.(null) @@ -192,7 +192,7 @@ describe('generateCommitMessageFromContext', () => { cancelGeneratePullRequestFieldsLocal('/repo') expect(children[0]?.kill).not.toHaveBeenCalled() - expectChildTerminated(children[1]!) + await expectChildTerminated(children[1]!) const commitStdout = children[0]?.listeners.get('stdout:data') commitStdout?.(Buffer.from('Update README\n')) @@ -250,7 +250,7 @@ describe('generateCommitMessageFromContext', () => { cancelGeneratePullRequestFieldsLocal('/repo') listeners.get('close')?.(null) - expectChildTerminated(child) + await expectChildTerminated(child) await expect(pullRequest).resolves.toEqual({ success: false, error: 'Generation canceled.', @@ -306,7 +306,7 @@ describe('generateCommitMessageFromContext', () => { ) cancelGenerateCommitMessageLocal('/repo') - expectChildTerminated(child) + await expectChildTerminated(child) await Promise.resolve() await Promise.resolve() await Promise.resolve() @@ -344,7 +344,7 @@ describe('generateCommitMessageFromContext', () => { error: 'Generation canceled.', canceled: true }) - expectChildTerminated(firstChild) + await expectChildTerminated(firstChild) const second = generateCommitMessageFromContext(context, params, { kind: 'local', @@ -377,7 +377,7 @@ describe('generateCommitMessageFromContext', () => { await vi.waitFor(() => expect(spawnMock).toHaveBeenCalledTimes(1)) cancelGenerateCommitMessageLocal('/descendant-repo') await expect(first).resolves.toMatchObject({ canceled: true }) - expectChildTerminated(firstChild) + await expectChildTerminated(firstChild) // SIGKILL reaches the codex process but not a grandchild that inherited its // stdout, so 'exit' arrives and 'close' never does. diff --git a/src/main/text-generation/commit-message-text-generation-local-subprocess.test.ts b/src/main/text-generation/commit-message-text-generation-local-subprocess.test.ts index c124cb85779..f73152255da 100644 --- a/src/main/text-generation/commit-message-text-generation-local-subprocess.test.ts +++ b/src/main/text-generation/commit-message-text-generation-local-subprocess.test.ts @@ -71,7 +71,7 @@ describe('generateCommitMessageFromContext', () => { error: 'agent CLI command produced too much output. Check the agent CLI configuration and try again.' }) - expectChildTerminated(child) + await expectChildTerminated(child) }) it('passes prepared provider environment to local agent subprocesses', async () => { diff --git a/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts b/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts index 4bc1a720f05..9a1f932f0a0 100644 --- a/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts +++ b/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts @@ -325,7 +325,7 @@ describe('discoverCommitMessageModelsLocal', () => { await vi.advanceTimersByTimeAsync(60_000) await assertion - expectChildTerminated(child) + await expectChildTerminated(child) expect(child.stdout.listenerCount('data')).toBe(0) expect(child.stderr.listenerCount('data')).toBe(0) expect(child.listenerCount('error')).toBe(0) @@ -352,7 +352,7 @@ describe('discoverCommitMessageModelsLocal', () => { success: false, error: 'Codex model discovery timed out after 60s.' }) - expectChildTerminated(firstChild) + await expectChildTerminated(firstChild) expect(spawnMock).toHaveBeenCalledTimes(1) firstChild.emit('close', null) @@ -413,7 +413,7 @@ describe('discoverCommitMessageModelsLocal', () => { success: false, error: 'Cursor returned too much model data.' }) - expectChildTerminated(child) + await expectChildTerminated(child) expect(child.stdout.listenerCount('data')).toBe(0) expect(child.stderr.listenerCount('data')).toBe(0) expect(child.listenerCount('error')).toBe(0) diff --git a/src/main/text-generation/commit-message-text-generation-test-harness.ts b/src/main/text-generation/commit-message-text-generation-test-harness.ts index 21103d71f40..8ba925ef087 100644 --- a/src/main/text-generation/commit-message-text-generation-test-harness.ts +++ b/src/main/text-generation/commit-message-text-generation-test-harness.ts @@ -33,15 +33,17 @@ export function withPlatform(platform: NodeJS.Platform, fn: () => T): T { // expectChildTerminated(child) with no extra argument. export function createChildTerminationExpectation( terminateWindowsProcessTreeMock: ReturnType -): (child: { pid: number; kill: ReturnType }) => void { - return (child) => { +): (child: { pid: number; kill: ReturnType }) => Promise { + return async (child) => { if (process.platform === 'win32') { expect(terminateWindowsProcessTreeMock).toHaveBeenCalledWith(child.pid, { site: 'source-control-text-generation' }) - expect(child.kill).not.toHaveBeenCalled() - return } - expect(child.kill).toHaveBeenCalledWith('SIGKILL') + // Every platform kills the root by its own handle. On win32 that is not a + // duplicate of the tree walk: it is what keeps a refused walk from resolving + // having killed nothing while the caller releases the managed-home lock. It + // runs after the walk there, so it can be a tick behind the caller. + await vi.waitFor(() => expect(child.kill).toHaveBeenCalledWith('SIGKILL')) } } diff --git a/src/main/text-generation/source-control-local-process.ts b/src/main/text-generation/source-control-local-process.ts index 170dead6b52..3f046494f0f 100644 --- a/src/main/text-generation/source-control-local-process.ts +++ b/src/main/text-generation/source-control-local-process.ts @@ -22,22 +22,25 @@ import type { TextGenerationOperation } from './source-control-text-generation-types' -export function killSourceControlAgentProcess( +export async function killSourceControlAgentProcess( child: SpawnedSourceControlAgentProcess ): Promise { const pid = child.pid if (!pid) { - return Promise.resolve() + return } if (process.platform === 'win32') { - return terminateWindowsProcessTree(pid, { site: 'source-control-text-generation' }) + // taskkill owns the tree, but the own-Chromium gate can refuse the + // pid-addressed walk; the handle-addressed root kill below cannot reach the + // recycled pid it refused, and callers release the managed-home lock on this + // promise, so it must not resolve having killed nothing. + await terminateWindowsProcessTree(pid, { site: 'source-control-text-generation' }) } try { child.kill('SIGKILL') } catch { // The process may exit between the PID check and kill. } - return Promise.resolve() } export function runLocalSourceControlPlan(input: { diff --git a/src/main/window/createMainWindow-close-confirmation.test.ts b/src/main/window/createMainWindow-close-confirmation.test.ts index a8c9f70c503..6f91dd4add8 100644 --- a/src/main/window/createMainWindow-close-confirmation.test.ts +++ b/src/main/window/createMainWindow-close-confirmation.test.ts @@ -73,8 +73,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const onQuitAborted = vi.fn() browserWindowMock.mockImplementation(function () { @@ -122,8 +122,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) browserWindowMock.mockImplementation(function () { @@ -178,8 +178,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn(), + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()), close: vi.fn(() => { windowHandlers.close({} as never) }) @@ -238,8 +238,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const updateUI = vi.fn() const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) @@ -296,8 +296,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) browserWindowMock.mockImplementation(function () { @@ -353,8 +353,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -401,8 +401,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -451,8 +451,8 @@ describe('createMainWindow', () => { maximize: vi.fn(), show: vi.fn(), destroy, - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } }) createMainWindow(null, { getIsQuitting: () => true }) @@ -500,8 +500,8 @@ describe('createMainWindow', () => { maximize: vi.fn(), show: vi.fn(), destroy, - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } }) createMainWindow(null, { getIsQuitting: () => true }) diff --git a/src/main/window/createMainWindow-markdown-editor-focus.test.ts b/src/main/window/createMainWindow-markdown-editor-focus.test.ts index 3189019f837..ef6fca229d3 100644 --- a/src/main/window/createMainWindow-markdown-editor-focus.test.ts +++ b/src/main/window/createMainWindow-markdown-editor-focus.test.ts @@ -54,8 +54,8 @@ describe('createMainWindow', () => { setWindowButtonPosition: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -104,8 +104,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -157,8 +157,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -216,8 +216,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -275,8 +275,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -355,8 +355,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -432,8 +432,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -496,8 +496,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow-recovery-reload-watchdog.test.ts b/src/main/window/createMainWindow-recovery-reload-watchdog.test.ts new file mode 100644 index 00000000000..e551579a7ab --- /dev/null +++ b/src/main/window/createMainWindow-recovery-reload-watchdog.test.ts @@ -0,0 +1,671 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type * as DurableCrashBreadcrumbModule from '../crash-reporting/durable-crash-breadcrumb' + +const { recordDurableCrashBreadcrumbMock } = vi.hoisted(() => ({ + recordDurableCrashBreadcrumbMock: vi.fn() +})) +vi.mock('../crash-reporting/durable-crash-breadcrumb', async (importOriginal) => ({ + ...(await importOriginal()), + recordDurableCrashBreadcrumb: recordDurableCrashBreadcrumbMock +})) + +vi.mock('electron', async () => + (await import('./createMainWindow-test-harness')).electronModuleMock() +) +vi.mock('@electron-toolkit/utils', async () => + (await import('./createMainWindow-test-harness')).electronToolkitUtilsMock() +) +vi.mock('./macos-tahoe-release', async () => + (await import('./createMainWindow-test-harness')).macosTahoeReleaseMock() +) +vi.mock('../app-icon', async () => (await import('./createMainWindow-test-harness')).appIconMock()) +vi.mock('../browser/browser-manager', async () => + (await import('./createMainWindow-test-harness')).browserManagerMock() +) +vi.mock('../browser/browser-client-page-renderer-runtime', async () => { + const harness = await import('./createMainWindow-test-harness') + return { + attachBrowserClientPageRenderer: harness.attachClientPageRendererMock, + retireBrowserClientPageRenderer: harness.retireClientPageRendererMock + } +}) + +import { createMainWindow } from './createMainWindow' +import { + browserWindowMock, + isMock, + powerMonitorOnMock, + resetMainWindowMocks +} from './createMainWindow-test-harness' +import { + RENDERER_RECOVERY_DEV_LOAD_TIMEOUT_MS, + RENDERER_RECOVERY_LOAD_TIMEOUT_MS +} from './renderer-recovery-reload-watchdog' + +const DOCUMENT_URL = 'file:///opt/orca/renderer/index.html' +// A real macOS install URL: the crash-report redactor's PATH_PATTERNS provably leave this one intact. +const INSTALL_PATH_LOAD_ERROR = + "ERR_FILE_NOT_FOUND (-6) loading 'file:///Users/jane.doe/Applications/Orca.app/Contents/Resources/app.asar/out/renderer/index.html'" +const CRASH = { reason: 'crashed', exitCode: 5 } as Electron.RenderProcessGoneDetails + +/** + * Regression cover for the field failure: the recovery reload is issued, never produces a document, and nothing + * notices — no did-fail-load, no breaker (it counts renderer deaths only), no retry, no prompt. + */ +describe('renderer recovery reload watchdog', () => { + beforeEach(() => { + resetMainWindowMocks() + recordDurableCrashBreadcrumbMock.mockClear() + vi.useFakeTimers() + }) + + const createHarness = () => { + // Why fan-out: dom-ready and did-finish-load have several real registrants on this one webContents, so + // last-writer-wins would silently drop the watchdog's listener if registration order ever changed. + const registered: Record void)[]> = {} + const windowHandlers: Record void> = {} + const register = (event: string, handler: (...args: any[]) => void): void => { + const handlers = (registered[event] ??= []) + handlers.push(handler) + windowHandlers[event] ??= (...args: any[]) => { + for (const listener of handlers.slice()) { + listener(...args) + } + } + } + // Loads stay pending unless a test settles one: that is exactly the stall being reproduced. + const settleLoad: { resolve: () => void; reject: (error: Error) => void }[] = [] + const pendingLoad = (): Promise => + new Promise((resolve, reject) => settleLoad.push({ resolve, reject })) + const webContents = { + id: 143, + getURL: vi.fn(() => DOCUMENT_URL), + isDestroyed: vi.fn(() => false), + on: vi.fn(register), + setZoomLevel: vi.fn(), + setBackgroundThrottling: vi.fn(), + invalidate: vi.fn(), + setWindowOpenHandler: vi.fn(), + send: vi.fn() + } + const browserWindowInstance = { + webContents, + on: vi.fn(register), + isDestroyed: vi.fn(() => false), + isMaximized: vi.fn(() => true), + isFullScreen: vi.fn(() => false), + getSize: vi.fn(() => [1200, 800]), + setSize: vi.fn(), + maximize: vi.fn(), + show: vi.fn(), + setWindowButtonPosition: vi.fn(), + loadFile: vi.fn(pendingLoad), + loadURL: vi.fn(pendingLoad) + } + browserWindowMock.mockImplementation(function () { + return browserWindowInstance + }) + const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) + const crashRenderer = (): void => { + windowHandlers['render-process-gone']?.({} as never, CRASH) + vi.advanceTimersByTime(250) + } + const reachMilestone = (milestone: 'committed' | 'dom-ready'): void => + windowHandlers[milestone === 'committed' ? 'did-navigate' : 'dom-ready']?.() + return { + browserWindowInstance, + consoleError, + crashRenderer, + reachMilestone, + settleLoad, + windowHandlers + } + } + + it('retries once when the recovery reload never produces a document, then hands the user the prompt', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + // 1 initial load + 1 recovery reload, which now stalls forever. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS - 1) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + + vi.advanceTimersByTime(1) + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith({ + status: 'timeout', + attempt: 1, + elapsedMs: RENDERER_RECOVERY_LOAD_TIMEOUT_MS, + progress: 'none' + }) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(onRecoveryReloadOutcome).toHaveBeenLastCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 2 }) + ) + // Retry budget spent: stop reloading and surface the only retry/quit surface the user has. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + expect(onRendererRecoveryExhausted).toHaveBeenCalledWith({ + details: CRASH, + webContentsId: 143, + recentRecoveryCount: 1, + cause: 'reload-stalled', + retry: expect.any(Function) + }) + + consoleError.mockRestore() + }) + + it('clears the watchdog when the recovery reload finishes loading', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(2_000) + settleLoad[1]?.resolve() + await vi.advanceTimersByTimeAsync(0) + + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith({ + status: 'loaded', + attempt: 1, + elapsedMs: 2_000 + }) + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 3) + expect(onRecoveryReloadOutcome).toHaveBeenCalledTimes(1) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + + consoleError.mockRestore() + }) + + it('keeps watching the retry when a stale did-finish-load arrives after it was issued', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, windowHandlers } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + // did-finish-load carries no attempt token: this one belongs to the load the timer just abandoned. Crediting + // the retry with it disarms the watchdog over a load still in flight — the exact hole this watchdog closes. + windowHandlers['did-finish-load']?.() + expect(onRecoveryReloadOutcome).not.toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded' }) + ) + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(onRecoveryReloadOutcome).toHaveBeenLastCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 2 }) + ) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('does not take an error page as the retry landing', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer, settleLoad, windowHandlers } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + settleLoad[1]?.reject(new Error('ERR_FILE_NOT_FOUND (-6)')) + await vi.advanceTimersByTimeAsync(0) + // Chromium commits an error document for the failed load, and that document emits did-finish-load too. + windowHandlers['did-finish-load']?.() + + expect(onRecoveryReloadOutcome).not.toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded' }) + ) + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('raises one prompt, however many times recovery gives up underneath it', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + // The renderer dies again while the box is up; the breaker never counted stalls, so it lets the reload go. + crashRenderer() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(4) + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + + // Nothing dismisses a native message box: a retry the user never asked for, or a second box, stacks on it. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(4) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + // The stall is still on the record, so the bundle does not read as a recovery that quietly worked. + expect(onRecoveryReloadOutcome).toHaveBeenLastCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 1 }) + ) + + // Answering the box with Reload hands the next verdict back to the user. + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(2) + + consoleError.mockRestore() + }) + + it('still reloads from a crash-loop prompt raised after an earlier recovery had landed', async () => { + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRendererRecoveryExhausted }) + // Every recovery reload lands, and every landed document then dies with its renderer. + for (let attempt = 1; attempt <= 3; attempt += 1) { + crashRenderer() + settleLoad[attempt]?.resolve() + await vi.advanceTimersByTimeAsync(0) + } + crashRenderer() + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + const loads = browserWindowInstance.loadFile.mock.calls.length + + // The last document landed, but the renderer took it down: declining Reload here strands the user. + + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(loads + 1) + + consoleError.mockRestore() + }) + + it('does not stack a crash-loop prompt on one that is already up', () => { + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRendererRecoveryExhausted }) + for (let attempt = 0; attempt < 5; attempt += 1) { + crashRenderer() + } + + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('escalates a rejected recovery load immediately instead of waiting out the watchdog', async () => { + const onRecoveryReloadOutcome = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome }) + crashRenderer() + settleLoad[1]?.reject(new Error("ERR_FILE_NOT_FOUND (-6) loading 'file:///opt/orca'")) + await vi.advanceTimersByTimeAsync(0) + + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ + status: 'failed', + attempt: 1, + errorCode: 'ERR_FILE_NOT_FOUND' + }) + ) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + consoleError.mockRestore() + }) + + it('ignores a superseded load rejection so ERR_ABORTED never escalates', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + // A second renderer death supersedes the first reload; Chromium rejects the abandoned load with ERR_ABORTED. + crashRenderer() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + settleLoad[1]?.reject(new Error('ERR_ABORTED (-3)')) + await vi.advanceTimersByTimeAsync(0) + + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + consoleError.mockRestore() + }) + + it('does not escalate when another navigation aborts the live recovery load', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad, windowHandlers } = + createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + // Chromium aborts the recovery load because something else replaced it — a user navigation, a close race, + // another loadURL caller. The attempt token still says this reload is live, so nothing else filters it. + settleLoad[1]?.reject(new Error(`ERR_ABORTED (-3) loading '${DOCUMENT_URL}'`)) + await vi.advanceTimersByTimeAsync(0) + + // A cold retry here would stomp the load that superseded this one. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + + // The replacement load lands, and the window the user sees was never worth a Reload/Quit prompt. The crumb + // says so: elapsedMs measures the replacement, and the budget analysis has to be able to leave it out. + windowHandlers['did-finish-load']?.() + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded', attempt: 1, superseded: true }) + ) + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + windowHandlers['did-finish-load']?.() + expect(onRecoveryReloadOutcome).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('still escalates on silence when an aborted recovery load has nothing behind it', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + settleLoad[1]?.reject(new Error('ERR_ABORTED (-3)')) + await vi.advanceTimersByTimeAsync(0) + + // Ignoring the abort must not disarm the watchdog: the cap still bounds a load that goes nowhere. + await vi.advanceTimersByTimeAsync(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 1 }) + ) + expect(onRendererRecoveryExhausted).toHaveBeenCalledWith( + expect.objectContaining({ cause: 'reload-stalled' }) + ) + + consoleError.mockRestore() + }) + + it('gives the dev server a longer budget than a packaged load', () => { + const onRecoveryReloadOutcome = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + isMock.dev = true + vi.stubEnv('ELECTRON_RENDERER_URL', 'http://localhost:5173/') + + try { + createMainWindow(null, { onRecoveryReloadOutcome }) + crashRenderer() + expect(browserWindowInstance.loadURL).toHaveBeenCalledTimes(2) + + vi.advanceTimersByTime(RENDERER_RECOVERY_DEV_LOAD_TIMEOUT_MS - 1) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + vi.advanceTimersByTime(1) + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 1 }) + ) + } finally { + vi.unstubAllEnvs() + consoleError.mockRestore() + } + }) + + it('stays silent when the stalled window is already closing', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, windowHandlers } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + windowHandlers.close?.({ preventDefault: vi.fn() } as never) + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + + consoleError.mockRestore() + }) + it('keeps the install path out of the outcome breadcrumb', async () => { + const onRecoveryReloadOutcome = vi.fn() + const { consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome }) + crashRenderer() + settleLoad[1]?.reject(new Error(INSTALL_PATH_LOAD_ERROR)) + await vi.advanceTimersByTimeAsync(0) + + const outcome = onRecoveryReloadOutcome.mock.calls[0]?.[0] + expect(outcome).toEqual({ + status: 'failed', + attempt: 1, + elapsedMs: 0, + progress: 'none', + errorCode: 'ERR_FILE_NOT_FOUND' + }) + // sanitizeCrashReportString cannot redact a file:///Users/... URL, so nothing path-shaped may reach the crumb. + expect(JSON.stringify(outcome)).not.toContain('/') + + consoleError.mockRestore() + }) + + it('records a durable breadcrumb for a rejected load, since console output never reaches the bundle', async () => { + const { consoleError, settleLoad } = createHarness() + + createMainWindow(null, {}) + settleLoad[0]?.reject(new Error(INSTALL_PATH_LOAD_ERROR)) + await vi.advanceTimersByTimeAsync(0) + + // Catching the rejection retired the main_unhandled_rejection crumb this used to produce. + expect(recordDurableCrashBreadcrumbMock).toHaveBeenCalledWith('main_window_load_failed', { + errorCode: 'ERR_FILE_NOT_FOUND' + }) + + consoleError.mockRestore() + }) + + it('escalates to the prompt when both attempts are rejected outright', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + settleLoad[1]?.reject(new Error('ERR_CONNECTION_REFUSED (-102)')) + await vi.advanceTimersByTimeAsync(0) + settleLoad[2]?.reject(new Error('ERR_CONNECTION_REFUSED (-102)')) + await vi.advanceTimersByTimeAsync(0) + + expect(onRecoveryReloadOutcome).toHaveBeenLastCalledWith( + expect.objectContaining({ status: 'failed', attempt: 2, errorCode: 'ERR_CONNECTION_REFUSED' }) + ) + expect(onRendererRecoveryExhausted).toHaveBeenCalledWith( + expect.objectContaining({ cause: 'reload-stalled', recentRecoveryCount: 1 }) + ) + + consoleError.mockRestore() + }) + + it('hands the prompt a watched retry so a stalled manual reload re-raises it', () => { + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + // Reload is the dialog's default button; unwatched it returned the user to the same unbounded silent hang. + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(4) + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(5) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(2) + + consoleError.mockRestore() + }) + + it('names the crash-loop cause and gives that prompt a watched retry too', () => { + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRendererRecoveryExhausted }) + for (let attempt = 0; attempt < 4; attempt += 1) { + crashRenderer() + } + + expect(onRendererRecoveryExhausted).toHaveBeenCalledWith( + expect.objectContaining({ cause: 'crash-loop' }) + ) + expect(typeof onRendererRecoveryExhausted.mock.calls[0]?.[0].retry).toBe('function') + + consoleError.mockRestore() + }) + + it('restarts the stall budget when the machine resumes mid-load', () => { + const onRecoveryReloadOutcome = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome }) + crashRenderer() + // Sleep freezes the timer; on wake it would otherwise fire against a load that never got its budget. + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS - 1) + const resume = powerMonitorOnMock.mock.calls.find(([event]) => event === 'resume')?.[1] as ( + ...args: unknown[] + ) => void + resume() + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS - 1) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + + vi.advanceTimersByTime(1) + // Why the full span: rewriting the issue time on resume publishes time-since-wake into the bundle, which is + // silently wrong on any laptop — the outcome crumb exists to be honest about how long the load actually ran. + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ + status: 'timeout', + attempt: 1, + elapsedMs: RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2 - 1 + }) + ) + + consoleError.mockRestore() + }) + + it('never restarts a load that reached a document, and gives it the rest of the cap', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, reachMilestone } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(10_000) + reachMilestone('committed') + + // 'no did-finish-load yet' is not a stall: a cold restart here throws away a load that already committed, and + // a machine that would have landed at ~60s misses the budget entirely. + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + + reachMilestone('dom-ready') + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith({ + status: 'timeout', + attempt: 1, + elapsedMs: RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2, + progress: 'dom-ready' + }) + // Still never restarted, and the cap keeps the ~90s worst case the no-document path already had. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('records a reload that lands after the prompt, and leaves the recovered window alone', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + onRecoveryReloadOutcome.mockClear() + vi.advanceTimersByTime(30_000) + settleLoad[2]?.resolve() + await vi.advanceTimersByTimeAsync(0) + + // Nothing cancels a pending Chromium load, so escalation must keep watching: a bundle that reads + // `exhausted` for a recovery that actually worked misleads the next triage round. + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith({ + status: 'loaded', + attempt: 2, + elapsedMs: RENDERER_RECOVERY_LOAD_TIMEOUT_MS + 30_000, + afterPrompt: true + }) + + // No API dismisses a native message box, so Reload is still aimed at a window that came back; taking it + // would destroy the session the recovery just restored. + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + consoleError.mockRestore() + }) + + it('separates the automatic recovery reload from the prompt-driven retry', () => { + const onBeforeRecoveryReload = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onBeforeRecoveryReload, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + + // The field counts keyed on renderer_recovery_reload mean 'automatic recovery'; a manual retry recorded + // under the same name silently redefines them. + expect(onBeforeRecoveryReload.mock.calls.map(([, trigger]) => trigger)).toEqual([ + 'automatic', + 'automatic', + 'manual-retry' + ]) + + consoleError.mockRestore() + }) + + it('keeps a shutdown-aborted load out of the crash breadcrumb stream', async () => { + const { consoleError, settleLoad } = createHarness() + + createMainWindow(null, {}) + settleLoad[0]?.reject(new Error(`ERR_ABORTED (-3) loading '${DOCUMENT_URL}'`)) + await vi.advanceTimersByTimeAsync(0) + + // A quit or close aborts the in-flight startup load; a healthy shutdown must not look like a launch failure. + expect(recordDurableCrashBreadcrumbMock).not.toHaveBeenCalledWith( + 'main_window_load_failed', + expect.anything() + ) + + consoleError.mockRestore() + }) +}) diff --git a/src/main/window/createMainWindow-renderer-crash-recovery.test.ts b/src/main/window/createMainWindow-renderer-crash-recovery.test.ts index 43dac231130..b3fc210be2c 100644 --- a/src/main/window/createMainWindow-renderer-crash-recovery.test.ts +++ b/src/main/window/createMainWindow-renderer-crash-recovery.test.ts @@ -21,7 +21,7 @@ vi.mock('../browser/browser-client-page-renderer-runtime', async () => { } }) -import { createMainWindow, loadMainWindow } from './createMainWindow' +import { createMainWindow } from './createMainWindow' import { ipcMain } from 'electron' import { shouldRecoverRendererAfterProcessGone } from '../crash-reporting/process-gone-classification' import { @@ -74,8 +74,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -120,8 +120,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -231,8 +231,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -272,8 +272,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) browserWindowMock.mockImplementation(function () { @@ -322,8 +322,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) browserWindowMock.mockImplementation(function () { @@ -353,6 +353,8 @@ describe('createMainWindow', () => { const windowHandlers: Record void> = {} const webContents = { id: 143, + getURL: vi.fn(() => 'file:///opt/orca/renderer/index.html'), + isDestroyed: vi.fn(() => false), on: vi.fn((event, handler) => { windowHandlers[event] = handler }), @@ -374,8 +376,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -415,10 +417,12 @@ describe('createMainWindow', () => { const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) const { browserWindowInstance, windowHandlers } = createRendererRecoveryWindowHarness() const onBeforeRecoveryReload = vi.fn() + const onRendererRecoveryExhausted = vi.fn() withPlatform('win32', () => { createMainWindow(null, { onBeforeRecoveryReload, + onRendererRecoveryExhausted, shouldRecoverRenderer: (details) => shouldRecoverRendererAfterProcessGone({ reason: details.reason, @@ -436,10 +440,14 @@ describe('createMainWindow', () => { {} as never, { reason: 'killed', exitCode: 1 } as Electron.RenderProcessGoneDetails ) - vi.runAllTimers() + vi.advanceTimersByTime(250) - expect(onBeforeRecoveryReload).toHaveBeenCalledWith(143) + expect(onBeforeRecoveryReload).toHaveBeenCalledWith(143, 'automatic') expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + // Why the watchdog must stay quiet here: this reload is deliberate during logoff, and a process that + // outlives the session-end signal must not put a native modal on screen mid-teardown. + vi.runAllTimers() + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() consoleError.mockRestore() }) @@ -615,8 +623,8 @@ describe('createMainWindow', () => { // 1 initial load + 3 recoveries; the 4th crash was refused. expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(4) - // The recovery prompt's Reload button goes straight to loadMainWindow, which the breaker never gates. - loadMainWindow(browserWindowInstance as unknown as Electron.BrowserWindow) + // The recovery prompt's Reload button takes the watched retry, which the breaker never gates. + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(5) // Still-poisoned machine: the next crash re-raises the prompt immediately instead of re-arming auto-reloads. diff --git a/src/main/window/createMainWindow-startup-reveal.test.ts b/src/main/window/createMainWindow-startup-reveal.test.ts index 1a623ab20e3..1200132d319 100644 --- a/src/main/window/createMainWindow-startup-reveal.test.ts +++ b/src/main/window/createMainWindow-startup-reveal.test.ts @@ -57,8 +57,8 @@ describe('createMainWindow', () => { setWindowButtonPosition: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow-system-resume-relay.test.ts b/src/main/window/createMainWindow-system-resume-relay.test.ts index e51f1be0f07..71bf0cd7381 100644 --- a/src/main/window/createMainWindow-system-resume-relay.test.ts +++ b/src/main/window/createMainWindow-system-resume-relay.test.ts @@ -56,8 +56,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return instance diff --git a/src/main/window/createMainWindow-terminal-focus-shortcuts.test.ts b/src/main/window/createMainWindow-terminal-focus-shortcuts.test.ts index 9b88c60800c..8287b9bcecb 100644 --- a/src/main/window/createMainWindow-terminal-focus-shortcuts.test.ts +++ b/src/main/window/createMainWindow-terminal-focus-shortcuts.test.ts @@ -51,8 +51,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -143,8 +143,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -237,8 +237,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -302,8 +302,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -366,8 +366,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -424,8 +424,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -486,8 +486,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -565,8 +565,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -630,8 +630,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -707,8 +707,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -777,8 +777,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow-tray-minimize-close.test.ts b/src/main/window/createMainWindow-tray-minimize-close.test.ts index 14f829b63d9..469d8313830 100644 --- a/src/main/window/createMainWindow-tray-minimize-close.test.ts +++ b/src/main/window/createMainWindow-tray-minimize-close.test.ts @@ -81,8 +81,8 @@ describe('createMainWindow', () => { maximize: vi.fn(), show: vi.fn(), hide: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return instance diff --git a/src/main/window/createMainWindow-zoom-and-tab-switch-shortcuts.test.ts b/src/main/window/createMainWindow-zoom-and-tab-switch-shortcuts.test.ts index d80349a52df..9d0c3cc9048 100644 --- a/src/main/window/createMainWindow-zoom-and-tab-switch-shortcuts.test.ts +++ b/src/main/window/createMainWindow-zoom-and-tab-switch-shortcuts.test.ts @@ -51,8 +51,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -116,8 +116,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -158,8 +158,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -206,8 +206,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -252,8 +252,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -300,8 +300,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -375,8 +375,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -423,8 +423,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -484,8 +484,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow.test.ts b/src/main/window/createMainWindow.test.ts index 18f4dbf5592..79b9a742533 100644 --- a/src/main/window/createMainWindow.test.ts +++ b/src/main/window/createMainWindow.test.ts @@ -78,8 +78,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -140,8 +140,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -320,8 +320,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } }) @@ -379,8 +379,8 @@ describe('createMainWindow', () => { setWindowButtonPosition: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -424,8 +424,8 @@ describe('createMainWindow', () => { setWindowButtonPosition: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -479,8 +479,8 @@ describe('createMainWindow', () => { }), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -554,8 +554,8 @@ describe('createMainWindow', () => { }), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -630,8 +630,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -678,8 +678,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -726,8 +726,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow.ts b/src/main/window/createMainWindow.ts index 085cc475d49..41d84d1ab93 100644 --- a/src/main/window/createMainWindow.ts +++ b/src/main/window/createMainWindow.ts @@ -14,7 +14,8 @@ import { installMainWindowCloseLifecycle, WINDOW_QUIT_RENDERER_ACK_TIMEOUT_MS } from './main-window-close-lifecycle' -import type { CreateMainWindowOptions } from './main-window-contracts' +import type { CreateMainWindowOptions, MainWindowLoadObserver } from './main-window-contracts' +import { mainWindowLoadErrorCode } from './main-window-load-error-code' import { installMainWindowFocusLifecycle } from './main-window-focus-lifecycle' import { installMainWindowShortcutRouting } from './main-window-shortcut-routing' import { installMainWindowStateLifecycle } from './main-window-state-lifecycle' @@ -33,12 +34,25 @@ import { installWindowsPathRegistryChangeListener } from '../pty/windows-path-re export { WINDOW_QUIT_RENDERER_ACK_TIMEOUT_MS } -export function loadMainWindow(mainWindow: BrowserWindow): void { - if (is.dev && process.env.ELECTRON_RENDERER_URL) { - void mainWindow.loadURL(process.env.ELECTRON_RENDERER_URL) - } else { - void mainWindow.loadFile(join(__dirname, '../renderer/index.html')) - } +export function loadMainWindow(mainWindow: BrowserWindow, observer?: MainWindowLoadObserver): void { + const load = + is.dev && process.env.ELECTRON_RENDERER_URL + ? mainWindow.loadURL(process.env.ELECTRON_RENDERER_URL) + : mainWindow.loadFile(join(__dirname, '../renderer/index.html')) + // Observe each load promise so failures cannot leave recovery waiting silently. + load.then( + () => observer?.onLoaded?.(), + (cause: unknown) => { + const error = cause instanceof Error ? cause : new Error(String(cause)) + const errorCode = mainWindowLoadErrorCode(error) + // Keep durable diagnostics path-free and exclude shutdown/navigation aborts. + if (!mainWindow.isDestroyed() && errorCode !== 'ERR_ABORTED') { + recordDurableCrashBreadcrumb('main_window_load_failed', { errorCode }) + } + console.error('[window] Main window load failed', error) + observer?.onError?.(error) + } + ) } export function createMainWindow( @@ -158,8 +172,9 @@ export function createMainWindow( } forceRepaint(mainWindow) mainWindow.webContents.send('system:resumed') + // Give a suspended recovery load its full budget on wake. + focus.notifySystemResume() } - powerMonitor.on('resume', onSystemResume) const state = installMainWindowStateLifecycle({ mainWindow, @@ -172,9 +187,11 @@ export function createMainWindow( isWindowClosing: state.isWindowClosing, mainWindow, opts, - reloadMainWindow: () => loadMainWindow(mainWindow), + reloadMainWindow: (observer) => loadMainWindow(mainWindow, observer), rendererWebContentsId }) + // Register after focus is initialized because the resume callback uses it. + powerMonitor.on('resume', onSystemResume) installMainWindowShortcutRouting({ focus, mainWindow, opts, store }) const closeLifecycle = installMainWindowCloseLifecycle({ focus, diff --git a/src/main/window/main-window-contracts.ts b/src/main/window/main-window-contracts.ts index ce5c6cfe0b2..5135be0fbe6 100644 --- a/src/main/window/main-window-contracts.ts +++ b/src/main/window/main-window-contracts.ts @@ -1,4 +1,15 @@ import type { KeybindingOverrides } from '../../shared/keybindings' +import type { + RecoveryExhaustionCause, + RecoveryReloadMilestone, + RecoveryReloadTrigger +} from './renderer-recovery-reload-watchdog' + +/** Per-load outcome from Electron's load promise, which is scoped to that one load unlike `did-finish-load`. */ +export type MainWindowLoadObserver = { + onLoaded?: () => void + onError?: (error: Error) => void +} export type CreateMainWindowOptions = { /** Returns true when a manual app.quit() (Cmd+Q) is in progress, so the renderer skips the running-process confirm dialog. */ @@ -14,11 +25,14 @@ export type CreateMainWindowOptions = { details: Electron.RenderProcessGoneDetails, webContentsId: number ) => boolean - /** Called when consecutive auto-recoveries hit the circuit-breaker limit so the host can prompt instead of crash-looping. */ + /** Called when auto-recovery gives up — the breaker opened, or the recovery reload never produced a document. */ onRendererRecoveryExhausted?: (info: { details: Electron.RenderProcessGoneDetails webContentsId: number recentRecoveryCount: number + cause?: RecoveryExhaustionCause + /** Watched manual retry for the recovery prompt; an unwatched one cannot re-raise the prompt when it stalls too. */ + retry?: () => void }) => void /** Defer renderer load until IPC handlers are registered, or eager renderer calls race into missing channels. */ deferLoad?: boolean @@ -27,6 +41,24 @@ export type CreateMainWindowOptions = { title?: string getKeybindings?: () => KeybindingOverrides | undefined onBeforeReload?: (options: { ignoreCache: boolean; webContentsId: number }) => void - /** Marks the in-place recovery reload so did-finish-load's PTY orphan sweep spares live sessions until restore re-attaches (#5787). */ - onBeforeRecoveryReload?: (webContentsId: number) => void + /** + * Marks the in-place recovery reload so did-finish-load's PTY orphan sweep spares live sessions until restore + * re-attaches (#5787). The prompt's manual Reload is one too, so `trigger` keeps the automatic-recovery + * breadcrumb counting only automatic recoveries. + */ + onBeforeRecoveryReload?: (webContentsId: number, trigger: RecoveryReloadTrigger) => void + /** Pairs an outcome with the recovery-reload intent crumb: bundles could not tell a landed reload from a stalled one. */ + onRecoveryReloadOutcome?: (outcome: { + status: 'loaded' | 'timeout' | 'failed' + attempt: number + elapsedMs: number + /** How far the load got: 'none' is the blank-window field failure, anything else a document that then hung. */ + progress?: RecoveryReloadMilestone + /** True when the load landed after the recovery prompt was already raised — the recovery worked. */ + afterPrompt?: boolean + /** True when a later navigation replaced this load: elapsedMs then measures the replacement, not the reload. */ + superseded?: boolean + /** `ERR_*` code only, for the same reason — Electron's load-error message embeds the URL. */ + errorCode?: string + }) => void } diff --git a/src/main/window/main-window-focus-lifecycle.ts b/src/main/window/main-window-focus-lifecycle.ts index a021d464d15..d494992e692 100644 --- a/src/main/window/main-window-focus-lifecycle.ts +++ b/src/main/window/main-window-focus-lifecycle.ts @@ -14,13 +14,14 @@ import { matchingRichMarkdownContextMenuTableTarget, parseRichMarkdownContextMenuTableTarget } from './editable-context-menu' -import type { CreateMainWindowOptions } from './main-window-contracts' +import type { CreateMainWindowOptions, MainWindowLoadObserver } from './main-window-contracts' import { browserRouteWebContentsRegistry } from '../browser/browser-route-session-runtime' import { attachBrowserClientPageRenderer, retireBrowserClientPageRenderer } from '../browser/browser-client-page-renderer-runtime' import { registerRendererDocumentNavigation } from './renderer-document-navigation' +import { createRendererRecoveryReloadWatchdog } from './renderer-recovery-reload-watchdog' export type MainWindowFocusLifecycle = { dispose: () => void @@ -30,13 +31,15 @@ export type MainWindowFocusLifecycle = { isRendererProcessGone: () => boolean isShortcutRecorderFocused: () => boolean isTerminalInputFocused: () => boolean + /** Relays powerMonitor 'resume' so a suspend-frozen recovery-reload timer does not fire against an unbudgeted load. */ + notifySystemResume: () => void } export function installMainWindowFocusLifecycle(args: { isWindowClosing: () => boolean mainWindow: BrowserWindow opts?: CreateMainWindowOptions - reloadMainWindow: () => void + reloadMainWindow: (observer: MainWindowLoadObserver) => void rendererWebContentsId: number }): MainWindowFocusLifecycle { const { isWindowClosing, mainWindow, opts, reloadMainWindow, rendererWebContentsId } = args @@ -162,6 +165,16 @@ export function installMainWindowFocusLifecycle(args: { rendererRecoveryTimer = null } } + // Why: the reload can stall with a live window and no document — no did-fail-load fires, and the breaker counts + // renderer deaths, so a load that never lands is invisible to every other observer on this path. + const recoveryReloadWatchdog = createRendererRecoveryReloadWatchdog({ + isRecoveryPending: () => rendererRecoveryTimer !== null, + isWindowClosing, + mainWindow, + opts, + reloadMainWindow, + rendererWebContentsId + }) const scheduleRendererRecovery = (details: Electron.RenderProcessGoneDetails): void => { if ( rendererRecoveryTimer || @@ -187,17 +200,16 @@ export function installMainWindowFocusLifecycle(args: { const recovery = rendererRecoveryCircuitBreaker.registerRecoveryAttempt(Date.now()) if (!recovery.allowed) { // Why: too many reloads means it will just crash again; stop and let the host surface a recovery prompt. - opts?.onRendererRecoveryExhausted?.({ - details, - webContentsId: rendererWebContentsId, - recentRecoveryCount: recovery.recentRecoveryCount - }) + // Why through the watchdog: it owns the one-prompt-at-a-time guard, and the prompt's manual retry is a + // recovery reload too — unwatched, one that stalls leaves a blank window and no further prompt. + recoveryReloadWatchdog.escalate( + { details, recentRecoveryCount: recovery.recentRecoveryCount }, + 'crash-loop' + ) return } // Why: a transient renderer/Network Service loss can blank Chromium; reload the app document once to recover. - // Why: mark this in-place reload so the did-finish-load orphan sweep spares live PTYs until session restore (#5787). - opts?.onBeforeRecoveryReload?.(mainWindow.webContents.id) - reloadMainWindow() + recoveryReloadWatchdog.issue(details, recovery.recentRecoveryCount) }, 250) } mainWindow.webContents.on('render-process-gone', (_event, details) => { @@ -229,6 +241,7 @@ export function installMainWindowFocusLifecycle(args: { rendererProcessGone = false attachBrowserClientPageRenderer(rendererWebContents) clearRendererRecoveryTimer() + recoveryReloadWatchdog.notifyDocumentLoaded() }) const dispose = (): void => { @@ -237,6 +250,7 @@ export function installMainWindowFocusLifecycle(args: { resetFloatingTerminalInputFocus() resetShortcutRecorderFocus() clearRendererRecoveryTimer() + recoveryReloadWatchdog.clear() ipcMain.removeListener(markdownFocusChannel, onMarkdownEditorFocused) ipcMain.removeListener(terminalInputFocusChannel, onTerminalInputFocused) ipcMain.removeListener(floatingFocusChannel, onFloatingFocus) @@ -250,6 +264,7 @@ export function installMainWindowFocusLifecycle(args: { isMarkdownEditorFocused: () => markdownEditorFocused, isRendererProcessGone: () => rendererProcessGone, isShortcutRecorderFocused: () => shortcutRecorderFocused, - isTerminalInputFocused: () => terminalInputFocused + isTerminalInputFocused: () => terminalInputFocused, + notifySystemResume: recoveryReloadWatchdog.notifySystemResume } } diff --git a/src/main/window/main-window-load-error-code.ts b/src/main/window/main-window-load-error-code.ts new file mode 100644 index 00000000000..6dd9a79e6d8 --- /dev/null +++ b/src/main/window/main-window-load-error-code.ts @@ -0,0 +1,12 @@ +// Record only the ERR_* code: Electron error messages embed private install URLs. +export function mainWindowLoadErrorCode(error: unknown): string { + const code = + typeof error === 'object' && error !== null && 'code' in error && typeof error.code === 'string' + ? error.code + : undefined + if (code && /^ERR_[A-Z0-9_]+$/.test(code)) { + return code + } + const message = error instanceof Error ? error.message : String(error) + return /\bERR_[A-Z0-9_]+/.exec(message)?.[0] ?? 'unknown' +} diff --git a/src/main/window/main-window-webview-security.ts b/src/main/window/main-window-webview-security.ts index da6e4159a58..a662449eb58 100644 --- a/src/main/window/main-window-webview-security.ts +++ b/src/main/window/main-window-webview-security.ts @@ -111,7 +111,7 @@ export function installMainWindowWebviewSecurity(mainWindow: BrowserWindow): voi mainWindow.webContents.on('did-attach-webview', (_event, guest) => { if (isDocPreviewSession(guest.session)) { - // Why: preview guests never join browser-tab routing, popups or anti-detection; the + // Why: preview guests never join browser-tab routing, popups or auth-identity tracking; the // workspace-doc profile is what refuses all three. The attach is also the point a live window // exists to receive read failures for that guest. setDocPreviewFailureSink(mainWindow.webContents) diff --git a/src/main/window/renderer-recovery-prompt.test.ts b/src/main/window/renderer-recovery-prompt.test.ts index d5bd7ce11c1..a26700720eb 100644 --- a/src/main/window/renderer-recovery-prompt.test.ts +++ b/src/main/window/renderer-recovery-prompt.test.ts @@ -1,11 +1,14 @@ import type { MessageBoxOptions, MessageBoxReturnValue } from 'electron' -import { describe, expect, it, vi } from 'vitest' +import { beforeEach, afterEach, describe, expect, it, vi } from 'vitest' +import { ensureMainI18n, mainI18n } from '../i18n/main-i18n' import type { InstallDirAclPoisonDiagnosis } from '../startup/windows-install-dir-acl-recovery' import { presentRendererRecoveryPrompt, type RendererRecoveryPromptDeps } from './renderer-recovery-prompt' +vi.mock('electron', () => ({ app: { getLocale: () => 'en-US' } })) + const POISON: InstallDirAclPoisonDiagnosis = { detail: "Windows permissions on Orca's install folder are blocking its own sandboxed processes.", commands: ['icacls "C:\\Orca" /grant "*S-1-15-2-2:(OI)(CI)(RX)"', 'icacls "C:\\Orca" /grant b'] @@ -43,17 +46,60 @@ function harness(overrides: Partial & { responses?: } describe('presentRendererRecoveryPrompt', () => { + beforeEach(async () => { + await ensureMainI18n() + await mainI18n.changeLanguage('en') + }) + + afterEach(() => { + mainI18n.removeResourceBundle('en', 'translation') + }) + + it('interpolates the recovery count', async () => { + const { run, shown } = harness({ recentRecoveryCount: 7 }) + await run() + expect(shown[0].detail).toContain('Orca tried to recover 7 times in a row') + expect(shown[0].detail).not.toContain('{{') + }) + + it.each([ + { responses: [1, 0], reloads: 1, quits: 0 }, + { responses: [1, 2], reloads: 0, quits: 1 } + ])( + 'dispatches translated buttons by response index: $responses', + async ({ responses, reloads, quits }) => { + mainI18n.addResourceBundle('en', 'translation', { + rendererRecovery: { reload: 'Recharger', copyCommands: 'Copier', quit: 'Quitter' } + }) + const { run, shown, copied, reload, quit } = harness({ diagnose: () => POISON, responses }) + await run() + expect(shown[0].buttons).toEqual(['Recharger', 'Copier', 'Quitter']) + expect(copied).toEqual([POISON.commands.join('\r\n')]) + expect(reload).toHaveBeenCalledTimes(reloads) + expect(quit).toHaveBeenCalledTimes(quits) + } + ) + it('offers reload and quit with the generic cause when nothing is diagnosed', async () => { const { run, shown, reload, quit } = harness({ responses: [0] }) await run() expect(shown).toHaveLength(1) expect(shown[0].buttons).toEqual(['Reload', 'Quit']) - expect(shown[0].cancelId).toBe(1) + // Escape lands on cancelId, and this box is window-modal over the window it is about: it must not quit. + expect(shown[0].cancelId).toBe(0) expect(shown[0].detail).toContain('graphics-driver or installation problem') expect(reload).toHaveBeenCalledOnce() expect(quit).not.toHaveBeenCalled() }) + it('names the stalled reload instead of claiming a repeated crash', async () => { + const { run, shown } = harness({ failure: 'reload-stalled', responses: [1] }) + await run() + expect(shown[0].message).toContain('stopped responding while reloading') + expect(shown[0].detail).toContain('never finished loading') + expect(shown[0].detail).not.toContain('times in a row') + }) + it('quits on the last button', async () => { const { run, reload, quit } = harness({ responses: [1] }) await run() @@ -65,7 +111,7 @@ describe('presentRendererRecoveryPrompt', () => { const { run, shown } = harness({ diagnose: () => POISON, responses: [0] }) await run() expect(shown[0].buttons).toEqual(['Reload', 'Copy Commands', 'Quit']) - expect(shown[0].cancelId).toBe(2) + expect(shown[0].cancelId).toBe(0) expect(shown[0].detail).toContain(POISON.detail) expect(shown[0].detail).toContain('graphics driver') }) diff --git a/src/main/window/renderer-recovery-prompt.ts b/src/main/window/renderer-recovery-prompt.ts index 2026d1f10b5..18ab02a8eca 100644 --- a/src/main/window/renderer-recovery-prompt.ts +++ b/src/main/window/renderer-recovery-prompt.ts @@ -1,19 +1,13 @@ import type { MessageBoxOptions, MessageBoxReturnValue } from 'electron' +import { translateMain } from '../i18n/main-i18n' import type { InstallDirAclPoisonDiagnosis } from '../startup/windows-install-dir-acl-recovery' +import type { RecoveryExhaustionCause } from './renderer-recovery-reload-watchdog' -/** - * The dialog shown when the renderer crash-loop breaker opens: the window is - * blank by then, so this is the only retry/quit surface the user has. - */ - -const GENERIC_DETAIL = - 'This is often a graphics-driver or installation problem. Reload to try again, or quit and relaunch Orca.' -// Why keep it alongside the ACL diagnosis: the probe cannot name-check every -// locale, so a driver crash on a healthy install must not lose its only hint. -const DRIVER_FALLBACK = 'If that does not help, the cause is usually a graphics driver.' +export type RendererRecoveryPromptFailure = RecoveryExhaustionCause export type RendererRecoveryPromptDeps = { recentRecoveryCount: number + failure?: RendererRecoveryPromptFailure isQuitting: () => boolean diagnose: () => InstallDirAclPoisonDiagnosis | null showMessageBox: (options: MessageBoxOptions) => Promise @@ -25,29 +19,59 @@ export type RendererRecoveryPromptDeps = { export async function presentRendererRecoveryPrompt( deps: RendererRecoveryPromptDeps ): Promise { - // Why a loop: copying the commands must not dismiss the only surface offering them. + const stalled = deps.failure === 'reload-stalled' + // Copying must preserve the only available recovery surface. while (!deps.isQuitting()) { const diagnosis = deps.diagnose() - const buttons = diagnosis ? ['Reload', 'Copy Commands', 'Quit'] : ['Reload', 'Quit'] + const buttons = [translateMain('rendererRecovery.reload', 'Reload')] + if (diagnosis) { + buttons.push(translateMain('rendererRecovery.copyCommands', 'Copy Commands')) + } + buttons.push(translateMain('rendererRecovery.quit', 'Quit')) + const recoveryDetail = stalled + ? translateMain( + 'rendererRecovery.stalledDetail', + 'Orca reloaded the window after a crash, but it never finished loading.' + ) + : translateMain( + 'rendererRecovery.crashLoopDetail', + 'Orca tried to recover {{recoveryCount}} times in a row without success.', + { recoveryCount: deps.recentRecoveryCount } + ) + const causeDetail = diagnosis + ? `${diagnosis.detail}\n\n${translateMain( + 'rendererRecovery.driverFallback', + 'If that does not help, the cause is usually a graphics driver.' + )}` + : translateMain( + 'rendererRecovery.genericDetail', + 'This is often a graphics-driver or installation problem. Reload to try again, or quit and relaunch Orca.' + ) const { response } = await deps.showMessageBox({ type: 'error', buttons, defaultId: 0, - cancelId: buttons.length - 1, - title: 'Orca keeps failing to load', - message: 'The app window crashed repeatedly and stopped reloading automatically.', - detail: `Orca tried to recover ${deps.recentRecoveryCount} times in a row without success.\n\n${ - diagnosis ? `${diagnosis.detail}\n\n${DRIVER_FALLBACK}` : GENERIC_DETAIL - }` + // Escape retries instead of destroying the session. + cancelId: 0, + title: translateMain('rendererRecovery.title', 'Orca keeps failing to load'), + message: stalled + ? translateMain( + 'rendererRecovery.stalledMessage', + 'The app window stopped responding while reloading after a crash.' + ) + : translateMain( + 'rendererRecovery.crashLoopMessage', + 'The app window crashed repeatedly and stopped reloading automatically.' + ), + detail: `${recoveryDetail}\n\n${causeDetail}` }) - const choice = buttons[response] - if (choice === 'Copy Commands' && diagnosis) { + if (response === 1 && diagnosis) { deps.copyToClipboard(diagnosis.commands.join('\r\n')) continue } - if (choice === 'Reload') { + if (response === 0) { deps.reload() - } else if (choice === 'Quit') { + } else if (response === buttons.length - 1) { deps.quit() } return diff --git a/src/main/window/renderer-recovery-reload-watchdog.test.ts b/src/main/window/renderer-recovery-reload-watchdog.test.ts new file mode 100644 index 00000000000..b6e854bbbf1 --- /dev/null +++ b/src/main/window/renderer-recovery-reload-watchdog.test.ts @@ -0,0 +1,136 @@ +import { EventEmitter } from 'node:events' +import type { BrowserWindow } from 'electron' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { MainWindowLoadObserver } from './main-window-contracts' +import { + createRendererRecoveryReloadWatchdog, + RENDERER_RECOVERY_LOAD_TIMEOUT_MS +} from './renderer-recovery-reload-watchdog' + +vi.mock('@electron-toolkit/utils', () => ({ is: { dev: false } })) + +function createHarness() { + const webContents = Object.assign(new EventEmitter(), { id: 143 }) + const mainWindow = { webContents, isDestroyed: () => false } as unknown as BrowserWindow + const loads: MainWindowLoadObserver[] = [] + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const watchdog = createRendererRecoveryReloadWatchdog({ + mainWindow, + rendererWebContentsId: webContents.id, + isRecoveryPending: () => false, + isWindowClosing: () => false, + reloadMainWindow: (observer) => loads.push(observer), + opts: { onRecoveryReloadOutcome, onRendererRecoveryExhausted } + }) + const abortLatestLoad = () => loads.at(-1)?.onError?.(new Error('ERR_ABORTED (-3)')) + watchdog.issue({ reason: 'crashed', exitCode: 5 }, 1) + return { + watchdog, + webContents, + loads, + abortLatestLoad, + onRecoveryReloadOutcome, + onRendererRecoveryExhausted + } +} + +describe('superseding recovery navigations', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => vi.useRealTimers()) + + it('removes listeners and the pending stall timer during teardown', () => { + const { watchdog, webContents, loads, onRecoveryReloadOutcome } = createHarness() + expect(webContents.eventNames().sort()).toEqual(['did-fail-load', 'did-navigate', 'dom-ready']) + expect(vi.getTimerCount()).toBe(1) + watchdog.clear() + expect(webContents.eventNames()).toEqual([]) + expect(vi.getTimerCount()).toBe(0) + loads[0]?.onLoaded?.() + loads[0]?.onError?.(new Error('ERR_FILE_NOT_FOUND')) + watchdog.notifySystemResume() + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(vi.getTimerCount()).toBe(0) + }) + + it('does not mistake a replacement error page for recovery', () => { + const { + watchdog, + webContents, + abortLatestLoad, + onRecoveryReloadOutcome, + onRendererRecoveryExhausted + } = createHarness() + abortLatestLoad() + webContents.emit('did-navigate') + webContents.emit('did-fail-load', {}, -6, 'ERR_FILE_NOT_FOUND', 'file:///missing', true) + watchdog.notifyDocumentLoaded() + + expect(onRecoveryReloadOutcome).not.toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded' }) + ) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'failed', errorCode: 'ERR_FILE_NOT_FOUND' }) + ) + watchdog.clear() + }) + + it('ignores subframe failures and aborted replacement navigations', () => { + const { watchdog, webContents, abortLatestLoad, onRecoveryReloadOutcome } = createHarness() + abortLatestLoad() + webContents.emit('did-fail-load', {}, -6, 'ERR_FILE_NOT_FOUND', 'file:///missing', false) + webContents.emit('did-fail-load', {}, -3, 'ERR_ABORTED', 'file:///previous', true) + watchdog.notifyDocumentLoaded() + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded', superseded: true }) + ) + watchdog.clear() + }) + + it('keeps Reload available if the replacement fails beneath an existing prompt', () => { + const { + watchdog, + webContents, + loads, + abortLatestLoad, + onRecoveryReloadOutcome, + onRendererRecoveryExhausted + } = createHarness() + abortLatestLoad() + webContents.emit('did-navigate') + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + webContents.emit('did-fail-load', {}, -6, 'ERR_FILE_NOT_FOUND', 'file:///missing', true) + watchdog.notifyDocumentLoaded() + expect(onRecoveryReloadOutcome).not.toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded' }) + ) + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(loads).toHaveLength(2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + watchdog.clear() + }) + + it('recognizes a successful replacement started after the stall prompt', () => { + const { + watchdog, + loads, + abortLatestLoad, + onRecoveryReloadOutcome, + onRendererRecoveryExhausted + } = createHarness() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + abortLatestLoad() + watchdog.notifyDocumentLoaded() + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded', superseded: true, afterPrompt: true }) + ) + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(loads).toHaveLength(2) + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + watchdog.clear() + }) +}) diff --git a/src/main/window/renderer-recovery-reload-watchdog.ts b/src/main/window/renderer-recovery-reload-watchdog.ts new file mode 100644 index 00000000000..1295ca13ac4 --- /dev/null +++ b/src/main/window/renderer-recovery-reload-watchdog.ts @@ -0,0 +1,310 @@ +import { is } from '@electron-toolkit/utils' +import type { BrowserWindow } from 'electron' +import { isSystemSessionEnding } from '../crash-reporting/expected-teardown-state' +import type { CreateMainWindowOptions, MainWindowLoadObserver } from './main-window-contracts' +import { mainWindowLoadErrorCode } from './main-window-load-error-code' + +// Field recoveries took up to 30.4s; allow 45s before retrying a load with no document. +export const RENDERER_RECOVERY_LOAD_TIMEOUT_MS = 45_000 +// Vite cold starts need a longer budget than packaged files. +export const RENDERER_RECOVERY_DEV_LOAD_TIMEOUT_MS = 180_000 +// Retry once before handing recovery back to the user. +const RENDERER_RECOVERY_LOAD_ATTEMPTS = 2 +// Milestones may extend the budget, but cannot postpone the prompt indefinitely. +const RENDERER_RECOVERY_LOAD_CAP_FACTOR = 2 + +/** Automatic recovery vs the prompt's manual Reload; they must not share one breadcrumb name. */ +export type RecoveryReloadTrigger = 'automatic' | 'manual-retry' + +/** How far a load got. Ranked, so an attempt's milestone only ever moves forward. */ +export type RecoveryReloadMilestone = 'none' | 'committed' | 'dom-ready' +const MILESTONE_RANK: Record = { + none: 0, + committed: 1, + 'dom-ready': 2 +} + +export type RecoveryExhaustionCause = 'crash-loop' | 'reload-stalled' + +export type RendererRecoveryReloadWatchdog = { + /** Issues a recovery reload and arms the stall watchdog. */ + issue: ( + details: Electron.RenderProcessGoneDetails, + recentRecoveryCount: number, + trigger?: RecoveryReloadTrigger + ) => void + /** Raises the recovery prompt at most once: a native message box cannot be dismissed, so a second one stacks. */ + escalate: (subject: RecoveryPromptSubject, cause: RecoveryExhaustionCause) => void + /** + * A main-frame document finished loading. Only an attempt whose load was superseded takes this as its outcome; + * every other attempt settles through its own load promise, which an error page or a later navigation cannot fool. + */ + notifyDocumentLoaded: () => void + /** Restarts the stall budget after a suspend froze the timer mid-load. */ + notifySystemResume: () => void + clear: () => void +} + +type RecoveryReload = { + attempt: number + details: Electron.RenderProcessGoneDetails + recentRecoveryCount: number + /** Never rewritten: the elapsedMs a crash bundle reads has to stay time-since-issue. */ + issuedAt: number + /** Absolute deadline. A suspend pushes it out; a milestone cannot. */ + capAt: number + milestone: RecoveryReloadMilestone + progressedSinceArm: boolean + /** Chromium aborted this load for a later navigation, which now owns the outcome. */ + superseded: boolean +} + +type RecoveryReloadSeed = Pick +/** What a raised prompt is about; the crash-loop breaker has no attempt to hand over, only the crash. */ +export type RecoveryPromptSubject = Pick + +/** Bounds stalled recovery reloads while still observing success after escalation. */ +export function createRendererRecoveryReloadWatchdog(args: { + /** True when a renderer death has already queued its own recovery, which then owns the next load. */ + isRecoveryPending: () => boolean + isWindowClosing: () => boolean + mainWindow: BrowserWindow + opts?: CreateMainWindowOptions + reloadMainWindow: (observer: MainWindowLoadObserver) => void + rendererWebContentsId: number +}): RendererRecoveryReloadWatchdog { + const { + isRecoveryPending, + isWindowClosing, + mainWindow, + opts, + reloadMainWindow, + rendererWebContentsId + } = args + // Cache before teardown: accessing a destroyed window's webContents throws. + const rendererWebContents = mainWindow.webContents + let inFlight: RecoveryReload | null = null + // Retain timed-out loads so a late success can disarm the prompt's Reload. + let latest: RecoveryReload | null = null + // Keep one prompt until answered; native message boxes cannot be dismissed programmatically. + let prompt: RecoveryPromptSubject | null = null + let documentLanded = false + let timer: ReturnType | null = null + + const clearTimer = (): void => { + if (timer) { + clearTimeout(timer) + timer = null + } + } + // Match loadMainWindow's dev/prod branch. + const timeoutMs = (): number => + is.dev && process.env.ELECTRON_RENDERER_URL + ? RENDERER_RECOVERY_DEV_LOAD_TIMEOUT_MS + : RENDERER_RECOVERY_LOAD_TIMEOUT_MS + + const armTimer = (reload: RecoveryReload): void => { + clearTimer() + reload.progressedSinceArm = false + timer = setTimeout( + () => onBudgetExpired(reload), + Math.max(0, Math.min(timeoutMs(), reload.capAt - Date.now())) + ) + timer.unref?.() + } + + const onBudgetExpired = (reload: RecoveryReload): void => { + if (inFlight !== reload) { + return + } + // Give a progressing load the remaining budget instead of restarting it cold. + if (reload.progressedSinceArm && Date.now() < reload.capAt) { + armTimer(reload) + return + } + fail(reload) + } + + const start = (seed: RecoveryReloadSeed, trigger: RecoveryReloadTrigger): void => { + const issuedAt = Date.now() + const reload: RecoveryReload = { + ...seed, + issuedAt, + capAt: issuedAt + timeoutMs() * RENDERER_RECOVERY_LOAD_CAP_FACTOR, + milestone: 'none', + progressedSinceArm: false, + superseded: false + } + inFlight = reload + latest = reload + documentLanded = false + // Preserve live PTYs until renderer session restore (#5787). + opts?.onBeforeRecoveryReload?.(mainWindow.webContents.id, trigger) + // Only this load's promise distinguishes success from stale events and error pages. + reloadMainWindow({ + onLoaded: () => settleLoaded(reload), + onError: (error) => onLoadRejected(reload, mainWindowLoadErrorCode(error)) + }) + armTimer(reload) + } + + const settleLoaded = (reload: RecoveryReload): void => { + // A replaced attempt's promise may resolve on the replacement document. + if (reload !== latest) { + return + } + latest = null + documentLanded = true + if (reload === inFlight) { + inFlight = null + clearTimer() + } + opts?.onRecoveryReloadOutcome?.({ + status: 'loaded', + attempt: reload.attempt, + elapsedMs: Math.max(0, Date.now() - reload.issuedAt), + // Record late recovery even if the prompt has already appeared. + ...(prompt ? { afterPrompt: true } : {}), + // Replacement timings must be excluded from recovery-load budget analysis. + ...(reload.superseded ? { superseded: true } : {}) + }) + } + + // ERR_ABORTED transfers ownership to a replacement; the cap still bounds a silent replacement. + const onLoadRejected = (reload: RecoveryReload, errorCode: string): void => { + if (errorCode !== 'ERR_ABORTED') { + fail(reload, errorCode) + return + } + if (latest !== reload) { + return + } + reload.superseded = true + if (inFlight === reload) { + armTimer(reload) + } + } + + const retryFrom = (subject: RecoveryPromptSubject): void => { + prompt = null + // A late recovery makes the prompt's Reload unnecessary. + if (documentLanded) { + return + } + start( + { attempt: 1, details: subject.details, recentRecoveryCount: subject.recentRecoveryCount }, + 'manual-retry' + ) + } + + const escalate = (subject: RecoveryPromptSubject, cause: RecoveryExhaustionCause): void => { + // A new crash invalidates any document that landed while the prompt was open. + documentLanded = false + if (prompt) { + return + } + prompt = subject + opts?.onRendererRecoveryExhausted?.({ + details: subject.details, + webContentsId: rendererWebContentsId, + recentRecoveryCount: subject.recentRecoveryCount, + cause, + // Watch manual retries too, so another stall can offer recovery again. + retry: () => retryFrom(subject) + }) + } + + const fail = (reload: RecoveryReload, errorCode?: string): void => { + // Only the live attempt owns a failure verdict. + if (inFlight !== reload) { + return + } + // Suppress shutdown verdicts; resume may re-arm the retained attempt. + if ( + isWindowClosing() || + opts?.getIsQuitting?.() || + mainWindow.isDestroyed() || + isSystemSessionEnding() + ) { + return + } + inFlight = null + clearTimer() + opts?.onRecoveryReloadOutcome?.({ + status: errorCode === undefined ? 'timeout' : 'failed', + attempt: reload.attempt, + // Wall-clock changes must not produce negative diagnostic durations. + elapsedMs: Math.max(0, Date.now() - reload.issuedAt), + progress: reload.milestone, + ...(errorCode === undefined ? {} : { errorCode }) + }) + // A pending prompt or crash recovery owns the next reload. + if (prompt || isRecoveryPending()) { + return + } + // Restart only loads with no document; preserve progress until the user chooses Reload. + if (reload.attempt < RENDERER_RECOVERY_LOAD_ATTEMPTS && reload.milestone === 'none') { + start({ ...reload, attempt: reload.attempt + 1 }, 'automatic') + return + } + escalate(reload, 'reload-stalled') + } + + // Commit and DOM-ready distinguish a blank load from a document still loading. + const observeMilestone = (milestone: RecoveryReloadMilestone) => (): void => { + if (!inFlight || MILESTONE_RANK[milestone] <= MILESTONE_RANK[inFlight.milestone]) { + return + } + inFlight.milestone = milestone + inFlight.progressedSinceArm = true + } + const onDidNavigate = observeMilestone('committed') + const onDomReady = observeMilestone('dom-ready') + const onDidFailLoad = ( + _event: Electron.Event, + errorCode: number, + errorDescription: string, + _validatedURL: string, + isMainFrame: boolean + ): void => { + if (!isMainFrame || errorCode === -3 || !latest?.superseded) { + return + } + // Error documents also finish loading; only a successful replacement may settle an aborted attempt. + latest.superseded = false + documentLanded = false + fail(latest, mainWindowLoadErrorCode(new Error(errorDescription))) + } + rendererWebContents.on('did-navigate', onDidNavigate) + rendererWebContents.on('dom-ready', onDomReady) + rendererWebContents.on('did-fail-load', onDidFailLoad) + + return { + issue: (details, recentRecoveryCount, trigger = 'automatic') => + start({ attempt: 1, details, recentRecoveryCount }, trigger), + escalate, + notifyDocumentLoaded: () => { + // Timed-out replacements can still recover beneath the prompt. + if (latest?.superseded) { + settleLoaded(latest) + } + }, + // Restore the budget after sleep without rewriting the diagnostic issue time. + notifySystemResume: () => { + if (!inFlight) { + return + } + inFlight.capAt = Date.now() + timeoutMs() * RENDERER_RECOVERY_LOAD_CAP_FACTOR + armTimer(inFlight) + }, + clear: () => { + inFlight = null + latest = null + prompt = null + clearTimer() + rendererWebContents.off?.('did-navigate', onDidNavigate) + rendererWebContents.off?.('dom-ready', onDomReady) + rendererWebContents.off?.('did-fail-load', onDidFailLoad) + } + } +} diff --git a/src/main/windows-descendant-exit-verification.test.ts b/src/main/windows-descendant-exit-verification.test.ts new file mode 100644 index 00000000000..392c44399e7 --- /dev/null +++ b/src/main/windows-descendant-exit-verification.test.ts @@ -0,0 +1,175 @@ +import { describe, expect, it, vi } from 'vitest' +import { + captureWindowsDescendantSnapshot, + terminateIdentifiedWindowsProcessTree, + verifyWindowsDescendantSnapshotExit, + type WindowsDescendantSnapshot +} from './windows-descendant-exit-verification' + +function snapshot( + descendants: { pid: number; creationTimeMs: number }[], + unidentifiedCount = 0 +): WindowsDescendantSnapshot { + return { + root: { pid: 100, creationTimeMs: 5 }, + descendants, + unidentifiedCount, + capturedAtMs: 1_700_000_000_000 + } +} + +describe('captureWindowsDescendantSnapshot', () => { + it('walks the whole subtree and keeps only rows a later read can re-identify', async () => { + const captured = await captureWindowsDescendantSnapshot(100, { + // 400 is a grandchild; 300 denied a creation-time query, so no later read + // could tell it from a recycled pid and signalling it would risk a stranger. + readTable: vi.fn(async () => [ + { pid: 100, ppid: 1, creationTimeMs: 5 }, + { pid: 200, ppid: 100, creationTimeMs: 7 }, + { pid: 300, ppid: 100 }, + { pid: 400, ppid: 200, creationTimeMs: 9 }, + { pid: 500, ppid: 1, creationTimeMs: 11 } + ]), + now: () => 42 + }) + + expect(captured).toEqual({ + root: { pid: 100, creationTimeMs: 5 }, + descendants: [ + { pid: 400, creationTimeMs: 9 }, + { pid: 200, creationTimeMs: 7 } + ], + // Seen but not re-identifiable: counted, so no later read can prove it gone. + unidentifiedCount: 1, + capturedAtMs: 42 + }) + }) + + it('reports an unreadable or rootless table as no snapshot rather than an empty one', async () => { + await expect( + captureWindowsDescendantSnapshot(100, { + readTable: vi.fn(async () => { + throw new Error('table unavailable') + }) + }) + ).resolves.toBeNull() + // A snapshot without the root is stale or filtered; only an observed root + // can authoritatively have no descendants. + await expect( + captureWindowsDescendantSnapshot(100, { + readTable: vi.fn(async () => [{ pid: 999, ppid: 1, creationTimeMs: 5 }]) + }) + ).resolves.toBeNull() + }) + + it('refuses an invalid root pid', async () => { + const readTable = vi.fn() + await expect(captureWindowsDescendantSnapshot(0, { readTable })).resolves.toBeNull() + expect(readTable).not.toHaveBeenCalled() + }) +}) + +describe('verifyWindowsDescendantSnapshotExit', () => { + it('proves an empty tree without reading the table', async () => { + const readTable = vi.fn() + await expect(verifyWindowsDescendantSnapshotExit(snapshot([]), { readTable })).resolves.toBe( + 'exited' + ) + expect(readTable).not.toHaveBeenCalled() + }) + + it('never proves a tree that held a descendant it could not identify', async () => { + // A descendant that denied the creation-time query was seen in the table; + // being unable to re-identify it is "could not look", never "it is gone". + const readTable = vi.fn() + await expect(verifyWindowsDescendantSnapshotExit(snapshot([], 1), { readTable })).resolves.toBe( + 'unverifiable' + ) + expect(readTable).not.toHaveBeenCalled() + + // The identified sibling leaving proves nothing about the unidentified one. + await expect( + verifyWindowsDescendantSnapshotExit(snapshot([{ pid: 200, creationTimeMs: 7 }], 1), { + readTable: vi.fn(async () => []), + wait: async () => {}, + now: vi.fn().mockReturnValueOnce(0).mockReturnValue(1) + }) + ).resolves.toBe('unverifiable') + }) + + it('reports exited once no identity-matched row remains', async () => { + const readTable = vi + .fn() + .mockResolvedValueOnce([{ pid: 200, ppid: 100, creationTimeMs: 7 }]) + // The pid came back on a different process; that is a recycle, not a survivor. + .mockResolvedValueOnce([{ pid: 200, ppid: 100, creationTimeMs: 99 }]) + + await expect( + verifyWindowsDescendantSnapshotExit(snapshot([{ pid: 200, creationTimeMs: 7 }]), { + readTable, + wait: async () => {}, + now: vi.fn().mockReturnValueOnce(0).mockReturnValue(1) + }) + ).resolves.toBe('exited') + expect(readTable).toHaveBeenCalledTimes(2) + }) + + it('reports live for a descendant still matched at the deadline', async () => { + let clock = 0 + await expect( + verifyWindowsDescendantSnapshotExit(snapshot([{ pid: 200, creationTimeMs: 7 }]), { + readTable: vi.fn(async () => [{ pid: 200, ppid: 100, creationTimeMs: 7 }]), + wait: async () => { + clock += 100 + }, + now: () => clock, + verifyMs: 250 + }) + ).resolves.toBe('live') + }) + + it('reports unverifiable when the table cannot be read at the deadline', async () => { + await expect( + verifyWindowsDescendantSnapshotExit(snapshot([{ pid: 200, creationTimeMs: 7 }]), { + readTable: vi.fn(async () => { + throw new Error('table unavailable') + }), + wait: async () => {}, + now: vi.fn().mockReturnValueOnce(0).mockReturnValue(9_999) + }) + ).resolves.toBe('unverifiable') + }) +}) + +describe('terminateIdentifiedWindowsProcessTree', () => { + it('never taskkills a replacement that reused the captured root pid', async () => { + const terminateTree = vi.fn(async () => {}) + + await expect( + terminateIdentifiedWindowsProcessTree( + { pid: 100, creationTimeMs: 5 }, + { + readTable: vi.fn(async () => [{ pid: 100, ppid: 1, creationTimeMs: 99 }]), + terminateTree + } + ) + ).resolves.toBe(false) + expect(terminateTree).not.toHaveBeenCalled() + }) + + it('rechecks retained-child ownership after the identity read settles', async () => { + const terminateTree = vi.fn(async () => {}) + + await expect( + terminateIdentifiedWindowsProcessTree( + { pid: 100, creationTimeMs: 5 }, + { + readTable: vi.fn(async () => [{ pid: 100, ppid: 1, creationTimeMs: 5 }]), + ownsRoot: () => false, + terminateTree + } + ) + ).resolves.toBe(false) + expect(terminateTree).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/windows-descendant-exit-verification.ts b/src/main/windows-descendant-exit-verification.ts new file mode 100644 index 00000000000..079833a2bd6 --- /dev/null +++ b/src/main/windows-descendant-exit-verification.ts @@ -0,0 +1,156 @@ +import type { DescendantTreeVerdict } from './pty-descendant-exit-verification' +import { windowsDescendantsFromRows } from './providers/windows-foreground-process-rows' +import { readWindowsProcessTableFresh } from './windows/windows-process-table' +import { terminateWindowsProcessTree } from './windows-process-tree-kill' + +export const WINDOWS_DESCENDANT_KILL_VERIFY_MS = 3_500 +const WINDOWS_DESCENDANT_POLL_MS = 100 + +/** + * A Windows descendant tree captured while its root was alive, with the + * PID-reuse guard the POSIX snapshot gets from ps lstart: a row only counts as + * the same process when its creation time still matches. Rows without a + * creation time are never signalled, because a bare pid cannot be re-identified, + * but they are counted: a descendant that was seen and denied identification + * is one no later read can prove gone. + */ +export type WindowsProcessIdentity = { pid: number; creationTimeMs: number } + +export type WindowsDescendantSnapshot = { + root: WindowsProcessIdentity + descendants: WindowsProcessIdentity[] + /** Descendants seen in the walk that denied the creation-time query. */ + unidentifiedCount: number + capturedAtMs: number + /** Per-PID boundaries retained when close refreshes merge snapshots. */ + capturedAtMsByPid?: Readonly> +} + +export type WindowsDescendantVerificationDeps = { + readTable?: () => Promise<{ pid: number; ppid: number; creationTimeMs?: number }[]> + now?: () => number + wait?: (ms: number) => Promise + verifyMs?: number +} + +/** Revalidate a Windows PID/creation-time identity immediately before a kill. */ +export async function verifyWindowsProcessIdentity( + target: WindowsProcessIdentity, + deps: Pick = {} +): Promise { + if (!Number.isInteger(target.pid) || target.pid <= 0 || !Number.isFinite(target.creationTimeMs)) { + return false + } + const table = await (deps.readTable ?? readWindowsProcessTableFresh)().catch(() => null) + const current = table?.filter((row) => row.pid === target.pid) ?? [] + return current.length === 1 && current[0]?.creationTimeMs === target.creationTimeMs +} + +function delay(ms: number): Promise { + return new Promise((resolve) => { + const timer = setTimeout(resolve, ms) + timer.unref?.() + }) +} + +/** + * Snapshot a Windows root's descendants while it is still alive. Resolves null + * (never rejects) when the table is unreadable or the root is absent — the same + * contract as the POSIX walk, because "cannot see" is never "nothing is there". + */ +export async function captureWindowsDescendantSnapshot( + rootPid: number, + deps: WindowsDescendantVerificationDeps = {} +): Promise { + if (!Number.isInteger(rootPid) || rootPid <= 0) { + return null + } + const capturedAtMs = (deps.now ?? Date.now)() + // One table read, not a walk plus an identity read: each is bounded in + // seconds, and this runs inside the close ladder's budget. + const table = await (deps.readTable ?? readWindowsProcessTableFresh)().catch(() => null) + const descendants = table && windowsDescendantsFromRows(table, rootPid) + const root = table?.find((row) => row.pid === rootPid) + if (!descendants || typeof root?.creationTimeMs !== 'number') { + return null + } + return { + root: { pid: root.pid, creationTimeMs: root.creationTimeMs }, + descendants: descendants.flatMap((row) => + // A descendant that denied a creation-time query cannot be told from a + // recycled pid later, so it is never signalled on a bare pid. + typeof row.creationTimeMs === 'number' + ? [{ pid: row.pid, creationTimeMs: row.creationTimeMs }] + : [] + ), + unidentifiedCount: descendants.filter((row) => typeof row.creationTimeMs !== 'number').length, + capturedAtMs + } +} + +export type IdentifiedWindowsTreeTerminationDeps = { + readTable?: WindowsDescendantVerificationDeps['readTable'] + terminateTree?: (target: WindowsProcessIdentity) => Promise + ownsRoot?: () => boolean +} + +/** Revalidate the captured root at the last async boundary before taskkill. */ +export async function terminateIdentifiedWindowsProcessTree( + target: WindowsProcessIdentity, + deps: IdentifiedWindowsTreeTerminationDeps = {} +): Promise { + if (!(await verifyWindowsProcessIdentity(target, { readTable: deps.readTable }))) { + return false + } + if (deps.ownsRoot?.() === false) { + return false + } + await ( + deps.terminateTree ?? + ((identified: WindowsProcessIdentity) => terminateWindowsProcessTree(identified.pid)) + )(target) + return true +} + +/** + * Whether a snapshotted Windows tree is gone, polled to a bounded deadline. + * + * Why a verification pass at all: `taskkill /T /F` resolves the same way on a + * timeout, an access denial and a recycled root as it does on a successful + * kill, so its completion is never evidence. Only a table read that no longer + * shows an identity-matched row is. + */ +export async function verifyWindowsDescendantSnapshotExit( + snapshot: WindowsDescendantSnapshot, + deps: WindowsDescendantVerificationDeps = {} +): Promise { + // The most a read can prove: a descendant that denied identification was seen + // and can never be matched gone, so "could not look" caps the verdict. + const proven: DescendantTreeVerdict = snapshot.unidentifiedCount > 0 ? 'unverifiable' : 'exited' + if (snapshot.descendants.length === 0) { + return proven + } + const now = deps.now ?? Date.now + const readTable = deps.readTable ?? readWindowsProcessTableFresh + const deadline = now() + (deps.verifyMs ?? WINDOWS_DESCENDANT_KILL_VERIFY_MS) + let verdict: DescendantTreeVerdict = 'unverifiable' + do { + const table = await readTable().catch(() => null) + if (!table) { + verdict = 'unverifiable' + } else { + const live = new Map(table.map((row) => [row.pid, row.creationTimeMs])) + verdict = snapshot.descendants.some((row) => live.get(row.pid) === row.creationTimeMs) + ? 'live' + : proven + if (verdict === proven) { + return verdict + } + } + if (now() >= deadline) { + return verdict + } + await (deps.wait ?? delay)(WINDOWS_DESCENDANT_POLL_MS) + } while (now() < deadline) + return verdict +} diff --git a/src/main/windows-live-tree-kill.win32.test.ts b/src/main/windows-live-tree-kill.win32.test.ts new file mode 100644 index 00000000000..45563e82c89 --- /dev/null +++ b/src/main/windows-live-tree-kill.win32.test.ts @@ -0,0 +1,197 @@ +import { existsSync, mkdtempSync, readFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { spawn, ChildProcess } from 'node:child_process' +import { subscribe, unsubscribe } from 'node:diagnostics_channel' +import { afterAll, afterEach, beforeEach, describe, expect, it } from 'vitest' +import { setAppEnvironment, type AppEnvironment } from '../shared/app-environment' +import { setProcessTreeKillGate } from '../shared/child-process/process-tree-kill-gate' +import { signalProcessTree } from '../shared/child-process/process-tree-termination' +import { removeTreeSync } from '../shared/windows-transient-lock-removal' +import { + findSelfInitiatedTreeKills, + resetSelfInitiatedTreeKillLogForTest +} from './crash-reporting/self-initiated-tree-kill-log' +import { installMainProcessTreeKillGate } from './own-chromium-tree-kill-guard' +import { terminateWindowsProcessTree } from './windows-process-tree-kill' + +/** + * The unit tests pin the gate's decision against a mocked `taskkill`; this pins + * what that decision does to real Windows processes. + * + * Both are needed. Every claim the gate makes is about a mechanism the mocks + * cannot show: that `taskkill /T /F` actually reaps a detached grandchild, that + * a refusal actually leaves that tree standing, and that the handle-addressed + * root kill the refusal path falls back to actually reaps the root while + * orphaning its descendants — the asymmetry the PR discloses rather than fixes. + * + * Runs only on win32; skipped elsewhere. + */ +const describeOnWindows = process.platform === 'win32' ? describe : describe.skip + +/** Read live by the guard on every kill, so a case can flip it mid-test. */ +let orcaChromiumPids: number[] = [] + +function appEnvironment(): AppEnvironment { + return { + getPath: () => process.cwd(), + getAppPath: () => process.cwd(), + getVersion: () => '0.0.0-live', + isPackaged: () => false, + onWillQuit: () => {}, + exit: () => {}, + getAppMetrics: (() => + orcaChromiumPids.map((pid) => ({ + pid, + type: 'Tab' + }))) as unknown as AppEnvironment['getAppMetrics'] + } +} + +function isAlive(pid: number): boolean { + try { + process.kill(pid, 0) + return true + } catch { + return false + } +} + +const sleep = (ms: number): Promise => new Promise((resolve) => setTimeout(resolve, ms)) + +async function waitFor(predicate: () => boolean, timeoutMs = 10_000): Promise { + const deadline = Date.now() + timeoutMs + while (Date.now() < deadline && !predicate()) { + await sleep(100) + } + return predicate() +} + +let markerDirectory = '' +let markerSequence = 0 +const spawnedRoots: ChildProcess[] = [] +const spawnedLeaves: number[] = [] +const observedSpawns: ChildProcess[] = [] + +function observeSpawn(message: unknown): void { + if ( + typeof message === 'object' && + message !== null && + 'process' in message && + message.process instanceof ChildProcess + ) { + observedSpawns.push(message.process) + } +} + +/** A real root with a real grandchild; the grandchild reports its pid on disk. */ +async function spawnLiveTree(): Promise<{ + child: ChildProcess + rootPid: number + leafPid: number +}> { + const marker = join(markerDirectory, `leaf-${markerSequence++}.pid`) + const leafSource = `require('node:fs').writeFileSync(${JSON.stringify(marker)}, String(process.pid)); setTimeout(() => {}, 600000)` + // Non-detached Windows children can die with the root's libuv Job Object. + const rootSource = `require('node:child_process').spawn(process.execPath, ['-e', ${JSON.stringify(leafSource)}], { stdio: 'ignore', detached: true, windowsHide: true }); setTimeout(() => {}, 600000)` + const child = spawn(process.execPath, ['-e', rootSource], { + stdio: 'ignore', + windowsHide: true + }) + spawnedRoots.push(child) + const rootPid = child.pid as number + expect(rootPid).toBeGreaterThan(0) + expect(await waitFor(() => existsSync(marker))).toBe(true) + const leafPid = Number(readFileSync(marker, 'utf8')) + spawnedLeaves.push(leafPid) + expect(await waitFor(() => isAlive(leafPid))).toBe(true) + return { child, rootPid, leafPid } +} + +describeOnWindows('own-Chromium gate against real Windows process trees', () => { + beforeEach(() => { + markerDirectory ||= mkdtempSync(join(tmpdir(), 'orca-live-tree-kill-')) + resetSelfInitiatedTreeKillLogForTest() + orcaChromiumPids = [] + setAppEnvironment(appEnvironment()) + installMainProcessTreeKillGate() + observedSpawns.length = 0 + subscribe('child_process', observeSpawn) + }) + + afterEach(async () => { + unsubscribe('child_process', observeSpawn) + orcaChromiumPids = [] + for (const leafPid of spawnedLeaves.splice(0)) { + await terminateWindowsProcessTree(leafPid, { site: 'live-tree-kill-cleanup' }) + } + for (const root of spawnedRoots.splice(0)) { + root.kill('SIGKILL') + } + setProcessTreeKillGate(null) + }) + + afterAll(() => { + if (markerDirectory) { + removeTreeSync(markerDirectory) + } + }) + + it('admitted: taskkill reaps the root and its detached grandchild, and the kill is recorded', async () => { + const { rootPid, leafPid } = await spawnLiveTree() + + await terminateWindowsProcessTree(rootPid, { site: 'live-tree-kill-admit' }) + + expect(await waitFor(() => !isAlive(rootPid))).toBe(true) + expect(await waitFor(() => !isAlive(leafPid))).toBe(true) + expect( + findSelfInitiatedTreeKills(Date.now()).some( + (kill) => kill.pid === rootPid && kill.site === 'live-tree-kill-admit' + ) + ).toBe(true) + }) + + it('refused: the tree survives, nothing is recorded, and the handle kill still reaps the root', async () => { + const { child, rootPid, leafPid } = await spawnLiveTree() + orcaChromiumPids = [rootPid] + + await terminateWindowsProcessTree(rootPid, { site: 'live-tree-kill-refuse' }) + + await sleep(1_000) + expect(isAlive(rootPid)).toBe(true) + expect(isAlive(leafPid)).toBe(true) + expect(findSelfInitiatedTreeKills(Date.now())).toEqual([]) + + // The fallback every gated site runs after a refusal. + child.kill('SIGKILL') + expect(await waitFor(() => !isAlive(rootPid))).toBe(true) + // Let root-owned job cleanup finish before asserting independent survival. + await sleep(250) + // Disclosed asymmetry: a refusal orphans descendants rather than reaping them. + expect(isAlive(leafPid)).toBe(true) + }) + + it('signalProcessTree refused: the root goes by handle and the barrier reports unverified', async () => { + const { child, rootPid, leafPid } = await spawnLiveTree() + orcaChromiumPids = [rootPid] + observedSpawns.length = 0 + + await expect(signalProcessTree(child, 'SIGKILL')).resolves.toBe(false) + + expect(observedSpawns).toHaveLength(0) + expect(await waitFor(() => !isAlive(rootPid))).toBe(true) + await sleep(250) + expect(isAlive(leafPid)).toBe(true) + }) + + it('signalProcessTree admitted: the whole tree goes and the barrier reports verified', async () => { + const { child, rootPid, leafPid } = await spawnLiveTree() + observedSpawns.length = 0 + + await expect(signalProcessTree(child, 'SIGKILL')).resolves.toBe(true) + + expect(observedSpawns.map((child) => child.spawnfile)).toEqual(['taskkill']) + expect(await waitFor(() => !isAlive(rootPid))).toBe(true) + expect(await waitFor(() => !isAlive(leafPid))).toBe(true) + }) +}) diff --git a/src/main/windows-process-tree-kill.ts b/src/main/windows-process-tree-kill.ts index ea7b756ada4..31e6b18da2a 100644 --- a/src/main/windows-process-tree-kill.ts +++ b/src/main/windows-process-tree-kill.ts @@ -11,8 +11,9 @@ export const WINDOWS_PROCESS_TREE_KILL_TIMEOUT_MS = 5_000 * Best-effort: missing/already-dead roots still resolve so callers can finish * their own handle cleanup via killRoot. * - * Nearly every main-process taskkill runs through here; the two account-login - * teardowns keep their own spawn but share the same gate, so the refusal and the + * Most main-process taskkills run through here; the families that keep their own + * spawn (account-login teardowns, codex app-server deadline, git-command abort, + * notebook and precheck timeouts) share the same gate, so the refusal and the * breadcrumb live in `admitSelfInitiatedTreeKill` rather than in this function. */ export function terminateWindowsProcessTree( diff --git a/src/preload/api/pty-api.ts b/src/preload/api/pty-api.ts index a1850398357..d2d547d98cb 100644 --- a/src/preload/api/pty-api.ts +++ b/src/preload/api/pty-api.ts @@ -69,6 +69,8 @@ export type PtyApi = { coldRestore?: { scrollback: string; cwd: string; cols?: number; rows?: number } startupCwdFallback?: { kind: 'worktree'; cwd: string } agentResumeUnavailable?: true + /** Host verdict on the shell-ready marker; absent when the execution host predates the field. */ + shellReadyArmed?: boolean }> write: (id: string, data: string) => void writeAccepted: (id: string, data: string) => Promise @@ -112,7 +114,11 @@ export type PtyApi = { getForegroundProcess: (id: string) => Promise inspectProcess: ( id: string, - options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } + options?: { + expectedIncarnationId?: string + scanChildProcesses?: boolean + steadyState?: boolean + } ) => Promise confirmForegroundProcess: (id: string) => Promise getCwd: (id: string) => Promise diff --git a/src/preload/api/pty-bridge-session-control.ts b/src/preload/api/pty-bridge-session-control.ts index ef6002e11c8..2854278c20b 100644 --- a/src/preload/api/pty-bridge-session-control.ts +++ b/src/preload/api/pty-bridge-session-control.ts @@ -66,6 +66,8 @@ export const ptySessionControlApi = { coldRestore?: { scrollback: string; cwd: string; cols?: number; rows?: number } startupCwdFallback?: { kind: 'worktree'; cwd: string } agentResumeUnavailable?: true + /** Host verdict on the shell-ready marker; absent when the execution host predates the field. */ + shellReadyArmed?: boolean }> => ipcRenderer.invoke('pty:spawn', opts), write: (id: string, data: string): void => { ipcRenderer.send('pty:write', { id, data }) diff --git a/src/preload/api/pty-bridge-stream-and-serialization.ts b/src/preload/api/pty-bridge-stream-and-serialization.ts index 414a5514bfa..0847291ba7e 100644 --- a/src/preload/api/pty-bridge-stream-and-serialization.ts +++ b/src/preload/api/pty-bridge-stream-and-serialization.ts @@ -7,7 +7,11 @@ import type { TerminalProcessInspection } from '../../shared/terminal-process-in export const ptyStreamAndSerializationApi = { inspectProcess: ( id: string, - options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } + options?: { + expectedIncarnationId?: string + scanChildProcesses?: boolean + steadyState?: boolean + } ): Promise => ipcRenderer.invoke('pty:inspectProcess', { id, ...options }), confirmForegroundProcess: (id: string): Promise => diff --git a/src/relay/fs-path-metadata-requests.ts b/src/relay/fs-path-metadata-requests.ts index 2a9717a6484..b0fa2347d95 100644 --- a/src/relay/fs-path-metadata-requests.ts +++ b/src/relay/fs-path-metadata-requests.ts @@ -1,6 +1,8 @@ import { readdir, stat, lstat, realpath } from 'node:fs/promises' +import type { Dirent } from 'node:fs' import { join } from 'node:path' import { sortDirEntries } from '../shared/file-name-sort' +import { forEachWithConcurrency } from '../shared/map-with-concurrency' import { expandTilde } from './context' async function resolveSymlinkDirectoryEntry( @@ -34,11 +36,18 @@ function fileStatFromLstat(stats: Awaited>) { } } +// Why bounded: a pnpm `node_modules` is hundreds-to-thousands of package symlinks, and one +// unbounded `Promise.all` of stats from a single readDir saturates libuv's four-thread pool — +// delaying every other relay filesystem operation, including the interactive reads the +// list-files scan coordinator exists to protect. Matches the cap every other bounded probe in +// this codebase uses. +const SYMLINK_DIRECTORY_PROBE_CONCURRENCY = 8 + export async function readRelayDir(params: Record) { const dirPath = expandTilde(params.dirPath as string) const entries = await readdir(dirPath, { withFileTypes: true }) const mapped: { name: string; isDirectory: boolean; isSymlink: boolean }[] = [] - const symlinkProbes: Promise[] = [] + const symlinkEntries: { entry: Dirent; mappedEntry: (typeof mapped)[number] }[] = [] for (const entry of entries) { const mappedEntry = { name: entry.name, @@ -47,15 +56,17 @@ export async function readRelayDir(params: Record) { } mapped.push(mappedEntry) if (!mappedEntry.isDirectory && mappedEntry.isSymlink) { - symlinkProbes.push( - resolveSymlinkDirectoryEntry(dirPath, entry).then((isDirectory) => { - mappedEntry.isDirectory = isDirectory - }) - ) + symlinkEntries.push({ entry, mappedEntry }) } } - if (symlinkProbes.length > 0) { - await Promise.all(symlinkProbes) + if (symlinkEntries.length > 0) { + await forEachWithConcurrency( + symlinkEntries, + SYMLINK_DIRECTORY_PROBE_CONCURRENCY, + async ({ entry, mappedEntry }) => { + mappedEntry.isDirectory = await resolveSymlinkDirectoryEntry(dirPath, entry) + } + ) } return sortDirEntries(mapped) } diff --git a/src/relay/fs-path-metadata-symlink-concurrency.test.ts b/src/relay/fs-path-metadata-symlink-concurrency.test.ts new file mode 100644 index 00000000000..3a342a79e7c --- /dev/null +++ b/src/relay/fs-path-metadata-symlink-concurrency.test.ts @@ -0,0 +1,70 @@ +import { mkdirSync, mkdtempSync, rmSync, symlinkSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type * as FsPromisesModule from 'node:fs/promises' + +const statCalls = vi.hoisted(() => ({ inFlight: 0, peak: 0, total: 0 })) + +vi.mock('node:fs/promises', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + stat: async (...args: Parameters) => { + statCalls.inFlight += 1 + statCalls.total += 1 + statCalls.peak = Math.max(statCalls.peak, statCalls.inFlight) + try { + return await actual.stat(...args) + } finally { + statCalls.inFlight -= 1 + } + } + } +}) + +const { readRelayDir } = await import('./fs-path-metadata-requests') + +describe('relay readDir symlink probes', () => { + let root: string + let targetRoot: string + + beforeEach(() => { + statCalls.inFlight = 0 + statCalls.peak = 0 + statCalls.total = 0 + root = mkdtempSync(join(tmpdir(), 'orca-relay-readdir-')) + // Kept outside `root` so the listing contains only the symlinks under test. + targetRoot = mkdtempSync(join(tmpdir(), 'orca-relay-readdir-target-')) + const target = join(targetRoot, 'target') + mkdirSync(target) + writeFileSync(join(target, 'index.js'), '') + // A pnpm-shaped node_modules: many package symlinks in one directory. Junctions on + // Windows: plain symlinks need Developer Mode there. + for (let index = 0; index < 60; index += 1) { + symlinkSync( + target, + join(root, `pkg-${index}`), + process.platform === 'win32' ? 'junction' : 'dir' + ) + } + }) + + afterEach(() => { + rmSync(root, { recursive: true, force: true }) + rmSync(targetRoot, { recursive: true, force: true }) + }) + + it('bounds concurrent symlink stats instead of issuing one per entry at once', async () => { + const entries = await readRelayDir({ dirPath: root }) + + expect(statCalls.total).toBe(60) + // Exactly the cap: every worker enters `stat` before any resolves, so the peak proves the + // probes overlap and that no more than 8 ever do. Unbounded, all 60 would be in flight, + // saturating libuv's four-thread pool and stalling every other relay filesystem read. + expect(statCalls.peak).toBe(8) + // Behaviour is unchanged: every symlink still resolves to its target's kind. + expect(entries).toHaveLength(60) + expect(entries.every((entry) => entry.isDirectory && entry.isSymlink)).toBe(true) + }) +}) diff --git a/src/relay/pty-handler-ownership-attestation.test.ts b/src/relay/pty-handler-ownership-attestation.test.ts index 94f462c46d2..ee1c144df09 100644 --- a/src/relay/pty-handler-ownership-attestation.test.ts +++ b/src/relay/pty-handler-ownership-attestation.test.ts @@ -33,7 +33,8 @@ import { endPtyHandlerTest, type MockDispatcher } from './pty-handler-test-harness' -import { PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS } from '../shared/process-table-snapshot-reader' +import * as processTableSnapshotReader from '../shared/process-table-snapshot-reader' +import { RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS } from '../shared/ssh-relay-pty-ownership-proof' const PANE_KEY = 'tab-agent:22222222-2222-4222-8222-222222222222' @@ -139,16 +140,41 @@ describe('PtyHandler publishes host-attested PTY ownership', () => { expect(entry?.ownerClientInstanceId).toBe('client-A') }) - it('dates the foreground observation instead of stamping it fresh', async () => { - // `capturedAgeMs` used to be a hardcoded 0 with no reader anywhere, so the one field that - // exists to bound staleness asserted the evidence was never stale. It now carries the - // actual age of the TTL-shared capture the record was derived from. - const { id } = await spawnFrom(7, { env: { ORCA_PANE_KEY: PANE_KEY } }) + it('publishes the age the capture reported, rather than restamping it fresh', async () => { + // This assertion used to read `capturedAgeMs <= PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS`, which + // could not fail: `beginPtyHandlerTest` installs fake timers, so `Date.now()` is frozen, the + // real reader reports exactly +0, and `0 <= 500` held identically for a hardcoded zero, for + // completion-stamping and for start-stamping. The one test guarding this field was blind to + // every change to it, while the real reader on a 2,002-process host returns thousands of ms. + // + // So drive a real age in from the reader. That the reader MEASURES the age correctly is + // pinned separately, against a controllable clock, by process-table-snapshot.test.ts; what + // belongs here is that the handler publishes what it was given instead of restamping. + const capturedAgeMs = 6_140 + const snapshot = vi + .spyOn(processTableSnapshotReader, 'getStrictProcessTableSnapshotWithAge') + .mockResolvedValue({ rows: [], capturedAgeMs }) + const { id } = await spawnFrom(7, { env: { ORCA_PANE_KEY: PANE_KEY } }) const entry = (await listProcesses()).find((process) => process.id === id) - expect(entry?.foregroundProcessEvidence?.capturedAgeMs).toBeLessThanOrEqual( - PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS + expect(snapshot).toHaveBeenCalled() + expect(entry?.foregroundProcessEvidence?.capturedAgeMs).toBe(capturedAgeMs) + }) + + it('publishes an age a destructive consumer will refuse, rather than one it will trust', async () => { + // The point of the field, stated as the consumer sees it: an observation this old cannot + // authorize a stop, and the whole bug was that it used to arrive claiming it could. + const snapshot = vi + .spyOn(processTableSnapshotReader, 'getStrictProcessTableSnapshotWithAge') + .mockResolvedValue({ rows: [], capturedAgeMs: 6_140 }) + + const { id } = await spawnFrom(7, { env: { ORCA_PANE_KEY: PANE_KEY } }) + const entry = (await listProcesses()).find((process) => process.id === id) + + expect(snapshot).toHaveBeenCalled() + expect(entry?.foregroundProcessEvidence?.capturedAgeMs).toBeGreaterThan( + RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS ) }) }) diff --git a/src/relay/pty-handler-spawn-admission.test.ts b/src/relay/pty-handler-spawn-admission.test.ts index 045ee2e412d..6fef02a9cc0 100644 --- a/src/relay/pty-handler-spawn-admission.test.ts +++ b/src/relay/pty-handler-spawn-admission.test.ts @@ -126,6 +126,46 @@ describe('PtyHandler', () => { expect(hasChildren).toHaveBeenLastCalledWith(mockPtyInstance.pid, { fresh: true }) }) + it('does not re-enter the shared capture after the evidence read gave up on it', async () => { + // The budget is worthless if the compatibility fields answer by joining the very capture the + // evidence read just abandoned: `inspectPtyChildProcesses` and `getForegroundProcessName` + // read the same TTL-shared table with no budget of their own, so on a slow host this call + // would still block for the whole capture -- once, then once per managed PTY in the listing. + const snapshot = vi + .spyOn(processTableSnapshotReader, 'getStrictProcessTableSnapshotWithAge') + .mockRejectedValue(new Error('process table unreadable: capture_over_budget')) + const hasChildren = vi.spyOn(ptyChildProcessInspection, 'inspectPtyChildProcesses') + const foregroundName = vi.spyOn(ptyShellUtils, 'getForegroundProcessName') + + const { id } = (await spawnPty({ cols: 80, rows: 24 })) as { id: string } + hasChildren.mockClear() + foregroundName.mockClear() + + const inspection = (await dispatcher.callRequest('pty.inspectProcess', { id })) as { + hasChildProcesses: boolean + childProcessEvidence?: string + foregroundProcessEvidence?: { verdict: string; reason?: string } + } + + expect(snapshot).toHaveBeenCalled() + expect(hasChildren).not.toHaveBeenCalled() + // The verdict the gates already handle, reached promptly instead of late. + expect(inspection.foregroundProcessEvidence?.verdict).toBe('unverifiable') + expect(inspection.foregroundProcessEvidence?.reason).toBe('process_table_unreadable') + // The honest verdict rather than a fabricated negative, reached without the wait. The + // compatibility boolean still spells `unverifiable` as `false` for older clients. + expect(inspection.childProcessEvidence).toBe('unverifiable') + expect(inspection.hasChildProcesses).toBe(false) + + const listing = (await dispatcher.callRequest('pty.listProcesses', {})) as { + id: string + title: string + }[] + + expect(foregroundName).not.toHaveBeenCalled() + expect(listing.find((entry) => entry.id === id)?.title).toBeTruthy() + }) + it('rejects strict process inspection for a missing relay PTY', async () => { await expect(dispatcher.callRequest('pty.inspectProcess', { id: 'missing' })).rejects.toThrow( 'terminal_gone' @@ -134,7 +174,13 @@ describe('PtyHandler', () => { it('spawns a PTY and returns an id', async () => { const result = await spawnPty({ cols: 80, rows: 24 }) - expect(result).toEqual({ id: testPtyId(1), incarnationId: expect.any(String) }) + // shellReadyArmed rides every spawn reply, false included: absent has to keep + // meaning "host predates the field", not "host did not arm". + expect(result).toEqual({ + id: testPtyId(1), + incarnationId: expect.any(String), + shellReadyArmed: false + }) expect(mockPtySpawn).toHaveBeenCalled() expect(handler.activePtyCount).toBe(1) }) @@ -153,7 +199,11 @@ describe('PtyHandler', () => { agentSessionCreateOperationId: operationId }) - expect(replayed).toEqual({ id: testPtyId(1), incarnationId: expect.any(String) }) + expect(replayed).toEqual({ + id: testPtyId(1), + incarnationId: expect.any(String), + shellReadyArmed: false + }) expect(mockPtySpawn).toHaveBeenCalledOnce() expect(mockPtyInstance.kill).not.toHaveBeenCalled() expect(handler.activePtyCount).toBe(1) diff --git a/src/relay/pty-handler-startup-command-delivery.test.ts b/src/relay/pty-handler-startup-command-delivery.test.ts index 817ca756385..9e7c29202c7 100644 --- a/src/relay/pty-handler-startup-command-delivery.test.ts +++ b/src/relay/pty-handler-startup-command-delivery.test.ts @@ -133,6 +133,77 @@ describe('PtyHandler', () => { } ) + it.skipIf(process.platform === 'win32')( + 'emits shell-ready markers for plain Codex on a line-editor shell', + async () => { + const oldShell = process.env.SHELL + const oldHome = process.env.HOME + const homeDir = mkdtempSync(join(tmpdir(), 'relay-plain-codex-spawn-')) + + process.env.SHELL = '/bin/bash' + process.env.HOME = homeDir + try { + // No prefill flag and no shell-ready hint: the host decides from its own + // shell, because the client cannot see it (#18767). + const reply = await dispatcher.callRequest('pty.spawn', { + env: { HOME: homeDir }, + command: 'codex' + }) + expect(reply).toMatchObject({ shellReadyArmed: true }) + } finally { + if (oldShell === undefined) { + delete process.env.SHELL + } else { + process.env.SHELL = oldShell + } + if (oldHome === undefined) { + delete process.env.HOME + } else { + process.env.HOME = oldHome + } + rmSync(homeDir, { recursive: true, force: true }) + } + + const spawnOptions = mockPtySpawn.mock.calls[0]?.[2] as + | { env?: Record } + | undefined + expect(spawnOptions?.env?.ORCA_SHELL_FEATURES).toContain('ready') + vi.advanceTimersByTime(15_000) + expect(handler.retainedStartupCommandCount).toBe(0) + } + ) + + it.skipIf(process.platform === 'win32')( + 'leaves plain Codex unwaited on a shell that emits the marker before its reader', + async () => { + const oldHome = process.env.HOME + const homeDir = mkdtempSync(join(tmpdir(), 'relay-plain-codex-fish-spawn-')) + + process.env.HOME = homeDir + try { + const reply = await dispatcher.callRequest('pty.spawn', { + env: { HOME: homeDir, SHELL: '/usr/bin/fish' }, + command: 'codex' + }) + // Why the reply carries it: the client cannot see this shell, and without + // the verdict it waits the full fallback for a marker fish never emits. + expect(reply).toMatchObject({ shellReadyArmed: false }) + } finally { + if (oldHome === undefined) { + delete process.env.HOME + } else { + process.env.HOME = oldHome + } + rmSync(homeDir, { recursive: true, force: true }) + } + + const spawnOptions = mockPtySpawn.mock.calls[0]?.[2] as + | { env?: Record } + | undefined + expect(spawnOptions?.env?.ORCA_SHELL_FEATURES ?? '').not.toContain('ready') + } + ) + it.skipIf(process.platform === 'win32')( 'emits shell-ready markers for renderer-delivered Codex native prefill commands', async () => { diff --git a/src/relay/pty-handler.ts b/src/relay/pty-handler.ts index 190bcabb3f9..cdf436bca2a 100644 --- a/src/relay/pty-handler.ts +++ b/src/relay/pty-handler.ts @@ -231,6 +231,10 @@ type ManagedPty = { gitCredentialPromptGuarded: boolean historyIsolationEnabled?: boolean startupCommand?: ManagedStartupCommand + /** Whether this host armed the shell-ready marker for a renderer-delivered startup command. + * Kept off `startupCommand`, which is dropped once delivered; the client reads it from the + * spawn reply to skip waiting for a marker that will never come (fish, sh, Windows). */ + shellReadyArmed?: boolean physicalExit?: PhysicalExitTracker forceKillSent?: boolean gracefulKillSent?: boolean @@ -253,6 +257,7 @@ type RelayAgentSessionCreateResult = { replay?: string agentSessionEnsure?: unknown sourceActivation?: PtySourceReceivingActivation + shellReadyArmed?: boolean } const AGENT_SESSION_CREATE_OPERATION_ID_PATTERN = /^[A-Za-z0-9_-]{43}$/ @@ -1804,7 +1809,10 @@ export class PtyHandler { incarnationId: managed.incarnationId, agentSessionEnsure: result, ...(sourceActivation ? { sourceActivation } : {}), - ...(adoptedReplay ? { replay: adoptedReplay } : {}) + ...(adoptedReplay ? { replay: adoptedReplay } : {}), + ...(managed.shellReadyArmed !== undefined + ? { shellReadyArmed: managed.shellReadyArmed } + : {}) } } catch (error) { if (!physicalSpawnCommitted) { @@ -1827,6 +1835,7 @@ export class PtyHandler { id: string incarnationId: string sourceActivation?: PtySourceReceivingActivation + shellReadyArmed?: boolean }> { const pty = await this.loadPty() if (!pty) { @@ -1896,12 +1905,16 @@ export class PtyHandler { isUnattended: launchAgent !== undefined, platform: process.platform }) + // Why the shell is part of the decision here and not on the client: the client + // cannot see which shell this host runs, and plain Codex must still wait where + // the marker rides the line editor rather than double-echoing an early write. const shouldEmitShellReadyMarker = launchCommandHint !== undefined && shouldUseShellReadyStartupDelivery({ command: launchCommandHint, startupCommandDelivery: - params.startupCommandDelivery === 'shell-ready' ? 'shell-ready' : undefined + params.startupCommandDelivery === 'shell-ready' ? 'shell-ready' : undefined, + shellPath: shell }) const managedStartupCommand = shouldProviderDeliverCommand ? command : launchCommandHint // Why: both renderer- and provider-delivered startup commands use this marker; the delivering side strips it from output. @@ -1994,6 +2007,7 @@ export class PtyHandler { }), ...(startupIngressIntent ? { startupIngressIntent } : {}), ...(terminalHandle ? { terminalHandle } : {}), + shellReadyArmed: rendererShellReadySupported, ...(managedStartupCommand && (shouldProviderDeliverCommand || rendererShellReadySupported) ? { startupCommand: { @@ -2035,7 +2049,8 @@ export class PtyHandler { return { id, incarnationId: managed.incarnationId, - ...(sourceActivation ? { sourceActivation } : {}) + ...(sourceActivation ? { sourceActivation } : {}), + shellReadyArmed: rendererShellReadySupported } } @@ -2663,6 +2678,9 @@ export class PtyHandler { } } let rows: readonly ProcessTableRow[] | null = null + // Set only when the budgeted evidence read gave up, so the compatibility fields below do not + // turn around and ask the same unreadable table again with no budget at all. + let tableUnavailable = false let evidence: RemoteForegroundEvidence | undefined if (process.platform === 'win32') { // Why SSH-to-Windows is always unverifiable: POSIX has a real foreground primitive @@ -2701,6 +2719,7 @@ export class PtyHandler { rows ) } catch { + tableUnavailable = true evidence = { authorityGeneration: this.ptyIdMintEpoch, observationEpoch: ++this.foregroundEvidenceEpoch, @@ -2727,13 +2746,19 @@ export class PtyHandler { // 1.36s CIM scan, and polling that would reinstate exactly the fork storm the shared table // exists to prevent (#15209, #15036). Close and cleanup decisions ask for the scan by name; // a poll gets the honest `unverifiable` instead of a fabricated negative. + // Why `tableUnavailable` first: it means the budgeted evidence read already gave up. Without + // this arm `inspectPtyChildProcesses` re-enters `getProcessTableSnapshot()` and joins the very + // capture this call just abandoned, blocking for all of it and spending the whole latency the + // budget exists to avoid. The destructive `pty.hasChildProcesses` RPC keeps its fresh probe. const childProcessEvidence: PtyChildProcessVerdict = rows ? rows.some((row) => row.ppid === managed.pty.pid) ? 'children' : 'no-children' - : process.platform === 'win32' && params.scanChildProcesses !== true + : tableUnavailable ? 'unverifiable' - : await inspectPtyChildProcesses(managed.pty.pid) + : process.platform === 'win32' && params.scanChildProcesses !== true + ? 'unverifiable' + : await inspectPtyChildProcesses(managed.pty.pid) return { foregroundProcess, // `unverifiable` keeps spelling itself `false` on the compatibility field, which is what @@ -2758,6 +2783,10 @@ export class PtyHandler { // process-table work on the host. const includeForegroundProcessEvidence = params.includeForegroundProcessEvidence !== false let evidenceRows: readonly ProcessTableRow[] | null = null + // Same reason as `inspectProcess`: once the budgeted read has given up, the per-PTY title + // fallback below must not re-enter the same capture without a budget -- and here it would do + // so once per managed PTY. + let evidenceTableUnavailable = false let evidenceResults: BatchedForegroundProcessResult[] = [] const evidenceEpoch = ++this.foregroundEvidenceEpoch // Worst-case capture time for the snapshot below, not the instant its await settled: the @@ -2783,6 +2812,7 @@ export class PtyHandler { } catch { // An unreadable capture is represented as unverifiable evidence below; // existing inventory fields remain available for old clients. + evidenceTableUnavailable = true } } for (const [entryIndex, [id, managed]] of managedEntries.entries()) { @@ -2797,7 +2827,7 @@ export class PtyHandler { const title = (evidenceRows ? (evidenceResults[entryIndex]?.processName ?? managed.pty.process ?? null) - : includeForegroundProcessEvidence + : includeForegroundProcessEvidence && !evidenceTableUnavailable ? await getForegroundProcessName(managed.pty.pid, managed.pty.process || null) : managed.pty.process || null) || 'shell' const foregroundProcessEvidence = diff --git a/src/renderer/src/assets/main.css b/src/renderer/src/assets/main.css index e3d267cb353..3527f11c44e 100644 --- a/src/renderer/src/assets/main.css +++ b/src/renderer/src/assets/main.css @@ -229,6 +229,10 @@ --git-decoration-untracked: #007100; --git-decoration-copied: #007acc; --git-decoration-ignored: #8c8c8c; + --diff-added-ground: color-mix(in srgb, var(--git-decoration-added) 13%, transparent); + --diff-added-gutter: color-mix(in srgb, var(--git-decoration-added) 26%, transparent); + --diff-removed-ground: color-mix(in srgb, var(--git-decoration-deleted) 11%, transparent); + --diff-removed-gutter: color-mix(in srgb, var(--git-decoration-deleted) 22%, transparent); --git-graph-ref: #007acc; --git-graph-remote-ref: #b66dff; --git-graph-base-ref: #ea5c00; @@ -341,6 +345,10 @@ --git-decoration-untracked: #73c991; --git-decoration-copied: #73c991; --git-decoration-ignored: #6e6e6e; + --diff-added-ground: color-mix(in srgb, var(--git-decoration-added) 16%, transparent); + --diff-added-gutter: color-mix(in srgb, var(--git-decoration-added) 30%, transparent); + --diff-removed-ground: color-mix(in srgb, var(--git-decoration-deleted) 18%, transparent); + --diff-removed-gutter: color-mix(in srgb, var(--git-decoration-deleted) 32%, transparent); --git-graph-ref: #3794ff; --git-graph-remote-ref: #b66dff; --git-graph-base-ref: #ea5c00; diff --git a/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.dead-guest.test.tsx b/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.dead-guest.test.tsx new file mode 100644 index 00000000000..f2fd2ad533b --- /dev/null +++ b/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.dead-guest.test.tsx @@ -0,0 +1,310 @@ +// @vitest-environment happy-dom +import { act, cleanup, render, screen } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { BrowserPage } from '../../../../shared/browser-workspace-types' + +const mocks = vi.hoisted(() => ({ + attach: vi.fn(), + detach: vi.fn(), + recordBreadcrumb: vi.fn() +})) + +vi.mock('./browser-client-page-renderer-installation', () => ({ + attachBrowserClientPageToViewport: mocks.attach +})) +vi.mock('@/lib/crash-breadcrumb-recorder', () => ({ + recordRendererCrashBreadcrumb: mocks.recordBreadcrumb +})) +vi.mock('sonner', () => ({ + toast: { error: vi.fn(), success: vi.fn(), loading: vi.fn(), message: vi.fn() } +})) + +import { TooltipProvider } from '@/components/ui/tooltip' +import { installClientHostedPaneApi } from './client-hosted-browser-pane-test-rig' +import { ClientHostedBrowserPagePane } from './ClientHostedBrowserPagePane' + +const PLACEMENT = { + kind: 'client' as const, + browserHostClientId: 'host-a', + browserHostGeneration: 3, + pageHostGeneration: 7 +} + +/** Verbatim from Electron 43.4.1: main destroyed the guest, the tag still holds its id. */ +function invalidGuestInstanceId(): Error { + return new Error('Invalid guestInstanceId: 7') +} + +/** Verbatim from Electron 43.4.1: focus() after the retained tag left the DOM. */ +function nullContentWindowFocus(): TypeError { + return new TypeError("Cannot read properties of null (reading 'focus')") +} + +function page(overrides?: Partial): BrowserPage { + return { + id: 'page-a', + workspaceId: 'workspace-a', + worktreeId: 'worktree-a', + url: 'https://example.internal/', + title: 'Example', + loading: false, + faviconUrl: null, + canGoBack: false, + canGoForward: false, + loadError: null, + createdAt: 1, + ...overrides + } +} + +function createGuest(): Electron.WebviewTag & { + getURL: ReturnType + reload: ReturnType +} { + const webview = document.createElement('webview') as Electron.WebviewTag & { + getURL: ReturnType + reload: ReturnType + } + Object.assign(webview, { + getURL: vi.fn(() => 'https://example.internal/'), + getTitle: vi.fn(() => 'Example'), + isLoading: vi.fn(() => false), + canGoBack: vi.fn(() => false), + canGoForward: vi.fn(() => false), + focus: vi.fn(), + blur: vi.fn(), + goBack: vi.fn(), + goForward: vi.fn(), + reload: vi.fn(), + loadURL: vi.fn(async () => {}) + }) + mocks.attach.mockReturnValue({ + webview, + detach: mocks.detach, + nextMetadataRevision: vi.fn(() => 1) + }) + return webview +} + +function paneElement( + isActive: boolean, + options?: { browserTab?: BrowserPage; onUpdatePageState?: (id: string, state: unknown) => void } +): React.JSX.Element { + return ( + + + + ) +} + +let webview: ReturnType + +beforeEach(() => { + mocks.attach.mockReset() + mocks.detach.mockReset() + mocks.recordBreadcrumb.mockReset() + installClientHostedPaneApi() + webview = createGuest() +}) + +afterEach(() => { + cleanup() + vi.clearAllMocks() +}) + +describe('client-hosted browser pane over a dead guest', () => { + it('degrades to the unavailable notice when the guest was destroyed in main', () => { + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + expect(() => render(paneElement(true))).not.toThrow() + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(mocks.detach).toHaveBeenCalled() + expect(mocks.recordBreadcrumb).toHaveBeenCalledWith('browser_client_page_guest_unavailable', { + browserPageId: 'page-a', + pageHostGeneration: PLACEMENT.pageHostGeneration, + reason: 'unreadable', + tagConnected: false + }) + // Why: the catch is total, so the swallowed error must stay visible to diagnostics. + expect(mocks.recordBreadcrumb).toHaveBeenCalledWith('browser_client_page_guest_read_failed', { + errorName: 'Error', + errorMessage: 'Invalid guestInstanceId: 7' + }) + }) + + it('stops the spinner it inherited from a page that died mid-load', () => { + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + const onUpdatePageState = vi.fn() + + render(paneElement(true, { browserTab: page({ loading: true }), onUpdatePageState })) + + expect(onUpdatePageState).toHaveBeenCalledWith('page-a', { loading: false }) + }) + + it('flips to the unavailable notice when the guest renderer goes away after attach', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { browserTab: page({ loading: true }), onUpdatePageState })) + expect(screen.queryByText('Client-hosted browser unavailable')).toBeNull() + onUpdatePageState.mockClear() + + // The registry pulls the tag out of the DOM on this event without telling the pane. + webview.remove() + act(() => { + webview.dispatchEvent(new Event('render-process-gone')) + }) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(mocks.detach).toHaveBeenCalled() + expect(onUpdatePageState).toHaveBeenCalledWith('page-a', { loading: false }) + expect(mocks.recordBreadcrumb).toHaveBeenCalledWith('browser_client_page_guest_unavailable', { + browserPageId: 'page-a', + pageHostGeneration: PLACEMENT.pageHostGeneration, + reason: 'render-process-gone', + tagConnected: false + }) + }) + + it('flips to the unavailable notice when main destroys the guest after attach', () => { + render(paneElement(true)) + + act(() => { + webview.dispatchEvent(new Event('destroyed')) + }) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(mocks.recordBreadcrumb).toHaveBeenCalledWith( + 'browser_client_page_guest_unavailable', + expect.objectContaining({ reason: 'destroyed' }) + ) + // The chrome must not keep driving the dead tag: Reload routes to the notice, not a throw. + webview.reload.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + expect(() => act(() => screen.getByRole('button', { name: 'Reload' }).click())).not.toThrow() + expect(webview.reload).not.toHaveBeenCalled() + }) + + it('does not freeze silently when a navigation event finds the guest gone', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { onUpdatePageState })) + onUpdatePageState.mockClear() + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + act(() => { + webview.dispatchEvent(new Event('did-navigate')) + }) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(onUpdatePageState).toHaveBeenCalledWith('page-a', { loading: false }) + }) + + it('stops the spinner when the guest dies as a load starts', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { onUpdatePageState })) + onUpdatePageState.mockClear() + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + act(() => { + webview.dispatchEvent(new Event('did-start-loading')) + }) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + // did-start-loading writes loading:true first; the loss must be the last word. + expect(onUpdatePageState.mock.calls.at(-1)).toEqual(['page-a', { loading: false }]) + }) + + it('ignores queued load events after guest loss', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { onUpdatePageState })) + act(() => webview.dispatchEvent(new Event('destroyed'))) + onUpdatePageState.mockClear() + + act(() => webview.dispatchEvent(new Event('did-start-loading'))) + + expect(onUpdatePageState).not.toHaveBeenCalled() + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + }) + + it('uses the guarded title snapshot if the guest dies immediately afterward', () => { + render(paneElement(true)) + webview.getTitle = vi.fn(() => { + webview.getTitle = vi.fn(() => { + throw invalidGuestInstanceId() + }) + return 'Last live title' + }) + + expect(() => act(() => webview.dispatchEvent(new Event('did-navigate')))).not.toThrow() + expect(webview.getTitle).not.toHaveBeenCalled() + }) + + it('shows unavailability when a load-failure fallback finds the guest gone', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { onUpdatePageState })) + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + act(() => webview.dispatchEvent(new Event('did-fail-load'))) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(onUpdatePageState.mock.calls.at(-1)).toEqual(['page-a', { loading: false }]) + }) + + it('removes loss listeners when the initial guest read fails', () => { + const removeListener = vi.spyOn(webview, 'removeEventListener') + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + render(paneElement(true)) + + expect(removeListener).toHaveBeenCalledWith('destroyed', expect.any(Function)) + expect(removeListener).toHaveBeenCalledWith('render-process-gone', expect.any(Function)) + }) + + it('stops listening for guest loss once the pane lets go of the tag', () => { + const onUpdatePageState = vi.fn() + const view = render(paneElement(true, { onUpdatePageState })) + view.unmount() + onUpdatePageState.mockClear() + mocks.recordBreadcrumb.mockClear() + + webview.dispatchEvent(new Event('destroyed')) + + expect(onUpdatePageState).not.toHaveBeenCalled() + expect(mocks.recordBreadcrumb).not.toHaveBeenCalled() + }) + + it('survives activation focus after the retained tag left the DOM', () => { + const view = render(paneElement(false)) + webview.focus = vi.fn(() => { + throw nullContentWindowFocus() + }) + + expect(() => + act(() => { + view.rerender(paneElement(true)) + }) + ).not.toThrow() + expect(webview.focus).toHaveBeenCalled() + }) +}) diff --git a/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.tsx b/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.tsx index f4f698f516b..349886a570f 100644 --- a/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.tsx +++ b/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.tsx @@ -7,7 +7,10 @@ import type { } from '../../../../shared/browser-workspace-types' import { toHttpsRecoveryUrl } from '../../../../shared/browser-url' import type { RuntimeBrowserClientPlacement } from '../../../../shared/runtime-browser-placement' -import { readBrowserClientPageGuestMetadata } from './browser-client-page-guest-metadata' +import { + readBrowserClientPageGuestMetadataIfLive, + createBrowserClientPageLoadFailureHandler +} from './browser-client-page-guest-metadata' import { forgetBrowserClientPageMetadataReports, startBrowserClientPageMetadataPublisher @@ -18,6 +21,7 @@ import { useBrowserClientHostedPopupNotices } from './browser-client-hosted-popu import { useBrowserClientHostedPermissionNotices } from './browser-client-hosted-permission-notices' import { useClientHostedBrowserIntroTour } from './use-client-hosted-browser-intro-tour' import { ClientHostedBrowserUnavailableNotice } from './client-hosted-browser-unavailable-notice' +import { watchBrowserClientPageGuestLoss } from './host-guest/browser-client-page-guest-loss' import { useRestoredClientHostedRecoveryWindow } from './restored-client-hosted-recovery-window' import BrowserFind from './assemble-chrome/BrowserFind' import { BrowserNavigationControlRow } from './assemble-chrome/browser-navigation-control-row' @@ -36,7 +40,6 @@ import { BrowserLoadFailureOverlay } from './navigate/browser-load-failure-overl import { useClientHostedPageUrlSubmission } from './navigate/use-client-hosted-page-url-submission' import { convertBrowserPageToWorkspaceDoc } from '@/lib/file-preview' import { useBrowserPageReloadActions } from './navigate/use-browser-page-reload-actions' -import { resolveBrowserWebviewLoadFailure } from './navigate/browser-webview-load-failure' import { resolveActiveBrowserLoadFailure } from './navigate/browser-load-failure-for-url' import { consumeBrowserPageDeferredNavigation } from './navigate/browser-page-deferred-navigation' import { @@ -46,7 +49,6 @@ import { } from './describe-page/browser-page-url-display' import type { BrowserChromeShortcutScope, - BrowserPageFailLoadEvent, BrowserPageUrlSetter, BrowserTabPageState } from './describe-page/browser-page-types' @@ -174,9 +176,7 @@ export function ClientHostedBrowserPagePane({ useLayoutEffect(() => { const viewport = viewportRef.current - // Why: no placement means the host has not minted this page yet. Attaching would throw for an - // id the retained registry has never seen and strand the pane on the unavailable notice, whose - // only exit is reopening on the server — so mount quiet and wait for adoption to supply it. + // Wait for host adoption before attaching an optimistic page the registry has not seen. if ( !viewport || pageHostGeneration === null || @@ -200,6 +200,24 @@ export function ClientHostedBrowserPagePane({ return } const webview = attachment.webview + // Guest loss uses the existing recovery notice and clears pending loading state. + let releaseGuest = (): void => attachment.detach() + const guestLoss = watchBrowserClientPageGuestLoss({ + webview, + webviewRef, + browserPageId: browserTab.id, + pageHostGeneration, + onLost: () => { + releaseGuest() + retryGuestRecoveryRef.current() + } + }) + // Main can destroy the guest while its tag still holds the stale id. + const attachedMetadata = readBrowserClientPageGuestMetadataIfLive(webview) + if (!attachedMetadata) { + guestLoss.lose('unreadable') + return guestLoss.dispose() + } const publisher = startBrowserClientPageMetadataPublisher({ browserPageId: browserTab.id, environmentId: runtimeEnvironmentId, @@ -213,23 +231,21 @@ export function ClientHostedBrowserPagePane({ }) webviewRef.current = webview setAttachmentError(null) - // Why: the failure carried in from the store is hearsay — this pane may be remounting over a - // guest that navigated on while nothing was listening — so it is checked once against where - // the guest actually is. Failures this session observes are trusted as they arrive, because a - // navigation that fails outright often never commits and leaves the guest on the old URL. + // Reconcile restored failures once; failed navigations this session may never commit a URL. activeLoadFailureRef.current = resolveActiveBrowserLoadFailure( activeLoadFailureRef.current, - readBrowserClientPageGuestMetadata(webview).url + attachedMetadata.url ) const syncNavigation = (event?: Event): void => { const eventUrl = (event as (Event & { url?: string }) | undefined)?.url - const metadata = readBrowserClientPageGuestMetadata(webview, eventUrl) - // Why: did-stop-loading fires after did-fail-load, so an unconditional null here would - // wipe the failure the overlay is about to show. + const metadata = readBrowserClientPageGuestMetadataIfLive(webview, eventUrl) + if (!metadata) { + guestLoss.lose('unreadable') + return + } + // did-stop-loading must preserve the preceding did-fail-load overlay. const activeLoadFailure = activeLoadFailureRef.current - // Why: a URL write drops the page's certificate challenge by design (challenges are - // transient across navigation), so a standing failure must not run through one — the - // local pane returns before its own setUrl for the same reason. + // URL writes clear certificate challenges, so preserve them while a failure stands. if (!activeLoadFailure) { setUrlFromGuest(browserTab.id, metadata.url, { preserveLoadError: true @@ -243,26 +259,41 @@ export function ClientHostedBrowserPagePane({ loadError: activeLoadFailure }) publisher.publish(metadata) - // Why: the address bar's suggestions read the client's shared URL history, so a page - // hosted here has to file its navigations there like a local guest does. - recordHistoryFromGuest(metadata.url, getBrowserDisplayTitle(webview.getTitle(), metadata.url)) + // Address-bar suggestions use the client's URL history, including client-hosted pages. + recordHistoryFromGuest(metadata.url, getBrowserDisplayTitle(metadata.title, metadata.url)) setAddressBarValueFromPage(toDisplayUrl(metadata.url)) } const onStart = (): void => { activeLoadFailureRef.current = null updatePageStateFromGuest(browserTab.id, { loading: true, loadError: null }) - publisher.publish(readBrowserClientPageGuestMetadata(webview, undefined, true)) - } - const onFailLoad = (event: Event): void => { - const loadError = resolveBrowserWebviewLoadFailure(event as BrowserPageFailLoadEvent, { - fallbackUrl: webview.getURL() - }) - if (!loadError) { + const startMetadata = readBrowserClientPageGuestMetadataIfLive(webview, undefined, true) + if (!startMetadata) { + guestLoss.lose('unreadable') return } - activeLoadFailureRef.current = loadError - updatePageStateFromGuest(browserTab.id, { loading: false, loadError }) + publisher.publish(startMetadata) } + const onFailLoad = createBrowserClientPageLoadFailureHandler( + webview, + () => guestLoss.lose('unreadable'), + (loadError) => { + activeLoadFailureRef.current = loadError + updatePageStateFromGuest(browserTab.id, { loading: false, loadError }) + } + ) + const cleanupGuest = (): void => { + webview.removeEventListener('did-start-loading', onStart) + webview.removeEventListener('did-stop-loading', syncNavigation) + webview.removeEventListener('did-navigate', syncNavigation) + webview.removeEventListener('did-navigate-in-page', syncNavigation) + webview.removeEventListener('page-title-updated', syncNavigation) + webview.removeEventListener('did-fail-load', onFailLoad) + guestLoss.dispose() + publisher.dispose() + forgetBrowserClientPageMetadataReports(browserTab.id) + attachment.detach() + } + releaseGuest = cleanupGuest webview.addEventListener('did-start-loading', onStart) webview.addEventListener('did-stop-loading', syncNavigation) webview.addEventListener('did-navigate', syncNavigation) @@ -270,26 +301,12 @@ export function ClientHostedBrowserPagePane({ webview.addEventListener('page-title-updated', syncNavigation) webview.addEventListener('did-fail-load', onFailLoad) syncNavigation() - // Why: the user pressed Enter while this page was still an optimistic stage, so the navigation - // was parked rather than sent to a host page that did not exist yet. The guest exists now. + // Resume navigation submitted before host adoption. const deferredUrl = consumeBrowserPageDeferredNavigation(browserTab.id) if (deferredUrl) { runDeferredNavigation(deferredUrl) } - return () => { - webview.removeEventListener('did-start-loading', onStart) - webview.removeEventListener('did-stop-loading', syncNavigation) - webview.removeEventListener('did-navigate', syncNavigation) - webview.removeEventListener('did-navigate-in-page', syncNavigation) - webview.removeEventListener('page-title-updated', syncNavigation) - webview.removeEventListener('did-fail-load', onFailLoad) - if (webviewRef.current === webview) { - webviewRef.current = null - } - publisher.dispose() - forgetBrowserClientPageMetadataReports(browserTab.id) - attachment.detach() - } + return cleanupGuest }, [ browserTab.id, browserHostClientId, @@ -299,7 +316,7 @@ export function ClientHostedBrowserPagePane({ setAddressBarValueFromPage ]) - useClientHostedGuestActivationFocus({ isActive, webviewRef, keepAddressBarFocusRef }) + useClientHostedGuestActivationFocus({ isActive, guestFocus, keepAddressBarFocusRef }) const showFailureOverlay = !attachmentError && Boolean(browserTab.loadError) // Why: the failure is about the URL that failed, not whatever page is still loaded — feeding diff --git a/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPane.webview-preferences.test.ts b/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPane.webview-preferences.test.ts index 5e6dc77f67d..aff65e4eb3e 100644 --- a/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPane.webview-preferences.test.ts +++ b/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPane.webview-preferences.test.ts @@ -56,7 +56,7 @@ describe('BrowserPane webview preferences', () => { 'persist:orca-browser-session-profile-1' ) expect(ensuredWebview?.webview.getAttribute('webpreferences')).toBe( - ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE + `${ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE},transparent=false` ) expect(registryMocks.registerPersistentWebview).toHaveBeenCalledWith( 'browser-page-1', diff --git a/src/renderer/src/components/browser-pane/browser-client-page-guest-metadata.ts b/src/renderer/src/components/browser-pane/browser-client-page-guest-metadata.ts index 5e4446f978a..93e62de4dd7 100644 --- a/src/renderer/src/components/browser-pane/browser-client-page-guest-metadata.ts +++ b/src/renderer/src/components/browser-pane/browser-client-page-guest-metadata.ts @@ -1,24 +1,67 @@ +import type { BrowserLoadError } from '../../../../shared/browser-workspace-types' +import type { BrowserPageFailLoadEvent } from './describe-page/browser-page-types' +import { resolveBrowserWebviewLoadFailure } from './navigate/browser-webview-load-failure' +import { recordRendererCrashBreadcrumb } from '@/lib/crash-breadcrumb-recorder' import { redactKagiSessionToken } from '../../../../shared/browser-url' import type { BrowserClientPageMetadataSnapshot } from './browser-client-page-metadata-publisher' /** - * What a client-hosted guest currently is, read straight off the webview. + * What a client-hosted guest currently is, read straight off the webview, or null once the tag + * can no longer reach its guest. * * `eventUrl` wins when a navigation event carries one: the tag's own getURL() can still report the * previous page while the event is being delivered. `loading` is forced for did-start-loading, * which fires before isLoading() flips. + * + * Why total rather than throwing: a guest destroyed in main leaves the tag holding its id, so + * every method on it throws `Invalid guestInstanceId` from then on — and every caller reads from + * a React effect, where that unwinds the whole workbench error boundary. */ -export function readBrowserClientPageGuestMetadata( +export function readBrowserClientPageGuestMetadataIfLive( webview: Electron.WebviewTag, eventUrl?: string, loading?: boolean -): BrowserClientPageMetadataSnapshot { - const url = redactKagiSessionToken(eventUrl || webview.getURL() || 'about:blank') - return { - url, - title: webview.getTitle() || url || 'Browser', - loading: loading ?? webview.isLoading(), - canGoBack: webview.canGoBack(), - canGoForward: webview.canGoForward() +): BrowserClientPageMetadataSnapshot | null { + try { + const url = redactKagiSessionToken(eventUrl || webview.getURL() || 'about:blank') + return { + url, + title: webview.getTitle() || url || 'Browser', + loading: loading ?? webview.isLoading(), + canGoBack: webview.canGoBack(), + canGoForward: webview.canGoForward() + } + } catch (error) { + // Why recorded: the catch is total, so a read failure that is NOT guest death would otherwise + // be indistinguishable from one — the breadcrumb carries the error text the console cannot. + console.warn('[browser-client-page] guest read failed, treating the page as gone:', error) + recordRendererCrashBreadcrumb('browser_client_page_guest_read_failed', { + errorName: error instanceof Error ? error.name : typeof error, + errorMessage: error instanceof Error ? error.message : String(error) + }) + return null + } +} + +export function createBrowserClientPageLoadFailureHandler( + webview: Electron.WebviewTag, + onUnavailable: () => void, + onFailure: (error: BrowserLoadError) => void +): (event: Event) => void { + return (event) => { + let guestUnavailable = false + const loadError = resolveBrowserWebviewLoadFailure(event as BrowserPageFailLoadEvent, { + // Discarded ERR_ABORTED/subframe events must not read the guest. + fallbackUrl: () => { + const metadata = readBrowserClientPageGuestMetadataIfLive(webview) + guestUnavailable = metadata === null + return metadata?.url ?? null + } + }) + if (guestUnavailable) { + onUnavailable() + } else if (loadError) { + onFailure(loadError) + } } } diff --git a/src/renderer/src/components/browser-pane/host-guest/browser-client-page-guest-loss.ts b/src/renderer/src/components/browser-pane/host-guest/browser-client-page-guest-loss.ts new file mode 100644 index 00000000000..b9e2f6a21f6 --- /dev/null +++ b/src/renderer/src/components/browser-pane/host-guest/browser-client-page-guest-loss.ts @@ -0,0 +1,54 @@ +import type { MutableRefObject } from 'react' +import { recordRendererCrashBreadcrumb } from '@/lib/crash-breadcrumb-recorder' + +export type BrowserClientPageGuestLossReason = 'unreadable' | 'destroyed' | 'render-process-gone' + +/** + * Tells a client-hosted pane, once, that its guest is gone. The retained registry fences the tag on + * `destroyed` / `render-process-gone` without telling the pane, which would otherwise sit mute or + * spinning over a tag whose every method throws; a failed guest read is the same verdict. + */ +export function watchBrowserClientPageGuestLoss(options: { + webview: Electron.WebviewTag + /** Released on loss and dispose: every chrome action null-checks it, so a dead tag is never driven. */ + webviewRef: MutableRefObject + browserPageId: string + pageHostGeneration: number + onLost: () => void +}): { lose(reason: BrowserClientPageGuestLossReason): void; dispose(): void } { + const { webview } = options + const releaseWebviewRef = (): void => { + if (options.webviewRef.current === webview) { + options.webviewRef.current = null + } + } + let lost = false + const lose = (reason: BrowserClientPageGuestLossReason): void => { + if (lost) { + return + } + lost = true + // Why the breadcrumb: the crash report this replaces was the only field signal for guest death. + recordRendererCrashBreadcrumb('browser_client_page_guest_unavailable', { + browserPageId: options.browserPageId, + pageHostGeneration: options.pageHostGeneration, + reason, + tagConnected: webview.isConnected + }) + releaseWebviewRef() + options.onLost() + } + const onDestroyed = (): void => lose('destroyed') + const onRendererGone = (): void => lose('render-process-gone') + webview.addEventListener('destroyed', onDestroyed) + webview.addEventListener('render-process-gone', onRendererGone) + return { + lose, + dispose: () => { + lost = true + releaseWebviewRef() + webview.removeEventListener('destroyed', onDestroyed) + webview.removeEventListener('render-process-gone', onRendererGone) + } + } +} diff --git a/src/renderer/src/components/browser-pane/host-guest/browser-page-webview-surface.test.ts b/src/renderer/src/components/browser-pane/host-guest/browser-page-webview-surface.test.ts new file mode 100644 index 00000000000..2d156ed926c --- /dev/null +++ b/src/renderer/src/components/browser-pane/host-guest/browser-page-webview-surface.test.ts @@ -0,0 +1,93 @@ +// @vitest-environment happy-dom +import { afterEach, describe, expect, it, vi } from 'vitest' +import { ORCA_BROWSER_BLANK_URL } from '../../../../../shared/constants' +import { ensureBrowserPageWebview } from './browser-page-webview' +import { webviewRegistry } from './webview-registry' + +vi.mock('./webview-registry', () => { + const webviewRegistry = new Map() + return { + webviewRegistry, + registerPersistentWebview: vi.fn((id, guest) => webviewRegistry.set(id, guest)), + replacePersistentWebview: vi.fn(), + destroyPersistentWebview: vi.fn() + } +}) + +afterEach(() => { + document.body.replaceChildren() + webviewRegistry.clear() +}) + +function createGuest(): Electron.WebviewTag { + const container = document.createElement('div') + document.body.appendChild(container) + return ensureBrowserPageWebview({ + browserTabId: 'surface-test', + container, + inputLocked: false, + webviewPartition: 'persist:browser-test', + resolveContainer: () => container + })!.webview +} + +function commit(guest: Electron.WebviewTag, url: string, isMainFrame = true): void { + guest.dispatchEvent(Object.assign(new Event('load-commit'), { url, isMainFrame })) +} + +describe('browser page surface ownership', () => { + it('themes the host before attach and uses an opaque native canvas for real pages', () => { + const guest = createGuest() + expect(guest.style.background).toBe('var(--background)') + expect(guest.getAttribute('webpreferences')).toContain('transparent=false') + expect(guest.getAttribute('webpreferences')).toContain('disableHtmlFullscreenWindowResize=true') + }) + + it.each(['about:blank', ORCA_BROWSER_BLANK_URL])( + 'keeps %s unavailable through first navigation, then reveals the committed page', + (url) => { + const guest = createGuest() + commit(guest, url) + expect(guest.style.visibility).toBe('hidden') + guest.dispatchEvent(new Event('did-start-loading')) + expect(guest.style.visibility).toBe('hidden') + commit(guest, 'https://example.test') + expect(guest.style.visibility).toBe('visible') + guest.dispatchEvent(new Event('did-start-loading')) + expect(guest.style.visibility).toBe('visible') + commit(guest, 'about:blank', false) + expect(guest.style.visibility).toBe('visible') + } + ) + + it('preserves a reused guest and initializes the same surface after a container remount', () => { + const guest = createGuest() + commit(guest, 'https://example.test') + const container = guest.parentElement as HTMLDivElement + const reused = ensureBrowserPageWebview({ + browserTabId: 'surface-test', + container, + inputLocked: false, + webviewPartition: 'persist:browser-test', + resolveContainer: () => container + })! + expect(reused.created).toBe(false) + expect(reused.webview).toBe(guest) + expect(reused.webview.style.visibility).toBe('visible') + const replacement = createGuest() + expect(replacement).not.toBe(guest) + expect(replacement.style.background).toBe('var(--background)') + expect(replacement.getAttribute('webpreferences')).toContain('transparent=false') + commit(replacement, ORCA_BROWSER_BLANK_URL) + expect(replacement.style.visibility).toBe('hidden') + }) + + it('exposes the themed host after renderer loss until a recovered document commits', () => { + const guest = createGuest() + commit(guest, 'https://example.test') + guest.dispatchEvent(new Event('render-process-gone')) + expect(guest.style.visibility).toBe('hidden') + commit(guest, 'https://example.test') + expect(guest.style.visibility).toBe('visible') + }) +}) diff --git a/src/renderer/src/components/browser-pane/host-guest/browser-page-webview.ts b/src/renderer/src/components/browser-pane/host-guest/browser-page-webview.ts index 30e1cc18442..3c959751051 100644 --- a/src/renderer/src/components/browser-pane/host-guest/browser-page-webview.ts +++ b/src/renderer/src/components/browser-pane/host-guest/browser-page-webview.ts @@ -1,3 +1,4 @@ +import { ORCA_BROWSER_BLANK_URL } from '../../../../../shared/constants' import { ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE } from '../../../../../shared/browser-guest-web-preferences' import { destroyPersistentWebview, @@ -59,16 +60,29 @@ export function ensureBrowserPageWebview({ webview.setAttribute('allowpopups', '') // Why: Electron spreads the webpreferences keys verbatim, so the shared // camelCase attribute must stay intact for fullscreen containment to work. - webview.setAttribute('webpreferences', ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE) + // Keep Chromium's normal page canvas opaque while the host underneath follows Orca's theme. + webview.setAttribute( + 'webpreferences', + `${ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE},transparent=false` + ) webview.style.display = 'flex' webview.style.flex = '1' webview.style.width = '100%' webview.style.height = '100%' webview.style.border = 'none' setBrowserPageWebviewInputLock(webview, inputLocked) - // Why: some pages never paint a background, and a white viewport matches - // normal browser behavior instead of leaking Orca chrome through the guest. - webview.style.background = '#ffffff' + webview.style.background = 'var(--background)' + const guest = webview + // A committed synthetic blank document belongs to New Tab, including while its first URL waits. + guest.addEventListener('load-commit', (event) => { + if (event.isMainFrame) { + guest.style.visibility = + event.url === 'about:blank' || event.url === ORCA_BROWSER_BLANK_URL ? 'hidden' : 'visible' + } + }) + guest.addEventListener('render-process-gone', () => { + guest.style.visibility = 'hidden' + }) registerPersistentWebview(browserTabId, webview) activeContainer.appendChild(webview) created = true diff --git a/src/renderer/src/components/browser-pane/host-guest/use-client-hosted-guest-activation-focus.ts b/src/renderer/src/components/browser-pane/host-guest/use-client-hosted-guest-activation-focus.ts index 95494fb043e..ac9fcf87d8b 100644 --- a/src/renderer/src/components/browser-pane/host-guest/use-client-hosted-guest-activation-focus.ts +++ b/src/renderer/src/components/browser-pane/host-guest/use-client-hosted-guest-activation-focus.ts @@ -1,4 +1,5 @@ import { useEffect, useRef, type RefObject } from 'react' +import type { BrowserPageGuestFocus } from '../assemble-chrome/browser-page-guest-focus' import { useWebviewDragPassthroughActive } from './use-webview-drag-passthrough-active' /** @@ -11,11 +12,12 @@ import { useWebviewDragPassthroughActive } from './use-webview-drag-passthrough- */ export function useClientHostedGuestActivationFocus({ isActive, - webviewRef, + guestFocus, keepAddressBarFocusRef }: { isActive: boolean - webviewRef: RefObject + /** Not the raw tag: a retired page's is out of the DOM, where focus() throws (STA-3448). */ + guestFocus: BrowserPageGuestFocus keepAddressBarFocusRef: RefObject }): void { const dragPassthroughActive = useWebviewDragPassthroughActive() @@ -41,6 +43,6 @@ export function useClientHostedGuestActivationFocus({ if (keepAddressBarFocusRef.current) { return } - webviewRef.current?.focus() - }, [dragPassthroughActive, isActive, keepAddressBarFocusRef, webviewRef]) + guestFocus.focus() + }, [dragPassthroughActive, guestFocus, isActive, keepAddressBarFocusRef]) } diff --git a/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.test.ts b/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.test.ts index 44e59fdbedd..1595b69f6aa 100644 --- a/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.test.ts +++ b/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, it } from 'vitest' +import { describe, expect, it, vi } from 'vitest' import { resolveBrowserWebviewLoadFailure } from './browser-webview-load-failure' describe('resolveBrowserWebviewLoadFailure', () => { @@ -49,6 +49,18 @@ describe('resolveBrowserWebviewLoadFailure', () => { ).toMatchObject({ validatedUrl: 'https://example.com/current' }) }) + it('never reads a lazy fallback URL for an event it discards', () => { + const fallbackUrl = vi.fn(() => 'https://example.com/current') + expect(resolveBrowserWebviewLoadFailure({ errorCode: -3 }, { fallbackUrl })).toBeNull() + expect(fallbackUrl).not.toHaveBeenCalled() + expect( + resolveBrowserWebviewLoadFailure( + { errorCode: -105, errorDescription: 'ERR_NAME_NOT_RESOLVED', validatedURL: '' }, + { fallbackUrl } + ) + ).toMatchObject({ validatedUrl: 'https://example.com/current' }) + }) + it('keeps a usable description when Chromium reports an empty one', () => { expect( resolveBrowserWebviewLoadFailure({ errorCode: -105, errorDescription: '' }) diff --git a/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.ts b/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.ts index c39c5afe53b..03228c3e1d6 100644 --- a/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.ts +++ b/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.ts @@ -8,20 +8,23 @@ import type { BrowserPageFailLoadEvent } from '../describe-page/browser-page-typ * cannot forget the ignore rules or build a differently-shaped BrowserLoadError. * * `fallbackUrl` covers failures that arrive without a validatedURL — pass the webview's - * current URL so the overlay names the page instead of about:blank. + * current URL so the overlay names the page instead of about:blank. Pass it as a function when + * reading it costs anything: discarded events never ask for it. */ export function resolveBrowserWebviewLoadFailure( event: BrowserPageFailLoadEvent, - options: { fallbackUrl?: string | null } = {} + options: { fallbackUrl?: string | null | (() => string | null) } = {} ): BrowserLoadError | null { // Why: Chromium reports redirect/cancel races as ERR_ABORTED (-3) even when the // replacement navigation succeeds; subframe failures never blank the page. if (event.isMainFrame === false || event.errorCode === -3) { return null } + const fallbackUrl = + typeof options.fallbackUrl === 'function' ? options.fallbackUrl() : options.fallbackUrl return { code: event.errorCode ?? -1, description: event.errorDescription || 'Unknown load failure', - validatedUrl: redactKagiSessionToken(event.validatedURL || options.fallbackUrl || 'about:blank') + validatedUrl: redactKagiSessionToken(event.validatedURL || fallbackUrl || 'about:blank') } } diff --git a/src/renderer/src/components/cmd-j/native-chat-split-quick-actions.ts b/src/renderer/src/components/cmd-j/native-chat-split-quick-actions.ts new file mode 100644 index 00000000000..d2019fb429a --- /dev/null +++ b/src/renderer/src/components/cmd-j/native-chat-split-quick-actions.ts @@ -0,0 +1,61 @@ +import { PanelBottomClose, PanelRightClose } from 'lucide-react' +import { translate } from '@/i18n/i18n' +import type { CmdJQuickAction } from './quick-actions' +import type { CmdJQuickActionContext } from './quick-action-context' +import type { NativeChatSplitDirection } from '@/components/native-chat/native-chat-split-shortcut' + +function availability(ctx: CmdJQuickActionContext) { + return ctx.canSplitActiveChat + ? ({ available: true } as const) + : ({ available: false, reason: 'no-active-chat' } as const) +} + +function splitAction( + direction: NativeChatSplitDirection, + action: Pick +): CmdJQuickAction { + return { + ...action, + kind: 'action', + isAvailable: availability, + run: async (ctx) => { + if (!availability(ctx).available || !ctx.splitActiveChat?.(direction)) { + return { status: 'unavailable', reason: 'no-active-chat' } + } + return { status: 'ok' } + } + } +} + +export function getNativeChatSplitQuickActions(): CmdJQuickAction[] { + return [ + splitAction('right', { + id: 'split-chat-right', + title: translate('auto.components.cmd.j.quick.actions.splitChatRight', 'Split Chat Right'), + description: translate( + 'auto.components.cmd.j.quick.actions.splitChatRightDescription', + 'Open the active chat in a split pane to the right.' + ), + icon: PanelRightClose, + verbKeywords: [ + translate('auto.components.cmd.j.quick.actions.verbs.splitChatRight', 'split chat right'), + translate('auto.components.cmd.j.quick.actions.verbs.moveChatRight', 'move chat right'), + translate('auto.components.cmd.j.quick.actions.verbs.chatPaneRight', 'chat pane right') + ] + }), + splitAction('down', { + id: 'split-chat-down', + title: translate('auto.components.cmd.j.quick.actions.splitChatDown', 'Split Chat Down'), + description: translate( + 'auto.components.cmd.j.quick.actions.splitChatDownDescription', + 'Open the active chat in a split pane below.' + ), + icon: PanelBottomClose, + verbKeywords: [ + translate('auto.components.cmd.j.quick.actions.verbs.splitChatDown', 'split chat down'), + translate('auto.components.cmd.j.quick.actions.verbs.moveChatDown', 'move chat down'), + translate('auto.components.cmd.j.quick.actions.verbs.chatPaneBelow', 'chat pane below') + ] + }) + ] +} diff --git a/src/renderer/src/components/cmd-j/quick-action-context.test.ts b/src/renderer/src/components/cmd-j/quick-action-context.test.ts index 325252c4dc2..843634309df 100644 --- a/src/renderer/src/components/cmd-j/quick-action-context.test.ts +++ b/src/renderer/src/components/cmd-j/quick-action-context.test.ts @@ -254,6 +254,62 @@ describe('Cmd+J quick action context', () => { expect(context.isLoading).toBe(true) }) + it('exposes chat split actions only on the workspace surface', () => { + const worktree = { + id: 'wt-1', + repoId: 'repo-1', + path: '/repo/wt', + displayName: 'Workspace', + branch: 'main', + createdAt: 0 + } as Worktree + const state = { + activeWorktreeId: 'wt-1', + worktreesByRepo: { 'repo-1': [worktree] }, + repos: [{ id: 'repo-1', path: '/repo', displayName: 'Repo', addedAt: 0 }], + sshConnectionStates: new Map(), + activeGroupIdByWorktree: { 'wt-1': 'group-1' }, + groupsByWorktree: { + 'wt-1': [ + { + id: 'group-1', + worktreeId: 'wt-1', + activeTabId: 'chat-1', + tabOrder: ['chat-1', 'other-1'] + } + ] + }, + unifiedTabsByWorktree: { + 'wt-1': [ + { + id: 'chat-1', + entityId: 'session-1', + groupId: 'group-1', + worktreeId: 'wt-1', + contentType: 'agent-session' + } + ] + }, + activeView: 'settings', + settings: null + } as unknown as AppState + + const buildContext = (activeView: AppState['activeView']) => + buildCmdJQuickActionContext({ + state: { ...state, activeView }, + activeGroupSnapshot: null, + openNewBrowserTab: async () => {}, + openNewMarkdownFile: async () => {}, + openNewTerminalTab: async () => {}, + openCreateWorkspace: () => {}, + deleteActiveWorkspace: () => {}, + openAddQuickCommand: () => {} + }) + + expect(buildContext('terminal').canSplitActiveChat).toBe(true) + expect(buildContext('settings').canSplitActiveChat).toBe(false) + }) + it('runtime re-check returns unavailable without invoking the action helper', async () => { const calls: string[] = [] const action = getCmdJQuickActions().find((entry) => entry.id === 'new-terminal-tab') @@ -335,4 +391,33 @@ describe('Cmd+J quick action context', () => { await expect(action?.run(context)).resolves.toEqual({ status: 'ok' }) expect(calls).toEqual(['delete']) }) + + it('offers and runs split actions only for an active movable chat', async () => { + const calls: string[] = [] + const action = getCmdJQuickActions().find((entry) => entry.id === 'split-chat-right') + const context = { + ...ctx({}), + activeWorktree: null, + runtimeMode: 'local-desktop' as const, + openNewBrowserTab: async () => {}, + openNewMarkdownFile: async () => {}, + openNewTerminalTab: async () => {}, + openCreateWorkspace: () => {}, + deleteActiveWorkspace: () => {}, + openAddQuickCommand: () => {}, + canSplitActiveChat: true, + splitActiveChat: (direction: string) => { + calls.push(direction) + return true + } + } satisfies CmdJQuickActionContext + + expect(action?.isAvailable(context)).toEqual({ available: true }) + await expect(action?.run(context)).resolves.toEqual({ status: 'ok' }) + expect(calls).toEqual(['right']) + expect(action?.isAvailable({ ...context, canSplitActiveChat: false })).toEqual({ + available: false, + reason: 'no-active-chat' + }) + }) }) diff --git a/src/renderer/src/components/cmd-j/quick-action-context.ts b/src/renderer/src/components/cmd-j/quick-action-context.ts index 88a8cfad533..bbed9671a4e 100644 --- a/src/renderer/src/components/cmd-j/quick-action-context.ts +++ b/src/renderer/src/components/cmd-j/quick-action-context.ts @@ -3,12 +3,19 @@ import { findWorktreeById } from '@/store/slices/worktree-helpers' import type { Worktree } from '../../../../shared/worktree/types' import type { SshConnectionStatus } from '../../../../shared/ssh-types' import { getClientCreationActionPolicy } from '@/lib/client-creation-action-policy' +import { + canRunNativeChatSplitTarget, + resolveActiveNativeChatSplitTarget, + runActiveNativeChatSplit +} from '@/components/native-chat/native-chat-layout-actions' +import type { NativeChatSplitDirection } from '@/components/native-chat/native-chat-split-shortcut' export type CmdJUnavailableReason = | 'loading' | 'no-active-workspace' | 'ssh-disconnected' | 'no-active-group' + | 'no-active-chat' | 'client-action-unsupported' export type CmdJQuickActionAvailability = @@ -35,6 +42,8 @@ export type CmdJQuickActionContext = { openCreateWorkspace: () => void deleteActiveWorkspace: () => void openAddQuickCommand: () => void + canSplitActiveChat?: boolean + splitActiveChat?: (direction: NativeChatSplitDirection) => boolean } export function resolveCmdJActiveGroupId( @@ -166,6 +175,10 @@ export function buildCmdJQuickActionContext(args: { const managedBrowserCreationEnabled = getClientCreationActionPolicy(args.state, activeWorktreeId)['managed-browser'].state === 'enabled' + const activeChatTarget = + args.state.activeView === 'terminal' + ? resolveActiveNativeChatSplitTarget(args.state, activeWorktreeId, activeGroupId) + : null return { activeView: args.state.activeView, @@ -181,7 +194,10 @@ export function buildCmdJQuickActionContext(args: { openNewTerminalTab: args.openNewTerminalTab, openCreateWorkspace: args.openCreateWorkspace, deleteActiveWorkspace: args.deleteActiveWorkspace, - openAddQuickCommand: args.openAddQuickCommand + openAddQuickCommand: args.openAddQuickCommand, + canSplitActiveChat: canRunNativeChatSplitTarget(args.state, activeChatTarget), + splitActiveChat: (direction) => + runActiveNativeChatSplit(activeWorktreeId, activeGroupId, direction) } } @@ -198,6 +214,8 @@ export function getUnavailableQuickActionMessage( return `Can't ${actionTitle.toLowerCase()} — workspace is disconnected.` case 'no-active-group': return `Can't ${actionTitle.toLowerCase()} — no tab group is available.` + case 'no-active-chat': + return `Can't ${actionTitle.toLowerCase()} — no movable chat is active.` case 'client-action-unsupported': return `Can't ${actionTitle.toLowerCase()} — this client and runtime do not support it.` } diff --git a/src/renderer/src/components/cmd-j/quick-actions.ts b/src/renderer/src/components/cmd-j/quick-actions.ts index 70f0e5f6859..00ccf127bad 100644 --- a/src/renderer/src/components/cmd-j/quick-actions.ts +++ b/src/renderer/src/components/cmd-j/quick-actions.ts @@ -8,6 +8,7 @@ import { } from './quick-action-context' import { translate } from '@/i18n/i18n' import { createLocalizedCatalog } from '@/i18n/localized-catalog' +import { getNativeChatSplitQuickActions } from './native-chat-split-quick-actions' export type CmdJQuickActionRunResult = | { status: 'ok' } @@ -125,6 +126,7 @@ export const getCmdJQuickActions = createLocalizedCatalog((): CmdJQuickAction[] isAvailable: workspaceActionAvailability, run: (ctx) => runWorkspaceAction(ctx, ctx.openNewTerminalTab) }, + ...getNativeChatSplitQuickActions(), { id: CREATE_WORKSPACE_QUICK_ACTION_ID, kind: 'action', diff --git a/src/renderer/src/components/confirmation-dialog-context.ts b/src/renderer/src/components/confirmation-dialog-context.ts index 4675c181fb7..b112191a4e6 100644 --- a/src/renderer/src/components/confirmation-dialog-context.ts +++ b/src/renderer/src/components/confirmation-dialog-context.ts @@ -1,4 +1,5 @@ import { createContext, useContext } from 'react' +import type { LucideIcon } from 'lucide-react' // Keep the context component-free so Fast Refresh preserves its identity. @@ -9,6 +10,9 @@ export type ConfirmationDialogOptions = { confirmLabel?: string cancelLabel?: string confirmVariant?: 'default' | 'destructive' + icon?: LucideIcon + cancelVariant?: 'outline' | 'ghost' + initialFocus?: 'confirm' /** Renders a "Don't ask again" checkbox. `onConfirmed` runs only when the user confirms with it checked. */ dontAskAgain?: { label?: string; onConfirmed: () => void } } diff --git a/src/renderer/src/components/confirmation-dialog.test.tsx b/src/renderer/src/components/confirmation-dialog.test.tsx index 25738ee4c17..5097a5d7786 100644 --- a/src/renderer/src/components/confirmation-dialog.test.tsx +++ b/src/renderer/src/components/confirmation-dialog.test.tsx @@ -44,6 +44,37 @@ function renderDialog(options: ConfirmationDialogOptions): { onSettled: ReturnTy describe('ConfirmationDialogProvider', () => { afterEach(cleanup) + it('focuses the primary action when requested and confirms with Enter', async () => { + const { onSettled } = renderDialog({ + title: 'Reveal hidden workspace?', + confirmLabel: 'Clear filters and reveal', + initialFocus: 'confirm' + }) + await userEvent.click(screen.getByRole('button', { name: 'ask' })) + await waitFor(() => + expect(screen.getByRole('button', { name: 'Clear filters and reveal' })).toHaveFocus() + ) + await userEvent.keyboard('{Enter}') + await waitFor(() => expect(onSettled).toHaveBeenCalledWith(true)) + }) + + it('still cancels with Escape when the primary action has focus', async () => { + const { onSettled } = renderDialog({ + title: 'Reveal hidden workspace?', + initialFocus: 'confirm' + }) + await userEvent.click(screen.getByRole('button', { name: 'ask' })) + await waitFor(() => expect(screen.getByRole('button', { name: 'Confirm' })).toHaveFocus()) + await userEvent.keyboard('{Escape}') + await waitFor(() => expect(onSettled).toHaveBeenCalledWith(false)) + }) + + it('keeps the default cancel focus for callers that do not opt in', async () => { + renderDialog({ title: 'Delete artifact?', confirmVariant: 'destructive' }) + await userEvent.click(screen.getByRole('button', { name: 'ask' })) + await waitFor(() => expect(screen.getByRole('button', { name: 'Cancel' })).toHaveFocus()) + }) + it('omits the checkbox unless the caller opts in', async () => { renderDialog({ title: 'Delete artifact?' }) diff --git a/src/renderer/src/components/confirmation-dialog.tsx b/src/renderer/src/components/confirmation-dialog.tsx index a1670b9bb22..615a2400bc6 100644 --- a/src/renderer/src/components/confirmation-dialog.tsx +++ b/src/renderer/src/components/confirmation-dialog.tsx @@ -32,6 +32,7 @@ export function ConfirmationDialogProvider({ children: React.ReactNode }): React.JSX.Element { const nextIdRef = useRef(0) + const confirmButtonRef = useRef(null) const [queue, setQueue] = useState([]) const [dontAskAgain, setDontAskAgain] = useState(false) const activeRequest = queue[0] ?? null @@ -46,6 +47,7 @@ export function ConfirmationDialogProvider({ } // Why: Radix keeps dialog content mounted while closing; keep labels stable without a post-render Effect. const displayedRequest = activeRequest ?? lastDisplayedRequestRef.current + const Icon = displayedRequest?.options.icon useEffect(() => { // Why: this provider's dialog is not represented by activeModal. Block @@ -96,18 +98,37 @@ export function ConfirmationDialogProvider({ open={activeRequest !== null} onOpenChange={(open) => !open && settleActiveRequest(false)} > - - - {displayedRequest?.options.title} - {displayedRequest?.options.description ? ( - // Callers pass multi-line descriptions (e.g. one path per line). - - {displayedRequest.options.description} - - ) : null} - + { + if (activeRequest?.options.initialFocus === 'confirm') { + event.preventDefault() + confirmButtonRef.current?.focus() + } + }} + > +
+ {Icon && ( +
+
+ )} + + {displayedRequest?.options.title} + {displayedRequest?.options.description ? ( + // Callers pass multi-line descriptions (e.g. one path per line). + + {displayedRequest.options.description} + + ) : null} + +
{displayedRequest?.options.dontAskAgain ? (
) : null} - - + + + {expanded ? ( +
+ {props.tasks.length > 0 ? ( +
    + {props.tasks.map((task) => ( +
  • +
  • + ))} +
+ ) : ( +

+ {translate( + 'components.native-chat.backgroundTasks.detailsUnavailable', + 'Task details are unavailable for this session.' + )} +

+ )} +
+ ) : null} + + + ) +} diff --git a/src/renderer/src/components/native-chat/NativeChatComposer.test.tsx b/src/renderer/src/components/native-chat/NativeChatComposer.test.tsx index a1aecd23419..a6f233b96a3 100644 --- a/src/renderer/src/components/native-chat/NativeChatComposer.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatComposer.test.tsx @@ -6,7 +6,6 @@ import type { SessionOptionDescriptor, SessionOptionsSurface } from '../../../../shared/native-chat-session-options' -import type * as nativeChatAgentProfiles from '../../../../shared/native-chat-agent-profiles' import { clearNativeChatSessionOptionCacheForTests } from './native-chat-session-option-cache' import { clearNativeChatModelEnrichmentForTests } from './native-chat-session-option-enrichment' @@ -27,6 +26,7 @@ const mocks = vi.hoisted(() => ({ sessionOptionsSnapshot?: SessionOptionDescriptor[] attachDisabled?: boolean sendButtonDisabled?: boolean + autocomplete?: { mode: string; items?: { kind: string; name: string }[] } } | null, modelSwitchOutcome: 'applied' as 'applied' | 'rejected' | 'unknown', confirmationObserver: null as { @@ -90,10 +90,6 @@ vi.mock('./claude-model-switch-confirmation', () => ({ createClaudeModelSwitchConfirmationObserver: (...args: unknown[]) => mocks.createClaudeModelSwitchConfirmationObserver(...args) })) -vi.mock('../../../../shared/native-chat-agent-profiles', async (importOriginal) => ({ - ...(await importOriginal()), - getVerifiedNativeChatCommands: () => [] -})) vi.mock('@/lib/native-chat-telemetry', () => ({ emitNativeChatMessageSent: vi.fn(), emitNativeChatPickerItemAccepted: vi.fn(), @@ -308,6 +304,43 @@ describe('NativeChatComposer', () => { expect(mocks.setDraft).toHaveBeenCalledWith('') }) + // The structured slash menu must offer the running agent's own catalog. Offering + // another agent's tokens sends them past the command guard as literal prompt text. + it.each([ + ['claude', 'compact', 'vim'], + ['codex', 'vim', 'help'] + ] as const)('offers %s its own structured slash commands', (agent, offered, withheld) => { + mocks.draft = '/' + render( + true), + dispatchCommand: vi.fn(async () => ({ handled: false, accepted: false, error: null })), + optionsSurface: { + getSnapshot: () => [], + setOption: vi.fn(), + invokeAction: vi.fn(), + subscribe: () => () => {} + }, + optionSnapshot: [], + onError: vi.fn(), + runtime: 'local' + }} + /> + ) + + const names = (mocks.fieldProps?.autocomplete?.items ?? []) + .filter((item) => item.kind === 'command') + .map((item) => item.name) + expect(names).toContain(offered) + expect(names).toContain('effort') + expect(names).not.toContain(withheld) + }) + it('sends structured image attachments through the durable transport', async () => { mocks.draft = '' mocks.imageAttachments = [{ id: 'image-1', path: '/tmp/image.png' }] diff --git a/src/renderer/src/components/native-chat/NativeChatComposer.tsx b/src/renderer/src/components/native-chat/NativeChatComposer.tsx index e0b1d97a057..06ae0ec5c8c 100644 --- a/src/renderer/src/components/native-chat/NativeChatComposer.tsx +++ b/src/renderer/src/components/native-chat/NativeChatComposer.tsx @@ -3,7 +3,7 @@ import { useAppStore } from '../../store' import { sendRuntimePtyInput } from '@/runtime/runtime-terminal-inspection' import { getSettingsForAgentTabRuntimeOwner } from '@/lib/agent-paste-draft' import { getVerifiedNativeChatCommands } from '../../../../shared/native-chat-agent-profiles' -import { STRUCTURED_AGENT_SESSION_SLASH_COMMANDS } from '../../../../shared/structured-agent-session-composer' +import { structuredSlashCommands } from '../../../../shared/structured-agent-session-composer' import { applyMentionSuggestion, EMPTY_HISTORY, @@ -111,9 +111,7 @@ const NativeChatComposerPane = forwardRef - structuredTransport - ? STRUCTURED_AGENT_SESSION_SLASH_COMMANDS - : getVerifiedNativeChatCommands(agent), + structuredTransport ? structuredSlashCommands(agent) : getVerifiedNativeChatCommands(agent), [agent, structuredTransport] ) const picker = useNativeChatPickerState({ diff --git a/src/renderer/src/components/native-chat/NativeChatCopyButton.tsx b/src/renderer/src/components/native-chat/NativeChatCopyButton.tsx index 36fa791f57d..0e933584d41 100644 --- a/src/renderer/src/components/native-chat/NativeChatCopyButton.tsx +++ b/src/renderer/src/components/native-chat/NativeChatCopyButton.tsx @@ -12,9 +12,12 @@ import { translate } from '@/i18n/i18n' */ export function NativeChatCopyButton({ text, + label: copyLabel, className }: { text: string + /** What this button copies, when it is not the whole message. */ + label?: string className?: string }): React.JSX.Element { const [copied, setCopied] = useState(false) @@ -46,7 +49,7 @@ export function NativeChatCopyButton({ const label = copied ? translate('components.native-chat.copyMessage.copied', 'Copied') - : translate('components.native-chat.copyMessage.copy', 'Copy message') + : (copyLabel ?? translate('components.native-chat.copyMessage.copy', 'Copy message')) return ( +
+ {file.oldPath ? ( + <> + + {baseName(file.oldPath)} + + → + + ) : null} + + {baseName(file.path)} + + + {file.truncated ? ( + // Beside the counts rather than under the rows: a collapsed card, and + // one clipped down to no rows at all, would otherwise say nothing. + + {translate('components.native-chat.tool.diffTruncated', 'Diff truncated')} + + ) : null} + +
+ {hasBody && expanded ? ( + // Focusable so the rows can be scrolled from the keyboard. +
+ {(() => { + const seen = new Map() + return file.lines.map((line) => { + const signature = `${line.kind}:${line.oldLineNumber}:${line.newLineNumber}:${line.text}` + const occurrence = seen.get(signature) ?? 0 + seen.set(signature, occurrence + 1) + return ( + + ) + }) + })()} +
+ ) : null} + + ) +} diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.stream-render.perf.test.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.stream-render.perf.test.tsx new file mode 100644 index 00000000000..361045b4642 --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.stream-render.perf.test.tsx @@ -0,0 +1,91 @@ +// @vitest-environment happy-dom + +import '@testing-library/jest-dom/vitest' + +import { cleanup, render } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type * as NativeChatProseModule from './native-chat-prose' +import type { NativeChatMessage } from '../../../../shared/native-chat-types' +import type { NativeChatLiveSession } from './use-native-chat-live-session' + +// Counting real per-row work rather than a render counter: a future refactor could keep the +// render count low while still re-deriving every row's markdown. +const proseCalls = vi.hoisted(() => ({ count: 0 })) +vi.mock('./native-chat-prose', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + nativeChatProseToMarkdown: (prose: Parameters[0]) => { + proseCalls.count += 1 + return actual.nativeChatProseToMarkdown(prose) + } + } +}) + +const { NativeChatMessageList } = await import('./NativeChatMessageList') + +afterEach(cleanup) + +const TRANSCRIPT_LENGTH = 120 + +function settledMessages(): NativeChatMessage[] { + return Array.from({ length: TRANSCRIPT_LENGTH }, (_, index) => ({ + id: `message-${index}`, + role: index % 2 === 0 ? ('user' as const) : ('assistant' as const), + blocks: [{ type: 'text' as const, text: `settled line ${index}` }], + timestamp: index + 1, + source: 'transcript' as const + })) +} + +function sessionWith(messages: NativeChatMessage[]): NativeChatLiveSession { + return { + messages, + status: 'ready', + sessionId: 'session-1', + agent: 'codex', + hasMore: false, + loadingEarlier: false, + loadEarlier: vi.fn(), + readPhase: 'ready' + } +} + +describe('native chat transcript re-render cost during a streaming turn', () => { + it('rebuilds only the rows whose blocks changed, not the whole transcript per frame', () => { + const messages = settledMessages() + const { rerender } = render( + + ) + + const afterFirstPaint = proseCalls.count + expect(afterFirstPaint).toBeGreaterThanOrEqual(TRANSCRIPT_LENGTH) + + // A streaming turn publishes a frame per SDK event; only the tail message's blocks change. + const STREAM_FRAMES = 20 + for (let frame = 1; frame <= STREAM_FRAMES; frame += 1) { + const streaming = messages.slice(0, -1).concat({ + ...messages.at(-1)!, + blocks: [{ type: 'text' as const, text: `streaming token ${frame}` }] + }) + rerender( + + ) + } + + const perFrame = (proseCalls.count - afterFirstPaint) / STREAM_FRAMES + // Without row memoization every settled row rebuilt its markdown on every frame. Settled + // rows keep their block identity, so only the streaming tail should rebuild. + expect(perFrame).toBeLessThan(TRANSCRIPT_LENGTH / 10) + }) +}) diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.test.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.test.tsx index 9695317f3af..860464ac65b 100644 --- a/src/renderer/src/components/native-chat/NativeChatMessageList.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.test.tsx @@ -215,7 +215,7 @@ describe('NativeChatMessageList assistant messages', () => { ) const user = screen.getByText('Run the checks') - const status = screen.getByText('Working for 0 seconds') + const status = screen.getByText('Working for 0s') const assistant = screen.getByText('I am checking now.') expect(user.compareDocumentPosition(status)).toBe(Node.DOCUMENT_POSITION_FOLLOWING) expect(status.compareDocumentPosition(assistant)).toBe(Node.DOCUMENT_POSITION_FOLLOWING) @@ -252,7 +252,7 @@ describe('NativeChatMessageList assistant messages', () => { /> ) - expect(screen.getByText('Working for 3 seconds')).toBeInTheDocument() + expect(screen.getByText('Working for 3s')).toBeInTheDocument() }) it('keeps the completed duration below the user message', () => { @@ -298,7 +298,7 @@ describe('NativeChatMessageList assistant messages', () => { ) const user = screen.getByText('Complete this task') - const status = screen.getByText('Worked for 3 seconds') + const status = screen.getByText('Worked for 3s') const assistant = screen.getByText('Task complete.') expect(user.compareDocumentPosition(status)).toBe(Node.DOCUMENT_POSITION_FOLLOWING) expect(status.compareDocumentPosition(assistant)).toBe(Node.DOCUMENT_POSITION_FOLLOWING) @@ -326,7 +326,7 @@ describe('NativeChatMessageList assistant messages', () => { /> ) - expect(screen.getByText('Worked for 3 seconds')).toBeInTheDocument() + expect(screen.getByText('Worked for 3s')).toBeInTheDocument() expect(screen.getByText('Thinking')).toBeInTheDocument() }) diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.tsx index 12b10e0f714..debcfb1c94b 100644 --- a/src/renderer/src/components/native-chat/NativeChatMessageList.tsx +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.tsx @@ -1,27 +1,17 @@ import { Fragment, useCallback, useEffect, useLayoutEffect, useMemo, useRef, useState } from 'react' import { ArrowDown } from 'lucide-react' -import CommentMarkdown, { - type CommentMarkdownLinkClickHandler -} from '@/components/sidebar/CommentMarkdown' -import { cn } from '@/lib/utils' +import type { CommentMarkdownLinkClickHandler } from '@/components/sidebar/CommentMarkdown' import { translate } from '@/i18n/i18n' -import type { NativeChatMessage } from '../../../../shared/native-chat-types' import type { NativeChatLiveSession } from './use-native-chat-live-session' import { orderNativeChatMessages } from './native-chat-message-grouping' import { stripNoiseMessages } from './native-chat-noise' -import { foldToolMessages, splitNativeChatBlocks } from './native-chat-tool-fold' +import { foldToolMessages } from './native-chat-tool-fold' import { isNearBottom, shouldShowJumpToLatest, type ScrollGeometry } from './native-chat-autoscroll' -import { NativeChatToolRun } from './NativeChatToolRun' +import { MessageRow } from './NativeChatMessageRow' import { shouldShowNativeChatTypingIndicator } from './native-chat-typing-indicator' import { NativeChatWorkingStatus } from './NativeChatWorkingStatus' import { useNativeChatTurnStatus } from './use-native-chat-turn-status' -import { nativeChatProseToMarkdown } from './native-chat-prose' import { NativeChatTypingIndicatorRow } from './NativeChatTypingIndicatorRow' -import { - NativeChatAgentControls, - NativeChatImageAttachments, - ProviderFrameRow -} from './NativeChatTranscriptChrome' import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' export { ProviderFrameRow } from './NativeChatTranscriptChrome' @@ -32,152 +22,6 @@ function geometryOf(el: HTMLElement): ScrollGeometry { const MAX_EXPANDED_TURNS = 128 -/** One message: its prose first, then a collapsible run folding all of the - * turn's tool activity. Monochrome per STYLEGUIDE: user prompts read as a - * lifted card, assistant prose as body copy, reasoning de-emphasized. */ -function MessageRow({ - message, - expandSignal, - activeTurnIsWorking, - onScrollMessageToTop, - onLinkClick, - allowFileUriLinks = false, - deliveryFailed = false, - activityExpandOverride, - structuredActivityUi = true, - runtimeContext -}: { - message: NativeChatMessage - expandSignal: boolean - activeTurnIsWorking?: boolean - /** Align this message's top to the top of the scroll viewport. */ - onScrollMessageToTop: (el: HTMLElement) => void - onLinkClick?: CommentMarkdownLinkClickHandler - allowFileUriLinks?: boolean - deliveryFailed?: boolean - activityExpandOverride?: boolean - structuredActivityUi?: boolean - runtimeContext?: RuntimeFileOperationArgs | null -}): React.JSX.Element | null { - const rowRef = useRef(null) - const { prose, tools } = useMemo(() => splitNativeChatBlocks(message.blocks), [message.blocks]) - const markdown = nativeChatProseToMarkdown(prose) - const hasImages = prose.some((block) => block.type === 'image-ref') - const isUser = message.role === 'user' - const isReasoning = message.role === 'reasoning' - const isSystem = message.role === 'system' - const providerFrame = message.blocks.find((block) => block.type === 'text' && block.providerFrame) - - const scrollToTop = useCallback(() => { - if (rowRef.current) { - onScrollMessageToTop(rowRef.current) - } - }, [onScrollMessageToTop]) - - // Skip rows with nothing renderable so the transcript shows no empty/ghost - // bubble. - // After all hooks, so hook order stays unconditional. - if (markdown.length === 0 && !hasImages && tools.length === 0) { - return null - } - - if (providerFrame) { - return ( -
- -
- ) - } - - if (isUser) { - return ( -
- {/* User turns get a distinct muted fill (not the card/canvas color) so - the prompt reads apart from the assistant's body copy. */} -
- {markdown ? ( - <> - - - - ) : ( - - )} -
- {deliveryFailed ? ( -
- {translate( - 'components.native-chat.launchPromptNotDelivered', - 'Not delivered — check the terminal' - )} -
- ) : null} -
- ) - } - - // Plain assistant prose is the copyable unit; reasoning/system asides stay - // chrome-free. The controls reveal on hover (and on keyboard focus-within). - const showControls = !isReasoning && !isSystem && markdown.length > 0 - - return ( -
- - {markdown ? ( - - ) : null} - {tools.length > 0 ? ( - - ) : null} - {showControls ? ( - - ) : null} -
- ) -} - export function NativeChatMessageList({ session, isWorking, @@ -200,7 +44,7 @@ export function NativeChatMessageList({ onLinkClick?: CommentMarkdownLinkClickHandler allowFileUriLinks?: boolean failedDeliveryMessageIds?: ReadonlySet - /** Turn timing/disclosure is available only on the structured Codex lane. */ + /** Turn timing and disclosure are available on structured agent sessions. */ showTurnStatus?: boolean runtimeContext?: RuntimeFileOperationArgs | null }): React.JSX.Element { diff --git a/src/renderer/src/components/native-chat/NativeChatMessageRow.tsx b/src/renderer/src/components/native-chat/NativeChatMessageRow.tsx new file mode 100644 index 00000000000..07ea51b5a62 --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatMessageRow.tsx @@ -0,0 +1,172 @@ +import { memo, useCallback, useMemo, useRef } from 'react' +import CommentMarkdown, { + type CommentMarkdownLinkClickHandler +} from '@/components/sidebar/CommentMarkdown' +import { cn } from '@/lib/utils' +import { translate } from '@/i18n/i18n' +import type { NativeChatMessage } from '../../../../shared/native-chat-types' +import { splitNativeChatBlocks } from './native-chat-tool-fold' +import { NativeChatToolRun } from './NativeChatToolRun' +import { nativeChatProseToMarkdown } from './native-chat-prose' +import { + NativeChatAgentControls, + NativeChatImageAttachments, + ProviderFrameRow +} from './NativeChatTranscriptChrome' +import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' + +/** One message: its prose first, then a collapsible run folding all of the + * turn's tool activity. Monochrome per STYLEGUIDE: user prompts read as a + * lifted card, assistant prose as body copy, reasoning de-emphasized. + * Memoized: a stream frame republishes the whole transcript, but settled rows + * keep their block identity, so only the changed row re-renders. */ +export const MessageRow = memo(function MessageRow({ + message, + expandSignal, + activeTurnIsWorking, + onScrollMessageToTop, + onLinkClick, + allowFileUriLinks = false, + deliveryFailed = false, + activityExpandOverride, + structuredActivityUi = true, + runtimeContext +}: { + message: NativeChatMessage + expandSignal: boolean + activeTurnIsWorking?: boolean + /** Align this message's top to the top of the scroll viewport. */ + onScrollMessageToTop: (el: HTMLElement) => void + onLinkClick?: CommentMarkdownLinkClickHandler + allowFileUriLinks?: boolean + deliveryFailed?: boolean + activityExpandOverride?: boolean + structuredActivityUi?: boolean + runtimeContext?: RuntimeFileOperationArgs | null +}): React.JSX.Element | null { + const rowRef = useRef(null) + // One pass per block set: a streaming turn re-renders this row on every frame, and these + // derivations used to re-run each time even though `message.blocks` had not changed. + const { hasImages, markdown, prose, tools } = useMemo(() => { + const split = splitNativeChatBlocks(message.blocks) + return { + ...split, + markdown: nativeChatProseToMarkdown(split.prose), + hasImages: split.prose.some((block) => block.type === 'image-ref') + } + }, [message.blocks]) + const isUser = message.role === 'user' + const isReasoning = message.role === 'reasoning' + const isSystem = message.role === 'system' + const providerFrame = message.blocks.find((block) => block.type === 'text' && block.providerFrame) + + const scrollToTop = useCallback(() => { + if (rowRef.current) { + onScrollMessageToTop(rowRef.current) + } + }, [onScrollMessageToTop]) + + // Skip rows with nothing renderable so the transcript shows no empty/ghost + // bubble. + // After all hooks, so hook order stays unconditional. + if (markdown.length === 0 && !hasImages && tools.length === 0) { + return null + } + + if (providerFrame) { + return ( +
+ +
+ ) + } + + if (isUser) { + return ( +
+ {/* User turns get a distinct muted fill (not the card/canvas color) so + the prompt reads apart from the assistant's body copy. */} +
+ {markdown ? ( + <> + + + + ) : ( + + )} +
+ {deliveryFailed ? ( +
+ {translate( + 'components.native-chat.launchPromptNotDelivered', + 'Not delivered — check the terminal' + )} +
+ ) : null} +
+ ) + } + + // Plain assistant prose is the copyable unit; reasoning/system asides stay + // chrome-free. The controls reveal on hover (and on keyboard focus-within). + const showControls = !isReasoning && !isSystem && markdown.length > 0 + + return ( +
+ + {markdown ? ( + + ) : null} + {tools.length > 0 ? ( + + ) : null} + {showControls ? ( + + ) : null} +
+ ) +}) diff --git a/src/renderer/src/components/native-chat/NativeChatQuestionCard.test.tsx b/src/renderer/src/components/native-chat/NativeChatQuestionCard.test.tsx index b6748fe7e82..9b1a2967682 100644 --- a/src/renderer/src/components/native-chat/NativeChatQuestionCard.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatQuestionCard.test.tsx @@ -28,7 +28,7 @@ afterEach(() => { function render( prompt: AskPrompt, onAnswer: (s: AskAnswerSelection[]) => void, - allowOther = true + allowOther: boolean | readonly boolean[] = true ): void { act(() => { root.render( @@ -155,4 +155,71 @@ describe('NativeChatQuestionCard', () => { expect(container.querySelector('input')).toBeNull() expect(container.textContent).not.toContain('Type your answer') }) + + it('applies free-text capability per question in a grouped prompt', () => { + render( + { + questions: [ + { + header: 'Listed', + question: 'Pick a listed value', + multiSelect: false, + options: [{ label: 'One' }] + }, + { + header: 'Custom', + question: 'Provide a custom value', + multiSelect: false, + options: [] + } + ] + }, + vi.fn(), + [false, true] + ) + + expect(container.querySelector('input')).toBeNull() + clickAction('Skip') + expect(container.querySelector('input')).not.toBeNull() + }) + + it('submits grouped multi-select and free-text answers together', () => { + const onAnswer = vi.fn() + render( + { + questions: [ + { + header: 'Targets', + question: 'Which targets?', + multiSelect: true, + options: [{ label: 'Web' }, { label: 'Mobile' }] + }, + { + header: 'Notes', + question: 'Anything else?', + multiSelect: false, + options: [] + } + ] + }, + onAnswer, + [false, true] + ) + + clickOption('Web') + clickOption('Mobile') + clickAction('Next') + const input = container.querySelector('input')! + act(() => { + const setter = Object.getOwnPropertyDescriptor(HTMLInputElement.prototype, 'value')!.set! + setter.call(input, 'SSH host') + input.dispatchEvent(new Event('input', { bubbles: true })) + }) + clickAction('Submit') + + expect(onAnswer).toHaveBeenCalledWith([ + { indices: [0, 1], other: '' }, + { indices: [], other: 'SSH host' } + ]) + }) }) diff --git a/src/renderer/src/components/native-chat/NativeChatQuestionCard.tsx b/src/renderer/src/components/native-chat/NativeChatQuestionCard.tsx index 4bc881ee3e1..1b1ce5a3547 100644 --- a/src/renderer/src/components/native-chat/NativeChatQuestionCard.tsx +++ b/src/renderer/src/components/native-chat/NativeChatQuestionCard.tsx @@ -10,7 +10,7 @@ export type NativeChatQuestionCardProps = { isSubmitting?: boolean /** Deliver the chosen answer (per-question option indices + free text). */ onAnswer: (selections: AskAnswerSelection[]) => void - allowOther?: boolean + allowOther?: boolean | readonly boolean[] /** Dismiss the prompt (sends Escape to the agent). */ onCancel: () => void /** Exposes the free-text row so pane-level Paste can target it while the @@ -42,6 +42,7 @@ export function NativeChatQuestionCard({ const total = prompt.questions.length const isLast = index === total - 1 const q = prompt.questions[index]! + const questionAllowsOther = Array.isArray(allowOther) ? (allowOther[index] ?? false) : allowOther const setOther = (qi: number, value: string): void => { setOtherText((prev) => { @@ -186,7 +187,7 @@ export function NativeChatQuestionCard({ /> ))}
- {allowOther ? ( + {questionAllowsOther ? ( <> diff --git a/src/renderer/src/components/native-chat/NativeChatResolvedView.tsx b/src/renderer/src/components/native-chat/NativeChatResolvedView.tsx index dc152df4c70..6f934f66a6e 100644 --- a/src/renderer/src/components/native-chat/NativeChatResolvedView.tsx +++ b/src/renderer/src/components/native-chat/NativeChatResolvedView.tsx @@ -52,6 +52,9 @@ import { useNativeChatFileLinkClick } from './use-native-chat-file-link-click' import type { NativeChatResolvedViewProps } from './native-chat-view-types' import { useNativeChatFileLinkContext } from './use-native-chat-file-link-context' import { NativeChatOrchestrationPausedNotice } from './NativeChatOrchestrationPausedNotice' +import { matchNativeChatSplitShortcut } from './native-chat-split-shortcut' +import { getShortcutPlatform } from '@/lib/shortcut-platform' +import { formatShortcutLabel } from '@/hooks/useShortcutLabel' /** Renders the bridge UI after NativeChatSessionGate resolves its agent session. */ export function NativeChatResolvedView({ @@ -73,6 +76,7 @@ export function NativeChatResolvedView({ const runtimeEnvironmentId = useAppStore((s) => selectNativeChatRuntimeEnvironmentId(s, terminalTabId) ) + const keybindings = useAppStore((s) => s.keybindings) const session = useNativeChatRetainedSession({ paneKey, agent, @@ -131,6 +135,10 @@ export function NativeChatResolvedView({ const contextMenu = useNativeChatContextMenu({ rootRef, onSwitchToTerminal, + splitShortcutLabels: { + right: formatShortcutLabel('terminal.splitRight', keybindings), + down: formatShortcutLabel('terminal.splitDown', keybindings) + }, actions: { onPaste: pasteClipboardIntoComposer, ...(contextMenuActions ?? emptyNativeChatContextMenuActions) @@ -337,6 +345,19 @@ export function NativeChatResolvedView({ } }} onKeyDownCapture={(event) => { + const splitDirection = event.repeat + ? null + : matchNativeChatSplitShortcut(event, getShortcutPlatform(), keybindings) + if (splitDirection && contextMenuActions) { + event.preventDefault() + event.stopPropagation() + if (splitDirection === 'right') { + contextMenuActions.onSplitRight() + } else { + contextMenuActions.onSplitDown() + } + return + } // Backspace/Delete outside an input focuses the composer (like typing) // but inserts nothing — let the now-focused field handle the keystroke. if (shouldFocusNativeChatComposerFromEditingKey(event)) { diff --git a/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.test.tsx b/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.test.tsx index d53e8f536d2..031ce4bcd15 100644 --- a/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.test.tsx @@ -165,6 +165,7 @@ function model(overrides: Partial = {}): SessionOptionD ] }, valueSource: 'applied', + transport: 'catalog', settable: true, ...overrides } @@ -183,6 +184,7 @@ const effort: SessionOptionDescriptor = { ] }, valueSource: 'applied', + transport: 'catalog', settable: true } @@ -192,6 +194,7 @@ const fast: SessionOptionDescriptor = { category: 'mode', kind: { type: 'boolean', currentValue: true }, valueSource: 'applied', + transport: 'catalog', settable: true } @@ -342,17 +345,48 @@ describe('NativeChatSessionOptionPickers', () => { expect(screen.queryByRole('button', { name: /^Effort/ })).toBeNull() }) - it('shows the unconfirmed hint for dispatched values', () => { + // The terminal transport typed the value at the agent and has not read it back, + // so the pill says so; the structured transport's own per-turn report is the + // confirmation, which makes the same hedge transient noise there. + it('hedges a dispatched value the terminal transport produced', () => { render( ) - expect(screen.getByText('Sent to the agent — not confirmed')).not.toBeNull() + expect(screen.getByText('Model')).not.toBeNull() + expect(screen.getAllByText('Sent to the agent — not confirmed').length).toBeGreaterThan(0) }) + it('does not hedge a dispatched value the structured transport produced', () => { + render( + + ) + expect(screen.getByText('Model')).not.toBeNull() + expect(screen.queryByText(/not confirmed/)).toBeNull() + }) + + it.each(['catalog', 'agent-session'] as const)( + 'does not hedge a reported value on the %s transport', + (transport) => { + render( + + ) + expect(screen.getByText('Model')).not.toBeNull() + expect(screen.queryByText(/not confirmed/)).toBeNull() + } + ) + it('renders agent-picker routes as one action instead of radio choices', async () => { const invokeAction = vi.fn().mockResolvedValue({ snapshot: [] }) const liveSurface = { ...surface, invokeAction } @@ -449,6 +483,7 @@ describe('NativeChatSessionOptionPickers', () => { category: 'mode', kind: { type: 'boolean' }, valueSource: 'unknown', + transport: 'catalog', settable: true } ]} @@ -463,26 +498,7 @@ describe('NativeChatSessionOptionPickers', () => { await waitFor(() => expect(setOption).toHaveBeenCalledWith('thinking', false)) }) - it('does not show unconfirmed for applied flip-only booleans', () => { - render( - - ) - expect(screen.queryByText('Sent to the agent — not confirmed')).toBeNull() - }) - - it('shows unconfirmed for confirmable dispatched booleans', () => { + it('tooltips a dispatched option pill with the category alone', () => { render( { category: 'mode', kind: { type: 'boolean', currentValue: true }, valueSource: 'dispatched', + transport: 'catalog', settable: true } ]} isWorking={false} /> ) - expect(screen.getByText('Sent to the agent — not confirmed')).not.toBeNull() + expect(screen.getAllByText('Thinking').length).toBeGreaterThan(0) + expect(screen.getAllByText('Sent to the agent — not confirmed').length).toBeGreaterThan(0) }) }) diff --git a/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.tsx b/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.tsx index 87a860662d2..31ff2cbdc4e 100644 --- a/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.tsx +++ b/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.tsx @@ -15,10 +15,11 @@ import { import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' import { translate } from '@/i18n/i18n' import { sortNativeChatSessionOptions } from '../../../../shared/native-chat-session-option-snapshot' -import type { - SessionOptionDescriptor, - SessionOptionsSurface, - SessionOptionValue +import { + sessionOptionDispatchUnconfirmed, + type SessionOptionDescriptor, + type SessionOptionsSurface, + type SessionOptionValue } from '../../../../shared/native-chat-session-options' import { nativeChatModelPillLabel, @@ -250,7 +251,7 @@ function NativeChatSessionOptionPickersInner({ tooltipLabel={optionsTooltip} disabled={isWorking || pendingId !== null} disabledReason={optionsReason} - dispatched={options.some((descriptor) => descriptor.valueSource === 'dispatched')} + dispatched={options.some(sessionOptionDispatchUnconfirmed)} /> {options.map((descriptor, index) => { @@ -283,7 +284,7 @@ function NativeChatSessionOptionPickersInner({ tooltipLabel={modelTooltip} disabled={isWorking || pendingId !== null} disabledReason={modelReason} - dispatched={model.valueSource === 'dispatched'} + dispatched={sessionOptionDispatchUnconfirmed(model)} /> {modelReason && !model.settable ? ( diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx index bfb3dd1ef52..afdf5ace7c5 100644 --- a/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx @@ -3,6 +3,10 @@ import { act, cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react' import React, { forwardRef, useImperativeHandle } from 'react' import { afterEach, describe, expect, it, vi } from 'vitest' +import type { AgentJournalRenderItem } from '../../../../shared/agent-session-journal-types' +import type { AgentSessionBackgroundTask } from '../../../../shared/agent-session-wire' +import { decodeAgentSessionQuestionAnswers } from '../../../../shared/agent-session-question-answer' +import type { NativeChatQuestionCardProps } from './NativeChatQuestionCard' const mocks = vi.hoisted(() => ({ call: vi.fn(), @@ -11,11 +15,22 @@ const mocks = vi.hoisted(() => ({ messageListProps: null as null | { allowFileUriLinks?: boolean onLinkClick?: (...args: unknown[]) => void + showTurnStatus?: boolean + runtimeContext?: unknown }, - composerProps: null as null | { structuredTransport?: Record }, + composerProps: null as null | { + structuredTransport?: Record + isWorking?: boolean + }, + questionCardProps: null as NativeChatQuestionCardProps | null, + promptItems: [] as AgentJournalRenderItem[], + respond: vi.fn(), handlePasteEvent: vi.fn(), pasteFromClipboard: vi.fn(), - submissions: [] as unknown[] + submissions: [] as unknown[], + monitoringBackgroundTasks: false, + backgroundTasks: [] as AgentSessionBackgroundTask[], + stopBackgroundTasks: vi.fn() })) vi.mock('@/runtime/structured-agent-session-client', () => ({ @@ -53,15 +68,18 @@ vi.mock('./use-structured-agent-session', async () => { hasOlder: false, loadingOlder: false, loadOlder: vi.fn(), - prompts: [], + prompts: mocks.promptItems, outbox: outbox.outbox, blockedClientMessageId: outbox.blockedClientMessageId, send: outbox.send, retry: outbox.retry, isWorking: false, + isMonitoringBackgroundTasks: mocks.monitoringBackgroundTasks, + backgroundTasks: mocks.backgroundTasks, turnId: null, cancel: vi.fn(), - respond: vi.fn(), + stopBackgroundTasks: mocks.stopBackgroundTasks, + respond: mocks.respond, optionSnapshot: [ { id: 'model', @@ -125,7 +143,12 @@ vi.mock('./NativeChatComposer', () => ({ })) vi.mock('./NativeChatEmptyState', () => ({ NativeChatEmptyState: () => null })) vi.mock('./NativeChatApprovalCard', () => ({ NativeChatApprovalCard: () => null })) -vi.mock('./NativeChatQuestionCard', () => ({ NativeChatQuestionCard: () => null })) +vi.mock('./NativeChatQuestionCard', () => ({ + NativeChatQuestionCard: (props: NativeChatQuestionCardProps) => { + mocks.questionCardProps = props + return null + } +})) import { NativeChatStructuredSession } from './NativeChatStructuredSession' @@ -136,9 +159,15 @@ describe('NativeChatStructuredSession', () => { mocks.mode = 'static' mocks.messageListProps = null mocks.composerProps = null + mocks.questionCardProps = null + mocks.promptItems = [] + mocks.respond.mockReset() mocks.handlePasteEvent.mockReset() mocks.pasteFromClipboard.mockReset() mocks.submissions = [] + mocks.monitoringBackgroundTasks = false + mocks.stopBackgroundTasks.mockReset() + mocks.backgroundTasks = [] }) it('routes app-menu paste into the structured composer', () => { @@ -149,7 +178,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-paste" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -160,15 +188,14 @@ describe('NativeChatStructuredSession', () => { expect(mocks.pasteFromClipboard).toHaveBeenCalledOnce() }) - it('wires local structured file links through the native chat opener', () => { + it('wires remote structured file links through the host-aware native chat opener', () => { render( ) @@ -176,6 +203,67 @@ describe('NativeChatStructuredSession', () => { expect(mocks.messageListProps?.onLinkClick).toBe(mocks.fileLinkClick) }) + // Turn status and transcript image previews shipped Codex-first. Every + // structured session renders through the same list, so neither is agent-gated. + it.each(['codex', 'claude'] as const)( + 'renders the same structured transcript chrome for %s', + (agent) => { + render( + + ) + + expect(mocks.messageListProps?.showTurnStatus).toBe(true) + expect(mocks.messageListProps?.runtimeContext).not.toBeUndefined() + } + ) + + it('places background monitoring above the usable composer and stops without an active turn', async () => { + mocks.monitoringBackgroundTasks = true + mocks.backgroundTasks = [ + { id: 'task-command', kind: 'command', description: 'sleep 180' }, + { id: 'task-agent', kind: 'agent' } + ] + mocks.stopBackgroundTasks.mockResolvedValue({ cancelled: true }) + + render( + + ) + + const status = screen + .getByText('Monitoring background tasks') + .closest('[data-native-chat-background-tasks="true"]') + const composer = screen.getByTestId('structured-composer') + if (!status) { + throw new Error('background task status was not rendered') + } + expect(status.compareDocumentPosition(composer) & Node.DOCUMENT_POSITION_FOLLOWING).toBeTruthy() + expect(mocks.composerProps?.isWorking).toBe(false) + expect(screen.queryByRole('list', { name: 'Running background tasks' })).toBeNull() + + const disclosure = screen.getByRole('button', { name: 'Monitoring background tasks' }) + expect(disclosure.getAttribute('aria-expanded')).toBe('false') + fireEvent.click(disclosure) + expect(disclosure.getAttribute('aria-expanded')).toBe('true') + expect(screen.getByRole('list', { name: 'Running background tasks' })).toBeTruthy() + expect(screen.getByText('sleep 180')).toBeTruthy() + expect(screen.getByText('Background agent')).toBeTruthy() + + fireEvent.click(screen.getByRole('button', { name: 'Stop' })) + await waitFor(() => expect(mocks.stopBackgroundTasks).toHaveBeenCalledOnce()) + }) + it('routes a bare model command to the native option picker', async () => { render( { sessionId="session-1" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) const dispatchCommand = mocks.composerProps?.structuredTransport?.dispatchCommand as @@ -220,7 +307,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-1" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -252,7 +338,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-wedge" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -284,7 +369,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-probe-flag" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -317,7 +401,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-parked" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -367,7 +450,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-churn" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) const { rerender } = render(makeView()) @@ -421,7 +503,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-target-switch" target={target} agent="codex" - allowFileUriLinks /> ) const { rerender } = render(makeView({ kind: 'local' })) @@ -456,7 +537,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-forced" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -495,7 +575,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-pending" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -525,7 +604,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-budget" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -546,4 +624,129 @@ describe('NativeChatStructuredSession', () => { vi.useRealTimers() } }, 30000) + + it('passes Claude grouped questions and one shared answer through the card', () => { + mocks.promptItems = [ + { + itemId: 'question-item', + revision: 1, + sequence: 1, + observedAt: 1, + body: { + kind: 'question', + question: '2 grouped questions from Claude', + options: [], + questions: [ + { + id: 'q1', + header: 'Targets', + question: 'Which targets?', + multiSelect: true, + options: [ + { id: 'target-web', label: 'Web' }, + { id: 'target-mobile', label: 'Mobile' } + ], + freeTextQuestionId: 'q1' + }, + { + id: 'q2', + header: 'Host', + question: 'Where should it run?', + multiSelect: false, + options: [], + freeTextQuestionId: 'q2' + } + ], + resolution: { + state: 'pending', + selectedOptionId: null, + resolvedBy: null, + resolvedAt: null + } + } + } + ] + + render( + + ) + + const card = mocks.questionCardProps + if (!card) { + throw new Error('question card was not rendered') + } + expect(card.prompt.questions).toHaveLength(2) + expect(card.prompt.questions[0]).toMatchObject({ + question: 'Which targets?', + multiSelect: true, + options: [{ label: 'Web' }, { label: 'Mobile' }] + }) + expect(card.allowOther).toEqual([true, true]) + + card.onAnswer([ + { indices: [0, 1], other: '' }, + { indices: [], other: 'SSH host' } + ]) + const encoded = mocks.respond.mock.calls[0]?.[1] + expect(decodeAgentSessionQuestionAnswers(encoded)).toEqual([ + { questionId: 'q1', optionIds: ['target-web', 'target-mobile'] }, + { questionId: 'q2', optionIds: [], other: 'SSH host' } + ]) + }) + + it('keeps legacy single-question option ids and free text behavior', () => { + mocks.promptItems = [ + { + itemId: 'legacy-question-item', + revision: 1, + sequence: 1, + observedAt: 1, + body: { + kind: 'question', + question: 'Pick a library', + options: [ + { id: 'q1:choice-1', label: 'React' }, + { id: 'q1:choice-2', label: 'Vue' } + ], + freeTextQuestionId: 'q1', + resolution: { + state: 'pending', + selectedOptionId: null, + resolvedBy: null, + resolvedAt: null + } + } + } + ] + + render( + + ) + + const card = mocks.questionCardProps + if (!card) { + throw new Error('question card was not rendered') + } + expect(card.prompt.questions).toEqual([ + { + question: 'Pick a library', + multiSelect: false, + options: [{ label: 'React' }, { label: 'Vue' }] + } + ]) + card.onAnswer([{ indices: [1], other: '' }]) + expect(mocks.respond).toHaveBeenCalledWith(mocks.promptItems[0], 'q1:choice-2') + }) }) diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx index f7d2fd62663..9ac354f8a73 100644 --- a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx +++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx @@ -1,13 +1,9 @@ import { useMemo, useRef, useState } from 'react' import { RotateCcw } from 'lucide-react' -import type { - AgentStatusOrchestrationContext, - AgentType -} from '../../../../shared/agent-status-types' +import { encodeAgentSessionQuestionAnswers } from '../../../../shared/agent-session-question-answer' import { dispatchStructuredAgentSessionComposerCommand } from '../../../../shared/structured-agent-session-composer' import { structuredAgentSessionPaneKey } from '../../../../shared/structured-agent-session-projection' import type { NativeChatLiveSession } from './use-native-chat-live-session' -import type { RuntimeClientTarget } from '@/runtime/runtime-rpc-client' import { Button } from '@/components/ui/button' import { NativeChatApprovalCard } from './NativeChatApprovalCard' import { NativeChatComposer, type NativeChatComposerHandle } from './NativeChatComposer' @@ -21,24 +17,21 @@ import { useNativeChatFileLinkContext } from './use-native-chat-file-link-contex import { useStructuredAgentSession } from './use-structured-agent-session' import { translate } from '@/i18n/i18n' import { NativeChatOrchestrationPausedNotice } from './NativeChatOrchestrationPausedNotice' -import { useNativeChatPasteBridge } from './use-native-chat-paste-bridge' import { useNativeChatImageRuntimeContext } from './native-chat-image-runtime-context' +import { useStructuredNativeChatPaneCommands } from './use-structured-native-chat-pane-commands' +import type { NativeChatStructuredViewProps } from './native-chat-view-types' +import { NativeChatBackgroundTasksStatus } from './NativeChatBackgroundTasksStatus' function encodeQuestionAnswer(questionId: string, answer: string): string { return `${encodeURIComponent(questionId)}:${encodeURIComponent(answer)}` } -export function NativeChatStructuredSession(props: { - tabId: string - sessionId: string - target: RuntimeClientTarget - agent: AgentType - isVisible: boolean - allowFileUriLinks: boolean - orchestrationDispatchStatus?: AgentStatusOrchestrationContext['dispatchStatus'] -}): React.JSX.Element { +export function NativeChatStructuredSession( + props: Omit +): React.JSX.Element { const controller = useStructuredAgentSession(props) const [composerError, setComposerError] = useState(null) + const [stoppingBackgroundTasks, setStoppingBackgroundTasks] = useState(false) const [optionPickerRequest, setOptionPickerRequest] = useState<{ id: string sequence: number @@ -49,7 +42,14 @@ export function NativeChatStructuredSession(props: { ) const rootRef = useRef(null) const composerRef = useRef(null) - useNativeChatPasteBridge({ rootRef, composerRef }) + const paneCommands = useStructuredNativeChatPaneCommands({ + tabId: props.tabId, + groupId: props.groupId, + isVisible: props.isVisible, + rootRef, + composerRef, + terminalPaneActions: props.contextMenuActions + }) const session = useMemo( () => ({ messages: controller.messages, @@ -82,9 +82,24 @@ export function NativeChatStructuredSession(props: { const fontScale = useNativeChatFontScale(viewState.kind === 'ready') const fileLinkContext = useNativeChatFileLinkContext(props.tabId) const imageRuntimeContext = useNativeChatImageRuntimeContext(props.tabId) - const fileLinkClick = useNativeChatFileLinkClick(props.allowFileUriLinks ? fileLinkContext : null) + const fileLinkClick = useNativeChatFileLinkClick(fileLinkContext) const prompt = controller.prompts[0] ?? null const questionBody = prompt?.body.kind === 'question' ? prompt.body : null + const questions = + questionBody?.questions ?? + (questionBody + ? [ + { + id: questionBody.freeTextQuestionId ?? 'q1', + question: questionBody.question, + options: questionBody.options, + multiSelect: false, + ...(questionBody.freeTextQuestionId + ? { freeTextQuestionId: questionBody.freeTextQuestionId } + : {}) + } + ] + : []) const retryableOutboxEntry = controller.outbox.find((entry) => entry.state === 'unconfirmed') ?? controller.outbox.find( @@ -127,6 +142,15 @@ export function NativeChatStructuredSession(props: { data-native-chat-root="true" data-native-chat-working={controller.isWorking ? 'true' : 'false'} tabIndex={-1} + onPointerDownCapture={(event) => { + if (event.button === 2) { + paneCommands.onSelectionCapture() + } + }} + onMouseUpCapture={paneCommands.onSelectionCapture} + onKeyUpCapture={paneCommands.onSelectionCapture} + onKeyDownCapture={paneCommands.onKeyDownCapture} + onContextMenuCapture={paneCommands.onContextMenuCapture} className="flex h-full min-h-0 w-full flex-col bg-background focus:outline-none" > @@ -144,10 +168,10 @@ export function NativeChatStructuredSession(props: { expandSignal={false} fontScale={fontScale.scale} workingStartedAt={null} - showTurnStatus={props.agent === 'codex'} + showTurnStatus onLinkClick={fileLinkClick} allowFileUriLinks={fileLinkClick !== undefined} - runtimeContext={props.agent === 'codex' ? imageRuntimeContext : undefined} + runtimeContext={imageRuntimeContext} /> )}
@@ -166,17 +190,39 @@ export function NativeChatStructuredSession(props: { ) : null} {prompt && questionBody ? ( ({ label: option.label })) - } - ] + questions: questions.map((question) => ({ + question: question.question, + ...(question.header ? { header: question.header } : {}), + multiSelect: question.multiSelect, + options: question.options.map((option) => ({ + label: option.label, + ...(option.description ? { description: option.description } : {}) + })) + })) }} - allowOther={Boolean(questionBody.freeTextQuestionId)} + allowOther={questions.map((question) => Boolean(question.freeTextQuestionId))} onAnswer={(answers) => { + if (questionBody.questions) { + const grouped = questions.map((question, questionIndex) => { + const answer = answers[questionIndex] + const other = answer?.other?.trim() + const optionIds = (answer?.indices ?? []).flatMap((optionIndex) => { + const optionId = question.options[optionIndex]?.id + return optionId ? [optionId] : [] + }) + return { + questionId: question.id, + optionIds: question.multiSelect || !other ? optionIds : [], + ...(other ? { other } : {}) + } + }) + if (grouped.every((answer) => answer.optionIds.length > 0 || answer.other)) { + void controller.respond(prompt, encodeAgentSessionQuestionAnswers(grouped)) + } + return + } const index = answers[0]?.indices[0] const other = answers[0]?.other?.trim() const optionId = @@ -228,6 +274,16 @@ export function NativeChatStructuredSession(props: { {controller.error ?? composerError}

) : null} + {controller.isMonitoringBackgroundTasks ? ( + { + setStoppingBackgroundTasks(true) + void controller.stopBackgroundTasks().finally(() => setStoppingBackgroundTasks(false)) + }} + /> + ) : null} {prompt ? null : ( )} + {paneCommands.menu} ) } diff --git a/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx b/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx index 41a70a8457d..050b3f7c84b 100644 --- a/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx @@ -2,8 +2,8 @@ import '@testing-library/jest-dom/vitest' -import { cleanup, render, screen } from '@testing-library/react' -import { afterEach, describe, expect, it } from 'vitest' +import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' import type { AgentJournalRenderItem } from '../../../../shared/agent-session-journal-types' import type { NativeChatBlock } from '../../../../shared/native-chat-types' import { projectStructuredItemToNativeChat } from '../../../../shared/structured-agent-session-projection' @@ -32,6 +32,9 @@ describe('NativeChatToolRun', () => { { type: 'tool-call', name: 'apply_patch', + // The patch lives on the call in this lane, so the provider's own + // completion is what says the edit landed. + state: 'completed', input: { changes: [ { @@ -46,8 +49,9 @@ describe('NativeChatToolRun', () => { const { container } = render() - expect(screen.getByText('+after')).toBeInTheDocument() - expect(screen.getByText('-before')).toBeInTheDocument() + expect(screen.getByText('after')).toBeInTheDocument() + expect(screen.getByText('before')).toBeInTheDocument() + expect(screen.getByText('Edited file')).toBeInTheDocument() expect(container.querySelector('pre')).toBeNull() }) @@ -78,18 +82,156 @@ describe('NativeChatToolRun', () => { ) - expect(screen.getByText('+after')).toHaveClass( - 'bg-emerald-500/10', - 'text-[var(--git-decoration-added)]' - ) - expect(screen.getByText('-before')).toHaveClass( - 'bg-rose-500/10', - 'text-[var(--git-decoration-deleted)]' - ) + // Row grounds come from the diff tokens, not a hardcoded palette value. + expect(screen.getByText('after').closest('div')).toHaveClass('bg-[var(--diff-added-ground)]') + expect(screen.getByText('before').closest('div')).toHaveClass('bg-[var(--diff-removed-ground)]') expect(container).not.toHaveTextContent('"changes"') expect(container.querySelector('pre')).toBeNull() }) + it('keeps the provider error visible for an edit the agent could not apply', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Edit', + input: { file_path: '/repo/a.ts', old_string: 'missing', new_string: 'now' } + }, + { type: 'tool-result', output: 'String to replace not found in file.', isError: true } + ] + + const { container } = render() + + expect(screen.queryByText('Edited file')).toBeNull() + const body = container.querySelector('pre') + expect(body).toHaveTextContent('String to replace not found in file.') + expect(body).toHaveClass('text-destructive') + }) + + it('leaves a `git diff` command as a command row rather than an edit card', () => { + const blocks: NativeChatBlock[] = [ + { type: 'tool-call', name: 'exec', input: { command: 'git diff' }, state: 'completed' }, + { + type: 'tool-result', + output: 'diff --git a/a.ts b/a.ts\n--- a/a.ts\n+++ b/a.ts\n@@ -1 +1 @@\n-was\n+now' + } + ] + + const { container } = render() + + expect(screen.queryByText('Edited file')).toBeNull() + expect(container).toHaveTextContent('git diff') + }) + + it('shows no gutter number for a snippet edit, which cannot locate itself', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Edit', + input: { file_path: '/repo/a.ts', old_string: 'was', new_string: 'now' }, + state: 'completed' + }, + { type: 'tool-result', output: 'ok' } + ] + + render() + + // Exact, because a snippet-relative number would sit ahead of the marker. + expect(screen.getByText('now').closest('div')?.textContent).toBe('+now') + expect(screen.getByText('was').closest('div')?.textContent).toBe('-was') + }) + + it('separates two regions of a file so the gutter jump is accounted for', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Edit', + input: { file_path: '/repo/a.ts' }, + state: 'completed' + }, + { + type: 'tool-result', + output: 'ok', + editPatch: { + filePath: '/repo/a.ts', + hunks: [ + { oldStart: 42, oldLines: 1, newStart: 42, newLines: 1, lines: ['-was', '+now'] }, + { oldStart: 310, oldLines: 1, newStart: 310, newLines: 1, lines: ['-old', '+new'] } + ] + } + } + ] + + render() + + const separators = screen.getAllByRole('separator') + expect(separators).toHaveLength(1) + expect(separators[0]).toHaveAccessibleName('Lines not shown') + }) + + it('offers no empty body for a delete, which names the file and nothing else', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'apply_patch', + input: { input: '*** Begin Patch\n*** Delete File: gone.ts\n*** End Patch' }, + state: 'completed' + } + ] + + render() + + expect(screen.getByTitle('gone.ts')).toBeInTheDocument() + // The header states the change; there is no body behind a disclosure. + expect(screen.getByText('Deleted file').closest('button')).not.toHaveAttribute('aria-expanded') + }) + + it('says a diff was clipped even while the card is collapsed', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Diff', + input: { path: 'src/a.ts' }, + state: 'completed' + }, + { type: 'tool-result', output: '@@ -1,3 +1,3 @@\n ctx\n-was\n+now\n… (48210 bytes)' } + ] + + // A defined expandOverride opens the run while leaving each card closed. + render() + + expect(screen.getByText('Diff truncated')).toBeInTheDocument() + expect(screen.queryByText('was')).toBeNull() + }) + + it('copies the diff as signed rows, with the region breaks left out', () => { + const writeClipboardText = vi.fn() + Object.assign(window, { api: { ui: { writeClipboardText } } }) + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Edit', + input: { file_path: '/repo/a.ts' }, + state: 'completed' + }, + { + type: 'tool-result', + output: 'ok', + editPatch: { + filePath: '/repo/a.ts', + hunks: [ + { oldStart: 1, oldLines: 2, newStart: 1, newLines: 2, lines: [' ctx', '-was', '+now'] }, + { oldStart: 90, oldLines: 1, newStart: 90, newLines: 1, lines: ['+tail'] } + ] + } + } + ] + + render() + fireEvent.click(screen.getByRole('button', { name: 'Copy diff' })) + + expect(writeClipboardText).toHaveBeenCalledWith(' ctx\n-was\n+now\n+tail') + }) + it('keeps a grouped active run to one stable row showing only the latest tool', () => { const blocks: NativeChatBlock[] = [ { type: 'tool-call', name: 'shell', input: { command: 'date' }, state: 'completed' }, @@ -99,7 +241,9 @@ describe('NativeChatToolRun', () => { const { container } = render() - expect(screen.getByText('Running cat package.json')).toBeInTheDocument() + const activeLabel = screen.getByText('Running cat package.json') + expect(activeLabel).toBeInTheDocument() + expect(activeLabel).toHaveClass('animate-pulse', 'motion-reduce:animate-none') expect(screen.queryByText('Running date')).toBeNull() expect(screen.queryByText('Running pwd')).toBeNull() expect(screen.queryByText('Ran 3 commands and used 1 tool')).toBeNull() diff --git a/src/renderer/src/components/native-chat/NativeChatToolRun.tsx b/src/renderer/src/components/native-chat/NativeChatToolRun.tsx index 716d293838e..e91177b375a 100644 --- a/src/renderer/src/components/native-chat/NativeChatToolRun.tsx +++ b/src/renderer/src/components/native-chat/NativeChatToolRun.tsx @@ -1,4 +1,4 @@ -import { useEffect, useState } from 'react' +import { useEffect, useMemo, useState } from 'react' import { Check, ChevronRight, SquareTerminal, Wrench } from 'lucide-react' import { cn } from '@/lib/utils' import { translate } from '@/i18n/i18n' @@ -8,53 +8,37 @@ import { type NativeChatBlock } from '../../../../shared/native-chat-types' import { diffFromText, diffFromToolCall, type DiffLine } from './native-chat-diff' +import { NativeChatDiffCard } from './NativeChatDiffCard' +import { pairToolBlocks } from './native-chat-tool-fold' +import { + editFilesFromToolPair, + isEditToolName +} from '../../../../shared/native-chat-edit-normalize' +import type { NativeChatEditFile } from '../../../../shared/native-chat-edit-model' import { countToolCalls, createToolInputDisplay, summarizeToolRun, truncateToolDetail } from './native-chat-tool-summary' +import { + describeActiveToolCall, + isCommandToolName, + NATIVE_CHAT_TOOL_ACTIVITY_COPY, + selectActiveToolCall +} from '../../../../shared/native-chat-tool-activity' import { NativeChatDiffView } from './NativeChatDiffView' -const COMMAND_TOOL_NAMES = new Set([ - 'bash', - 'shell', - 'powershell', - 'terminal', - 'execute', - 'run_command', - 'run_shell_command', - 'shell_command', - 'exec_command', - 'run_terminal_cmd', - 'run_terminal_command' -]) - -function normalizedToolName(name: string): string { - return name.trim().toLowerCase() -} - function activeToolLabel(call: Extract): string { - const preview = createToolInputDisplay(call.input).label - if (COMMAND_TOOL_NAMES.has(normalizedToolName(call.name))) { - return preview - ? translate('components.native-chat.tool.runningPreview', 'Running {{preview}}', { - preview - }) - : translate('components.native-chat.tool.runningCommand', 'Running command') - } - return preview - ? translate( - 'components.native-chat.tool.runningNamedPreview', - 'Running {{toolName}} {{preview}}', - { - toolName: call.name, - preview - } - ) - : translate('components.native-chat.tool.runningNamed', 'Running {{toolName}}', { - toolName: call.name - }) + const { key, toolName, preview } = describeActiveToolCall(call) + const copy = NATIVE_CHAT_TOOL_ACTIVITY_COPY[key] + return key === 'runningPreview' + ? translate('components.native-chat.tool.runningPreview', copy, { preview }) + : key === 'runningCommand' + ? translate('components.native-chat.tool.runningCommand', copy) + : key === 'runningNamedPreview' + ? translate('components.native-chat.tool.runningNamedPreview', copy, { toolName, preview }) + : translate('components.native-chat.tool.runningNamed', copy, { toolName }) } /** A single inline tool line — `▸ ToolName preview` — that expands in place to @@ -151,6 +135,50 @@ function ToolLine({ ) } +type EditCardModel = { + editCards: Map + /** Result blocks the card already speaks for, so they render no second row. */ + consumedResults: Set +} + +const NO_EDIT_CARDS: EditCardModel = { editCards: new Map(), consumedResults: new Set() } + +/** An edit renders as one card, so its result block is folded into the call. The + * model decides which calls have landed; a call that has not keeps the generic + * tool view, its result still visible as the provider's own error. */ +function buildEditCards(blocks: NativeChatBlock[]): EditCardModel { + const editCards: EditCardModel['editCards'] = new Map() + const consumedResults: EditCardModel['consumedResults'] = new Set() + for (const [index, pair] of pairToolBlocks(blocks).entries()) { + const call = pair.call + if (!call || !isEditToolName(call.name)) { + continue + } + const files = editFilesFromToolPair({ + name: call.name, + input: call.input, + ...(call.state ? { state: call.state } : {}), + ...(pair.result + ? { + result: { + output: pair.result.output, + isError: pair.result.isError, + editPatch: pair.result.editPatch + } + } + : {}) + }) + if (!files || files.length === 0) { + continue + } + editCards.set(call, { files, key: `${call.name}:${index}` }) + if (pair.result) { + consumedResults.add(pair.result) + } + } + return { editCards, consumedResults } +} + /** A run of a message's tool calls/results, collapsed to a one-line summary that * expands to the individual inline tool lines. `expandSignal` lets the global * toolbar toggle drive every run at once while still allowing per-run override. */ @@ -176,27 +204,25 @@ export function NativeChatToolRun({ const callCount = countToolCalls(blocks) || blocks.length const summary = summarizeToolRun(blocks) - const calls = blocks.filter(isToolCallBlock) - const activeCalls = structuredActivityUi - ? calls.filter( - (call) => - (call.state === 'running' || (call.state == null && activeTurnIsWorking === true)) && - activeTurnIsWorking !== false - ) - : [] - const latestActiveCall = activeCalls.at(-1) + const latestActiveCall = structuredActivityUi + ? selectActiveToolCall(blocks, { activeTurnIsWorking }) + : null const isSettled = latestActiveCall == null // The turn caret opens the activity group, while each child tool remains // collapsed. The global expand toolbar still opens child details together. const expandToolLines = expandOverride === undefined ? open : false + // Diffing every edit is the run's most expensive work, so a collapsed run — + // which renders none of it — never pays for it. + const { editCards, consumedResults } = useMemo( + () => (open ? buildEditCards(blocks) : NO_EDIT_CARDS), + [open, blocks] + ) const ActiveToolIcon = - latestActiveCall && COMMAND_TOOL_NAMES.has(normalizedToolName(latestActiveCall.name)) - ? SquareTerminal - : Wrench + latestActiveCall && isCommandToolName(latestActiveCall.name) ? SquareTerminal : Wrench const fallbackLabel = callCount === 1 - ? translate('components.native-chat.tool.countOne', '1 tool call') - : translate('components.native-chat.tool.countN', '{{value0}} tool calls', { + ? translate('components.native-chat.tool.countOne', NATIVE_CHAT_TOOL_ACTIVITY_COPY.countOne) + : translate('components.native-chat.tool.countN', NATIVE_CHAT_TOOL_ACTIVITY_COPY.countN, { value0: callCount }) @@ -227,7 +253,7 @@ export function NativeChatToolRun({ - + {activeToolLabel(latestActiveCall)} {open ? : null} @@ -264,6 +290,23 @@ export function NativeChatToolRun({ {(() => { const seen = new Map() return blocks.map((block) => { + const edit = editCards.get(block) + if (edit) { + return ( +
+ {edit.files.map((file, fileIndex) => ( + + ))} +
+ ) + } + if (consumedResults.has(block)) { + return null + } const signature = block.type === 'tool-call' ? `${block.type}:${block.name}:${JSON.stringify(block.input)}` diff --git a/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx b/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx index a883429f658..e6c38de83f5 100644 --- a/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx +++ b/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx @@ -2,6 +2,14 @@ import { useState } from 'react' import { ChevronRight } from 'lucide-react' import { translate } from '@/i18n/i18n' import { useNow } from '@/hooks/use-now' +import { + describeNativeChatTurnStatus, + formatNativeChatDuration, + NATIVE_CHAT_TURN_STATUS_COPY, + nativeChatElapsedSeconds +} from '../../../../shared/native-chat-turn-status' + +export { formatNativeChatDuration } export function NativeChatWorkingStatus({ startedAt, @@ -24,21 +32,29 @@ export function NativeChatWorkingStatus({ // Why: preserves the old effect's `startedAt ?? Date.now()` epoch for the // single frame before the turn's startedAt lands. const [mountedAt] = useState(() => Date.now()) - const elapsedSeconds = counting - ? Math.max(0, Math.floor((now - (startedAt ?? mountedAt)) / 1000)) - : 0 + const elapsedSeconds = counting ? nativeChatElapsedSeconds(startedAt, mountedAt, now) : 0 + const { key, duration } = describeNativeChatTurnStatus({ + thinking, + workedSeconds, + elapsedSeconds + }) const label = - workedSeconds != null - ? translate('components.native-chat.status.workedFor', 'Worked for {{value0}} seconds', { - value0: workedSeconds - }) - : thinking - ? translate('components.native-chat.status.thinking', 'Thinking') - : translate('components.native-chat.status.workingFor', 'Working for {{value0}} seconds', { - value0: elapsedSeconds - }) - + key === 'workedFor' + ? translate( + 'components.native-chat.status.workedFor', + NATIVE_CHAT_TURN_STATUS_COPY.workedFor, + { + value0: duration + } + ) + : key === 'thinking' + ? translate('components.native-chat.status.thinking', NATIVE_CHAT_TURN_STATUS_COPY.thinking) + : translate( + 'components.native-chat.status.workingFor', + NATIVE_CHAT_TURN_STATUS_COPY.workingFor, + { value0: duration } + ) const className = `flex min-h-8 items-center gap-1 text-sm text-muted-foreground${thinking ? '' : ' border-b border-border'}` const caret = workedSeconds != null ? ( @@ -52,7 +68,10 @@ export function NativeChatWorkingStatus({ + + ) : owner === 'native' && phase === 'idle' ? ( + isWorking ? ( + <> + + + + ) : ( + + ) + ) : owner === 'tui' && phase === 'idle' ? ( + + ) : null} + + + {owner === 'tui' && phase === 'idle' ? ( +
+ + {status?.hostLabel + ? translate( + 'components.native-chat.handoff.agentOpenOnHost', + 'Agent is open in terminal on {{value0}}.', + { value0: status.hostLabel } + ) + : translate('components.native-chat.handoff.agentOpen', 'Agent is open in terminal.')} + + +
+ ) : null} + {switching ? ( +
+ {phase === 'waiting-for-exit' + ? translate( + 'components.native-chat.handoff.exitTerminal', + 'Exit the agent terminal to continue in chat.' + ) + : status?.stage + ? handoffStageCopy(status) + : translate( + 'components.native-chat.handoff.switchingOwner', + 'Switching session owner…' + )} +
+ ) : null} + {phase === 'failed' && status?.error ? ( +
+
+ {status.error.message} + {status.direction && status.error.canRetryProof ? ( + + ) : status.direction && status.error.recoverableOwner !== 'none' ? ( + + ) : null} +
+ {status.error.details ? ( +
+ {translate('components.native-chat.handoff.details', 'Details')} +

{status.error.details}

+
+ ) : null} +
+ ) : null} + + ) +} diff --git a/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.test.tsx b/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.test.tsx index 25991eac3ff..05101f14fd1 100644 --- a/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.test.tsx +++ b/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.test.tsx @@ -8,7 +8,6 @@ type MockAppState = { unifiedTabsByWorktree: Record groupsByWorktree: Record runtimeEnvironmentId: string | null - executionHostId: string focusGroup: (worktreeId: string, groupId: string) => void } @@ -16,7 +15,8 @@ const mocks = vi.hoisted(() => ({ store: null as null | { setState: (state: Partial) => void }, focusGroup: vi.fn(), mountsByTabId: new Map(), - unmountsByTabId: new Map() + unmountsByTabId: new Map(), + groupIdByTabId: new Map() })) vi.mock('@/store', async () => { @@ -25,7 +25,6 @@ vi.mock('@/store', async () => { unifiedTabsByWorktree: {}, groupsByWorktree: {}, runtimeEnvironmentId: null, - executionHostId: 'local', focusGroup: mocks.focusGroup })) mocks.store = useAppStore @@ -33,8 +32,7 @@ vi.mock('@/store', async () => { }) vi.mock('@/lib/worktree-runtime-owner', () => ({ - getRuntimeEnvironmentIdForWorktree: (state: MockAppState) => state.runtimeEnvironmentId, - getExecutionHostIdForWorktree: (state: MockAppState) => state.executionHostId + getRuntimeEnvironmentIdForWorktree: (state: MockAppState) => state.runtimeEnvironmentId })) vi.mock('@/runtime/runtime-rpc-client', () => ({ @@ -53,11 +51,14 @@ vi.mock('./NativeChatView', async () => { return { default: function MockNativeChatView({ tabId, + groupId, isVisible }: { tabId: string + groupId?: string isVisible: boolean }) { + mocks.groupIdByTabId.set(tabId, groupId) useEffect(() => { mocks.mountsByTabId.set(tabId, (mocks.mountsByTabId.get(tabId) ?? 0) + 1) return () => { @@ -87,6 +88,7 @@ describe('StructuredAgentSessionPaneOverlayLayer', () => { mocks.focusGroup.mockClear() mocks.mountsByTabId.clear() mocks.unmountsByTabId.clear() + mocks.groupIdByTabId.clear() mocks.store?.setState(createState(FIRST_TAB_ID)) }) @@ -125,6 +127,12 @@ describe('StructuredAgentSessionPaneOverlayLayer', () => { expect(mocks.mountsByTabId.get(FIRST_TAB_ID)).toBe(1) expect(mocks.mountsByTabId.get(SECOND_TAB_ID)).toBe(1) expect(mocks.unmountsByTabId.size).toBe(0) + expect(mocks.groupIdByTabId).toEqual( + new Map([ + [FIRST_TAB_ID, GROUP_ID], + [SECOND_TAB_ID, GROUP_ID] + ]) + ) }) it('routes overlay interaction back to the owning split group', () => { @@ -166,7 +174,6 @@ function createState(activeTabId: string): MockAppState { }, groupsByWorktree: { [WORKTREE_ID]: [createGroup(activeTabId)] }, runtimeEnvironmentId: null, - executionHostId: 'local', focusGroup: mocks.focusGroup } } diff --git a/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.tsx b/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.tsx index 555d23b9ec4..01b8c256277 100644 --- a/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.tsx +++ b/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.tsx @@ -3,10 +3,7 @@ import { useShallow } from 'zustand/react/shallow' import type { Tab, TabGroup } from '../../../../shared/tab-types' import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' import { useAppStore } from '@/store' -import { - getExecutionHostIdForWorktree, - getRuntimeEnvironmentIdForWorktree -} from '@/lib/worktree-runtime-owner' +import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner' import { getActiveRuntimeTarget, type RuntimeClientTarget } from '@/runtime/runtime-rpc-client' import { tabGroupBodyAnchorName } from '../tab-group/tab-group-body-anchor' import NativeChatView from './NativeChatView' @@ -24,14 +21,12 @@ const StructuredAgentSessionOverlaySlot = memo(function StructuredAgentSessionOv groupId, isActive, target, - allowFileUriLinks, onFocusOwningGroup }: { tab: StructuredAgentSessionTab groupId: string | undefined isActive: boolean target: RuntimeClientTarget - allowFileUriLinks: boolean onFocusOwningGroup: ((groupId: string) => void) | undefined }): React.JSX.Element { const anchorName = groupId !== undefined ? tabGroupBodyAnchorName(groupId) : undefined @@ -69,11 +64,11 @@ const StructuredAgentSessionOverlaySlot = memo(function StructuredAgentSessionOv ) @@ -87,12 +82,11 @@ const StructuredAgentSessionPaneOverlayLayer = memo( worktreeId: string isWorktreeActive: boolean }): React.JSX.Element { - const { unifiedTabs, groups, runtimeEnvironmentId, allowFileUriLinks } = useAppStore( + const { unifiedTabs, groups, runtimeEnvironmentId } = useAppStore( useShallow((state) => ({ unifiedTabs: state.unifiedTabsByWorktree[worktreeId] ?? EMPTY_UNIFIED_TABS, groups: state.groupsByWorktree[worktreeId] ?? EMPTY_GROUPS, - runtimeEnvironmentId: getRuntimeEnvironmentIdForWorktree(state, worktreeId), - allowFileUriLinks: getExecutionHostIdForWorktree(state, worktreeId) === 'local' + runtimeEnvironmentId: getRuntimeEnvironmentIdForWorktree(state, worktreeId) })) ) const focusGroup = useAppStore((state) => state.focusGroup) @@ -127,7 +121,6 @@ const StructuredAgentSessionPaneOverlayLayer = memo( groupId={tab.groupId} isActive={Boolean(isWorktreeActive && groupActiveTabById.get(tab.groupId) === tab.id)} target={target} - allowFileUriLinks={allowFileUriLinks} onFocusOwningGroup={focusOwningGroup} /> ))} diff --git a/src/renderer/src/components/native-chat/native-chat-layout-actions.test.ts b/src/renderer/src/components/native-chat/native-chat-layout-actions.test.ts new file mode 100644 index 00000000000..6f3a3868bdf --- /dev/null +++ b/src/renderer/src/components/native-chat/native-chat-layout-actions.test.ts @@ -0,0 +1,57 @@ +import { describe, expect, it } from 'vitest' +import type { AppState } from '@/store/types' +import { + canRunNativeChatSplitTarget, + resolveActiveNativeChatSplitTarget +} from './native-chat-layout-actions' + +function stateWithActiveTab(tab: Record, tabOrder = ['chat', 'other']) { + return { + groupsByWorktree: { + workspace: [{ id: 'group', worktreeId: 'workspace', activeTabId: 'chat', tabOrder }] + }, + unifiedTabsByWorktree: { + workspace: [ + { + id: 'chat', + entityId: 'session', + groupId: 'group', + worktreeId: 'workspace', + ...tab + } + ] + } + } as unknown as Pick +} + +describe('native chat layout actions', () => { + it('resolves structured chats to the reusable workspace-tab move path', () => { + const state = stateWithActiveTab({ contentType: 'agent-session' }) + const target = resolveActiveNativeChatSplitTarget(state, 'workspace', 'group') + + expect(target).toEqual({ kind: 'workspace-tab', unifiedTabId: 'chat', groupId: 'group' }) + expect(canRunNativeChatSplitTarget(state, target)).toBe(true) + expect( + canRunNativeChatSplitTarget( + stateWithActiveTab({ contentType: 'agent-session' }, ['chat']), + target + ) + ).toBe(false) + }) + + it('resolves terminal-backed chat mode to the existing pane split path', () => { + const state = stateWithActiveTab({ contentType: 'terminal', viewMode: 'chat' }) + + expect(resolveActiveNativeChatSplitTarget(state, 'workspace', 'group')).toEqual({ + kind: 'terminal-pane', + terminalTabId: 'session' + }) + expect( + resolveActiveNativeChatSplitTarget( + stateWithActiveTab({ contentType: 'terminal', viewMode: 'terminal' }), + 'workspace', + 'group' + ) + ).toBeNull() + }) +}) diff --git a/src/renderer/src/components/native-chat/native-chat-layout-actions.ts b/src/renderer/src/components/native-chat/native-chat-layout-actions.ts new file mode 100644 index 00000000000..4f24a8a663c --- /dev/null +++ b/src/renderer/src/components/native-chat/native-chat-layout-actions.ts @@ -0,0 +1,74 @@ +import type { AppState } from '@/store/types' +import { useAppStore } from '@/store' +import { + canMoveTabToNewPaneColumnFromState, + moveTabToNewPaneColumn +} from '@/components/tab-bar/tab-move-to-pane-column' +import { requestActiveTerminalPaneSplit } from '@/components/tab-bar/request-active-terminal-pane-split' +import type { NativeChatSplitDirection } from './native-chat-split-shortcut' + +export type NativeChatSplitTarget = + | { kind: 'terminal-pane'; terminalTabId: string } + | { kind: 'workspace-tab'; unifiedTabId: string; groupId: string } + +export function resolveActiveNativeChatSplitTarget( + state: Pick, + worktreeId: string | null, + groupId: string | null +): NativeChatSplitTarget | null { + if (!worktreeId || !groupId) { + return null + } + const group = (state.groupsByWorktree?.[worktreeId] ?? []).find((entry) => entry.id === groupId) + const tab = (state.unifiedTabsByWorktree?.[worktreeId] ?? []).find( + (entry) => entry.id === group?.activeTabId && entry.groupId === groupId + ) + if (tab?.contentType === 'agent-session') { + return { kind: 'workspace-tab', unifiedTabId: tab.id, groupId } + } + if (tab?.contentType === 'terminal' && tab.viewMode === 'chat') { + return { kind: 'terminal-pane', terminalTabId: tab.entityId } + } + return null +} + +export function canRunNativeChatSplitTarget( + state: Pick, + target: NativeChatSplitTarget | null +): boolean { + if (!target) { + return false + } + return ( + target.kind === 'terminal-pane' || + canMoveTabToNewPaneColumnFromState(state, target.unifiedTabId, target.groupId) + ) +} + +export function runNativeChatSplitTarget( + target: NativeChatSplitTarget, + direction: NativeChatSplitDirection +): boolean { + if (target.kind === 'terminal-pane') { + requestActiveTerminalPaneSplit({ + tabId: target.terminalTabId, + direction: direction === 'right' ? 'vertical' : 'horizontal' + }) + return true + } + return moveTabToNewPaneColumn({ + unifiedTabId: target.unifiedTabId, + groupId: target.groupId, + direction + }) +} + +export function runActiveNativeChatSplit( + worktreeId: string | null, + groupId: string | null, + direction: NativeChatSplitDirection +): boolean { + const state = useAppStore.getState() + const target = resolveActiveNativeChatSplitTarget(state, worktreeId, groupId) + return target ? runNativeChatSplitTarget(target, direction) : false +} diff --git a/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts b/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts index 9dae22da197..8cf82b69ca1 100644 --- a/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts @@ -117,6 +117,7 @@ describe('native chat PTY session options', () => { expect(effortResult.snapshot.map(({ id }) => id)).toEqual(['model', 'effort', 'fastMode']) expect(effortResult.snapshot.find(({ id }) => id === 'effort')).toMatchObject({ valueSource: 'dispatched', + transport: 'catalog', kind: { currentValue: 'high' } }) expect(listener).toHaveBeenCalledOnce() diff --git a/src/renderer/src/components/native-chat/native-chat-pty-session-options.ts b/src/renderer/src/components/native-chat/native-chat-pty-session-options.ts index aa7562b5944..3e3587e68d1 100644 --- a/src/renderer/src/components/native-chat/native-chat-pty-session-options.ts +++ b/src/renderer/src/components/native-chat/native-chat-pty-session-options.ts @@ -98,7 +98,8 @@ export function createNativeChatPtySessionOptions( catalog, models: activeModels(), record, - mode: args.mode + mode: args.mode, + liveTransport: 'catalog' }) const listeners = new Set<(value: SessionOptionDescriptor[]) => void>() @@ -108,7 +109,8 @@ export function createNativeChatPtySessionOptions( catalog, models: activeModels(), record, - mode: args.mode + mode: args.mode, + liveTransport: 'catalog' }) for (const listener of listeners) { listener(snapshot) diff --git a/src/renderer/src/components/native-chat/native-chat-session-option-labels.test.ts b/src/renderer/src/components/native-chat/native-chat-session-option-labels.test.ts index 60d0ffe6aba..7cdf621b272 100644 --- a/src/renderer/src/components/native-chat/native-chat-session-option-labels.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-session-option-labels.test.ts @@ -18,6 +18,7 @@ function modelDescriptor( id: 'model', label: 'Model', valueSource, + transport: 'catalog', settable: true, kind: { type: 'select', diff --git a/src/renderer/src/components/native-chat/native-chat-session-option-snapshot.ts b/src/renderer/src/components/native-chat/native-chat-session-option-snapshot.ts index 9010aa66f27..4acb1a1932a 100644 --- a/src/renderer/src/components/native-chat/native-chat-session-option-snapshot.ts +++ b/src/renderer/src/components/native-chat/native-chat-session-option-snapshot.ts @@ -7,6 +7,7 @@ import { buildNativeChatSessionOptionSnapshot as buildSharedSnapshot, resolveEffectiveNativeChatModelId, withTrackedNativeChatModel, + type NativeChatLiveOptionTransport, type NativeChatSessionOptionMode } from '../../../../shared/native-chat-session-option-snapshot' import { @@ -15,7 +16,7 @@ import { } from '../../../../shared/native-chat-session-option-state' import { translate } from '@/i18n/i18n' -export type { NativeChatSessionOptionMode } +export type { NativeChatLiveOptionTransport, NativeChatSessionOptionMode } export { flattenNativeChatSessionOptionRecord, resolveEffectiveNativeChatModelId, @@ -27,6 +28,7 @@ export function buildNativeChatSessionOptionSnapshot(args: { models: readonly CatalogModel[] record: NativeChatSessionOptionRecord mode: NativeChatSessionOptionMode + liveTransport: NativeChatLiveOptionTransport }): SessionOptionDescriptor[] { return buildSharedSnapshot({ ...args, diff --git a/src/renderer/src/components/native-chat/native-chat-shared-copy-matches-catalog.test.ts b/src/renderer/src/components/native-chat/native-chat-shared-copy-matches-catalog.test.ts new file mode 100644 index 00000000000..d73bdc81802 --- /dev/null +++ b/src/renderer/src/components/native-chat/native-chat-shared-copy-matches-catalog.test.ts @@ -0,0 +1,43 @@ +import { describe, expect, it } from 'vitest' +import en from '@/i18n/locales/en.json' +import { NATIVE_CHAT_TOOL_ACTIVITY_COPY } from '../../../../shared/native-chat-tool-activity' +import { NATIVE_CHAT_TURN_STATUS_COPY } from '../../../../shared/native-chat-turn-status' + +// The shared copy is desktop's i18n fallback and mobile's actual rendered string. +// If the two drift, desktop keeps showing en.json while mobile shows the constant — +// silently, since neither side errors. These are the exact keys that made these +// strings runtime-required (i18next can no longer rebuild them from a literal +// call-site default), so they must stay byte-identical to the catalog. +const catalog = en as unknown as { + components: { + 'native-chat': { + status: Record + tool: Record + } + } +} + +describe('native-chat shared copy matches the English catalog', () => { + it.each(Object.entries(NATIVE_CHAT_TURN_STATUS_COPY))( + 'status.%s matches en.json', + (key, value) => { + expect(catalog.components['native-chat'].status[key]).toBe(value) + } + ) + + it.each(Object.entries(NATIVE_CHAT_TOOL_ACTIVITY_COPY))( + 'tool.%s matches en.json', + (key, value) => { + expect(catalog.components['native-chat'].tool[key]).toBe(value) + } + ) + + it('keeps the interpolation placeholders the catalog expects', () => { + expect(NATIVE_CHAT_TURN_STATUS_COPY.workedFor).toContain('{{value0}}') + expect(NATIVE_CHAT_TURN_STATUS_COPY.workingFor).toContain('{{value0}}') + expect(NATIVE_CHAT_TOOL_ACTIVITY_COPY.countN).toContain('{{value0}}') + expect(NATIVE_CHAT_TOOL_ACTIVITY_COPY.runningPreview).toContain('{{preview}}') + expect(NATIVE_CHAT_TOOL_ACTIVITY_COPY.runningNamedPreview).toContain('{{toolName}}') + expect(NATIVE_CHAT_TOOL_ACTIVITY_COPY.runningNamedPreview).toContain('{{preview}}') + }) +}) diff --git a/src/renderer/src/components/native-chat/native-chat-split-shortcut.test.ts b/src/renderer/src/components/native-chat/native-chat-split-shortcut.test.ts new file mode 100644 index 00000000000..0a5b2bd6688 --- /dev/null +++ b/src/renderer/src/components/native-chat/native-chat-split-shortcut.test.ts @@ -0,0 +1,38 @@ +import { describe, expect, it } from 'vitest' +import { matchNativeChatSplitShortcut } from './native-chat-split-shortcut' + +describe('matchNativeChatSplitShortcut', () => { + it('uses the existing platform split bindings', () => { + expect( + matchNativeChatSplitShortcut( + { key: 'd', metaKey: true, ctrlKey: false, altKey: false, shiftKey: false }, + 'darwin', + {} + ) + ).toBe('right') + expect( + matchNativeChatSplitShortcut( + { key: 'd', metaKey: false, ctrlKey: true, altKey: false, shiftKey: true }, + 'win32', + {} + ) + ).toBe('right') + expect( + matchNativeChatSplitShortcut( + { key: 'd', metaKey: false, ctrlKey: false, altKey: true, shiftKey: true }, + 'linux', + {} + ) + ).toBe('down') + }) + + it('respects customized bindings', () => { + expect( + matchNativeChatSplitShortcut( + { key: 'ArrowRight', metaKey: false, ctrlKey: true, altKey: true, shiftKey: false }, + 'linux', + { 'terminal.splitRight': ['Ctrl+Alt+Right'] } + ) + ).toBe('right') + }) +}) diff --git a/src/renderer/src/components/native-chat/native-chat-split-shortcut.ts b/src/renderer/src/components/native-chat/native-chat-split-shortcut.ts new file mode 100644 index 00000000000..5ded10fa6bb --- /dev/null +++ b/src/renderer/src/components/native-chat/native-chat-split-shortcut.ts @@ -0,0 +1,21 @@ +import { + keybindingMatchesAction, + type KeybindingInput, + type KeybindingOverrides +} from '../../../../shared/keybindings' + +export type NativeChatSplitDirection = 'right' | 'down' + +export function matchNativeChatSplitShortcut( + input: KeybindingInput, + platform: NodeJS.Platform, + keybindings: KeybindingOverrides +): NativeChatSplitDirection | null { + if (keybindingMatchesAction('terminal.splitRight', input, platform, keybindings)) { + return 'right' + } + if (keybindingMatchesAction('terminal.splitDown', input, platform, keybindings)) { + return 'down' + } + return null +} diff --git a/src/renderer/src/components/native-chat/native-chat-view-types.ts b/src/renderer/src/components/native-chat/native-chat-view-types.ts index 106b29b429d..920bede7028 100644 --- a/src/renderer/src/components/native-chat/native-chat-view-types.ts +++ b/src/renderer/src/components/native-chat/native-chat-view-types.ts @@ -37,11 +37,12 @@ export type NativeChatBridgeViewProps = NativeChatOrchestrationProps & { export type NativeChatStructuredViewProps = NativeChatOrchestrationProps & { mode: 'structured' tabId: string + groupId?: string sessionId: string target: RuntimeClientTarget agent: AgentType isVisible: boolean - allowFileUriLinks: boolean + contextMenuActions?: Omit } export type NativeChatResolvedViewProps = NativeChatOrchestrationProps & { diff --git a/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx b/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx index dfea16ae0e4..e4bae292004 100644 --- a/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx +++ b/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx @@ -3,7 +3,7 @@ import { act } from 'react' import { createRoot, type Root } from 'react-dom/client' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import { NativeChatWorkingStatus } from './NativeChatWorkingStatus' +import { formatNativeChatDuration, NativeChatWorkingStatus } from './NativeChatWorkingStatus' let container: HTMLDivElement let root: Root @@ -49,34 +49,50 @@ afterEach(() => { }) describe('native chat working status elapsed clock', () => { + it.each([ + [0, '0s'], + [59, '59s'], + [60, '1m 0s'], + [69, '1m 9s'], + [3_725, '1h 2m 5s'] + ])('formats %s seconds as %s', (seconds, expected) => { + expect(formatNativeChatDuration(seconds)).toBe(expected) + }) + + it('renders the compact duration in the completed status label', () => { + act(() => { + root.render( + + ) + }) + + expect(elapsedLabels()).toEqual(['Worked for 1m 9s']) + }) + it('collapses every in-flight turn onto one shared visibility-gated timer', () => { renderTurns(3, 1_000_000) // One shared 1s clock for all three turns, not one interval per turn. expect(vi.getTimerCount()).toBe(1) act(() => vi.advanceTimersByTime(3_000)) - expect(elapsedLabels()).toEqual([ - 'Working for 3 seconds', - 'Working for 3 seconds', - 'Working for 3 seconds' - ]) + expect(elapsedLabels()).toEqual(['Working for 3s', 'Working for 3s', 'Working for 3s']) }) it('stops ticking while hidden and re-syncs the elapsed value on return', () => { renderTurns(1, 1_000_000) act(() => vi.advanceTimersByTime(3_000)) - expect(elapsedLabels()).toEqual(['Working for 3 seconds']) + expect(elapsedLabels()).toEqual(['Working for 3s']) setDocumentVisibility('hidden') expect(vi.getTimerCount()).toBe(0) // A minute of hidden wall-clock: no callbacks, no commits, label frozen. act(() => vi.advanceTimersByTime(60_000)) - expect(elapsedLabels()).toEqual(['Working for 3 seconds']) + expect(elapsedLabels()).toEqual(['Working for 3s']) // Returning re-derives elapsed from startedAt, so nothing was lost. setDocumentVisibility('visible') - expect(elapsedLabels()).toEqual(['Working for 63 seconds']) + expect(elapsedLabels()).toEqual(['Working for 1m 3s']) expect(vi.getTimerCount()).toBe(1) }) diff --git a/src/renderer/src/components/native-chat/use-native-chat-context-menu.test.tsx b/src/renderer/src/components/native-chat/use-native-chat-context-menu.test.tsx index f4e8b3efc72..14841e6c56d 100644 --- a/src/renderer/src/components/native-chat/use-native-chat-context-menu.test.tsx +++ b/src/renderer/src/components/native-chat/use-native-chat-context-menu.test.tsx @@ -3,7 +3,8 @@ */ import React, { createRef, type ReactNode } from 'react' import { renderToStaticMarkup } from 'react-dom/server' -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { cleanup, render } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { emptyNativeChatContextMenuActions, useNativeChatContextMenu, @@ -52,6 +53,10 @@ vi.mock('@/i18n/i18n', () => ({ translate: (_key: string, fallback: string) => fallback })) +vi.mock('@/components/tab-bar/TabWorkspaceLayoutMenuSection', () => ({ + TabWorkspaceLayoutMenuSection: () => 'Move Tab to Split' +})) + function childrenText(children: ReactNode): string { return React.Children.toArray(children) .map((child) => { @@ -65,11 +70,22 @@ function childrenText(children: ReactNode): string { .join('') } -function Harness({ onSwitchToTerminal }: { onSwitchToTerminal?: () => void }) { +function Harness({ + onSwitchToTerminal, + structured = false, + enabled = true +}: { + onSwitchToTerminal?: () => void + structured?: boolean + enabled?: boolean +}) { const rootRef = createRef() const { menu } = useNativeChatContextMenu({ rootRef, + enabled, onSwitchToTerminal, + showTerminalPaneActions: !structured, + workspaceLayout: structured ? { unifiedTabId: 'chat-tab', groupId: 'group-1' } : undefined, actions: { ...emptyNativeChatContextMenuActions, onPaste: vi.fn() @@ -83,6 +99,11 @@ describe('useNativeChatContextMenu', () => { items.list = [] }) + afterEach(() => { + cleanup() + vi.restoreAllMocks() + }) + it('restores the bridge switch-to-terminal action when supplied', () => { const onSwitchToTerminal = vi.fn() @@ -107,4 +128,31 @@ describe('useNativeChatContextMenu', () => { items.list.some((candidate) => childrenText(candidate.children) === 'Switch to terminal view') ).toBe(false) }) + + it('reuses workspace layout actions without terminal-only pane commands', () => { + const markup = renderToStaticMarkup() + + expect(markup).toContain('Move Tab to Split') + expect(markup).not.toContain('Split Terminal Right') + expect(markup).not.toContain('Fork Agent Session') + }) + + it('subscribes to selection changes only while its retained chat is visible', () => { + const getSelection = vi.spyOn(window, 'getSelection').mockReturnValue(null) + const view = render() + + getSelection.mockClear() + document.dispatchEvent(new Event('selectionchange')) + expect(getSelection).not.toHaveBeenCalled() + + view.rerender() + getSelection.mockClear() + document.dispatchEvent(new Event('selectionchange')) + expect(getSelection).toHaveBeenCalledOnce() + + view.rerender() + getSelection.mockClear() + document.dispatchEvent(new Event('selectionchange')) + expect(getSelection).not.toHaveBeenCalled() + }) }) diff --git a/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx b/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx index a2d7dae4c93..48d896d4aea 100644 --- a/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx +++ b/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx @@ -30,6 +30,8 @@ import { } from '@/components/ui/dropdown-menu' import { translate } from '@/i18n/i18n' import { isMacPlatform, nativeChatToggleShortcutLabel } from './native-chat-shortcut' +import { TabWorkspaceLayoutMenuSection } from '@/components/tab-bar/TabWorkspaceLayoutMenuSection' +import type { TabSplitDirection } from '@/store/slices/tabs' type NativeChatContextMenuState = { open: boolean @@ -39,9 +41,17 @@ type NativeChatContextMenuState = { type UseNativeChatContextMenuArgs = { rootRef: RefObject - /** Bridge-only escape hatch; structured sessions never mount this menu. */ + enabled?: boolean + /** Bridge-only escape hatch; structured sessions have no terminal view. */ onSwitchToTerminal?: () => void actions: NativeChatContextMenuActions + showTerminalPaneActions?: boolean + splitShortcutLabels?: { right: string; down: string } + workspaceLayout?: { + unifiedTabId: string + groupId: string + shortcutLabels?: Partial> + } } export type NativeChatContextMenuActions = { @@ -88,8 +98,12 @@ export const emptyNativeChatContextMenuActions: Omit onSelectionCapture: () => void @@ -112,9 +126,18 @@ export function useNativeChatContextMenu({ }, [rootRef]) useEffect(() => { + if (!enabled) { + return + } document.addEventListener('selectionchange', rememberCurrentSelection) return () => document.removeEventListener('selectionchange', rememberCurrentSelection) - }, [rememberCurrentSelection]) + }, [enabled, rememberCurrentSelection]) + + useEffect(() => { + if (!enabled) { + setState((current) => (current.open ? { ...current, open: false } : current)) + } + }, [enabled]) const onContextMenuCapture = useCallback( (event: React.MouseEvent) => { @@ -142,7 +165,7 @@ export function useNativeChatContextMenu({ onContextMenuCapture, onSelectionCapture: rememberCurrentSelection, menu: ( - + +
+ +
+ + ) +} + +let container: HTMLDivElement +let root: Root + +beforeEach(() => { + // Radix Presence retains closing content when the production exit animation runs. + const getStyle = window.getComputedStyle.bind(window) + vi.spyOn(window, 'getComputedStyle').mockImplementation((element, ...args) => { + const style = getStyle(element, ...args) + if (element.getAttribute('data-slot') === 'popover-content') { + return new Proxy(style, { + get: (target, property) => + property === 'animationName' + ? element.getAttribute('data-state') === 'closed' + ? 'exit' + : 'enter' + : Reflect.get(target, property) + }) + } + return style + }) + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) +}) + +afterEach(async () => { + await act(async () => root.unmount()) + container.remove() + vi.restoreAllMocks() +}) + +function projectField(): HTMLInputElement { + return document.querySelector('[role="combobox"][aria-label="Project"]')! +} + +function addOption(): HTMLElement { + return Array.from(document.querySelectorAll('[role="option"]')).find((option) => + option.textContent?.includes('Add a new project') + )! +} + +describe.each([false, true])( + 'project selector to nested dialog handoff (populated: %s)', + (populated) => { + it.each(['mouse', 'keyboard'] as const)( + 'removes the selector before the creation dialog is active via %s', + async (input) => { + await act(async () => root.render()) + await act(async () => projectField().click()) + expect(document.querySelector('[role="listbox"]')).not.toBeNull() + + await act(async () => { + if (input === 'mouse') { + addOption().dispatchEvent( + new MouseEvent('mousedown', { bubbles: true, cancelable: true }) + ) + addOption().click() + } else if (populated) { + projectField().dispatchEvent( + new KeyboardEvent('keydown', { key: 'ArrowDown', bubbles: true, cancelable: true }) + ) + } + }) + if (input === 'keyboard') { + await act(async () => { + projectField().dispatchEvent( + new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }) + ) + }) + } + + expect(document.querySelector('[role="listbox"]')).toBeNull() + expect(document.querySelector('[data-slot="popover-content"]')).toBeNull() + expect(projectField().getAttribute('aria-expanded')).toBe('false') + expect(projectField().hasAttribute('aria-activedescendant')).toBe(false) + const path = document.querySelector('[aria-label="Project path"]')! + expect(document.activeElement).toBe(path) + expect(path.closest('[aria-hidden="true"]')).toBeNull() + expect(projectField().closest('[aria-hidden="true"]')).not.toBeNull() + + await act(async () => { + Array.from(document.querySelectorAll('button')) + .find((b) => b.textContent === 'Cancel')! + .click() + }) + await vi.waitFor(() => + expect(document.activeElement).toBe( + document.querySelector('[aria-label="Workspace name"]') + ) + ) + await act(async () => projectField().click()) + expect(projectField().getAttribute('aria-expanded')).toBe('true') + expect(document.querySelector('[role="listbox"]')).not.toBeNull() + } + ) + } +) diff --git a/src/renderer/src/components/new-workspace/ProjectCombobox.test.tsx b/src/renderer/src/components/new-workspace/ProjectCombobox.test.tsx index 3b7931238b3..aa6b8df7b2d 100644 --- a/src/renderer/src/components/new-workspace/ProjectCombobox.test.tsx +++ b/src/renderer/src/components/new-workspace/ProjectCombobox.test.tsx @@ -72,7 +72,7 @@ function field(): HTMLInputElement { function openList(): void { act(() => { - field().dispatchEvent(new FocusEvent('focus', { bubbles: true })) + field().focus() }) } diff --git a/src/renderer/src/components/new-workspace/ProjectCombobox.tsx b/src/renderer/src/components/new-workspace/ProjectCombobox.tsx index c305964ca0e..8561465d57e 100644 --- a/src/renderer/src/components/new-workspace/ProjectCombobox.tsx +++ b/src/renderer/src/components/new-workspace/ProjectCombobox.tsx @@ -229,129 +229,132 @@ export default function ProjectCombobox({ - event.preventDefault()} - onCloseAutoFocus={(event) => event.preventDefault()} - // Why: the field lives in the anchor, not inside the content, so Radix - // sees a focus/pointer event "outside" the layer and dismisses it the - // instant you tab in. Keep the layer open whenever the interaction is - // within this control; genuine outside events still close it. - onFocusOutside={(event) => { - if (isWithinComboboxRoot(event.target, ROOT_ATTRIBUTE)) { - event.preventDefault() - } - }} - onInteractOutside={(event) => { - if (isWithinComboboxRoot(event.target, ROOT_ATTRIBUTE)) { - event.preventDefault() - } - }} - > - {/* The listbox wraps a scrolling pane plus the pinned Add row, so both - stay `option` children of one listbox. */} -
event.preventDefault()} + onCloseAutoFocus={(event) => event.preventDefault()} + // Why: the field lives in the anchor, not inside the content, so Radix + // sees a focus/pointer event "outside" the layer and dismisses it the + // instant you tab in. Keep the layer open whenever the interaction is + // within this control; genuine outside events still close it. + onFocusOutside={(event) => { + if (isWithinComboboxRoot(event.target, ROOT_ATTRIBUTE)) { + event.preventDefault() + } + }} + onInteractOutside={(event) => { + if (isWithinComboboxRoot(event.target, ROOT_ATTRIBUTE)) { + event.preventDefault() + } + }} > - {/* Why: `presentation` — this element exists to scroll, and an + {/* The listbox wraps a scrolling pane plus the pinned Add row, so both + stay `option` children of one listbox. */} +
+ {/* Why: `presentation` — this element exists to scroll, and an unroled div between a listbox and its options breaks the ownership relationship assistive tech relies on. */} -
- {matches.length === 0 ? ( - // Why: row-height rather than a tall centred block — a 60px panel - // next to 32px rows reads as a different kind of surface and - // makes an empty result feel like an error. -

- {options.length === 0 - ? translate( - 'auto.components.new.workspace.ProjectCombobox.noProjects', - 'No projects yet.' - ) - : translate( - 'auto.components.new.workspace.ProjectCombobox.empty', - 'No projects match your search.' - )} -

- ) : null} - {sections.map((section) => ( - // Why: `role="group"` — a bare div between a listbox and its - // options breaks the ownership relationship for screen readers, - // which is the only thing that makes the headings announceable. -
- {section.heading ? ( - - ) : null} - {section.items.map((scored) => ( - arm(scored.option.id)} - onCommit={() => commit(scored.option.id)} - /> - ))} -
- ))} -
- {onAddProject ? (
event.preventDefault()} - onMouseMove={() => arm(ADD_PROJECT_KEY)} - onClick={() => commit(ADD_PROJECT_KEY)} - className={cn( - 'flex h-9 shrink-0 cursor-default items-center gap-2 border-t border-border px-2 text-sm', - armedKey === ADD_PROJECT_KEY && 'bg-accent text-accent-foreground' - )} + ref={setListNode} + role="presentation" + className="max-h-72 min-h-0 flex-1 overflow-y-auto p-1 scrollbar-sleek" > - - - {translate( - 'auto.components.new.workspace.ProjectCombobox.addProject', - 'Add a new project' - )} - + {matches.length === 0 ? ( + // Why: row-height rather than a tall centred block — a 60px panel + // next to 32px rows reads as a different kind of surface and + // makes an empty result feel like an error. +

+ {options.length === 0 + ? translate( + 'auto.components.new.workspace.ProjectCombobox.noProjects', + 'No projects yet.' + ) + : translate( + 'auto.components.new.workspace.ProjectCombobox.empty', + 'No projects match your search.' + )} +

+ ) : null} + {sections.map((section) => ( + // Why: `role="group"` — a bare div between a listbox and its + // options breaks the ownership relationship for screen readers, + // which is the only thing that makes the headings announceable. +
+ {section.heading ? ( + + ) : null} + {section.items.map((scored) => ( + arm(scored.option.id)} + onCommit={() => commit(scored.option.id)} + /> + ))} +
+ ))}
- ) : null} -
- + {onAddProject ? ( +
event.preventDefault()} + onMouseMove={() => arm(ADD_PROJECT_KEY)} + onClick={() => commit(ADD_PROJECT_KEY)} + className={cn( + 'flex h-9 shrink-0 cursor-default items-center gap-2 border-t border-border px-2 text-sm', + armedKey === ADD_PROJECT_KEY && 'bg-accent text-accent-foreground' + )} + > + + + {translate( + 'auto.components.new.workspace.ProjectCombobox.addProject', + 'Add a new project' + )} + +
+ ) : null} +
+
+ ) : null} ) } diff --git a/src/renderer/src/components/right-sidebar/source-control-tree.ts b/src/renderer/src/components/right-sidebar/source-control-tree.ts index d855201646b..df0938e9097 100644 --- a/src/renderer/src/components/right-sidebar/source-control-tree.ts +++ b/src/renderer/src/components/right-sidebar/source-control-tree.ts @@ -128,12 +128,16 @@ export function buildSourceControlTree< } let parent = root + // Why accumulated and only materialized on a miss: `slice().join('/')` per segment made + // tree building O(files x depth^2) in characters copied, and the Source Control filter + // rebuilds this whole tree on every keystroke. + let ancestorPath = '' for (let index = 0; index < segments.length - 1; index += 1) { const name = segments[index] - const path = segments.slice(0, index + 1).join('/') + ancestorPath = ancestorPath ? `${ancestorPath}/${name}` : name let dir = parent.directoryChildren.get(name) if (!dir) { - dir = makeDirectoryNode(area, path, name, index) + dir = makeDirectoryNode(area, ancestorPath, name, index) parent.directoryChildren.set(name, dir) parent.children.push(dir) } diff --git a/src/renderer/src/components/settings/ExperimentalPane.test.tsx b/src/renderer/src/components/settings/ExperimentalPane.test.tsx index b421b77bfc2..ba8e1518921 100644 --- a/src/renderer/src/components/settings/ExperimentalPane.test.tsx +++ b/src/renderer/src/components/settings/ExperimentalPane.test.tsx @@ -258,8 +258,12 @@ describe('ExperimentalPane', () => { }) expect(container.textContent).toContain('Use updated structured native chat') + // The one opt-in gates both providers, so its copy must not name only Codex. expect(container.textContent).toContain( - 'Local macOS and Linux sessions only for now. Windows, WSL, and remote execution hosts (including SSH) continue to use terminal chat.' + 'Opt in to the host-owned structured chat runtime for Codex and Claude.' + ) + expect(container.textContent).toContain( + 'Local sessions only for now. WSL and remote execution hosts (including SSH) continue to use terminal chat, and Windows falls back to it unless Orca can read process start times.' ) expect(container.textContent).toContain('Default view') root.unmount() diff --git a/src/renderer/src/components/settings/MobileSettingsPane.tsx b/src/renderer/src/components/settings/MobileSettingsPane.tsx index 2e98a5ccb43..ff8de2b16e5 100644 --- a/src/renderer/src/components/settings/MobileSettingsPane.tsx +++ b/src/renderer/src/components/settings/MobileSettingsPane.tsx @@ -13,7 +13,7 @@ export { getMobileSettingsPaneSearchEntries } const ORCA_IOS_APP_STORE_URL = 'https://apps.apple.com/app/orca-ide/id6766130217' const ORCA_ANDROID_APK_URL = - 'https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk' + 'https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk' export function MobileSettingsPane(): React.JSX.Element { const showMobileButton = useAppStore((s) => s.settings?.showMobileButton !== false) diff --git a/src/renderer/src/components/settings/NativeChatExperimentalSetting.tsx b/src/renderer/src/components/settings/NativeChatExperimentalSetting.tsx index 93c4b1c899d..85d27dae2e3 100644 --- a/src/renderer/src/components/settings/NativeChatExperimentalSetting.tsx +++ b/src/renderer/src/components/settings/NativeChatExperimentalSetting.tsx @@ -126,13 +126,13 @@ export function NativeChatExperimentalSetting({

{translate( 'auto.components.settings.ExperimentalPane.nativeChat.structuredCopy', - 'Opt in to the host-owned structured Codex runtime. Off keeps the existing terminal-backed chat path.' + 'Opt in to the host-owned structured chat runtime for Codex and Claude. Off keeps the existing terminal-backed chat path.' )}

{translate( 'auto.components.settings.ExperimentalPane.nativeChat.structuredScope', - 'Local macOS and Linux sessions only for now. Windows, WSL, and remote execution hosts (including SSH) continue to use terminal chat.' + 'Local sessions only for now. WSL and remote execution hosts (including SSH) continue to use terminal chat, and Windows falls back to it unless Orca can read process start times.' )}

diff --git a/src/renderer/src/components/sidebar/CommentMarkdown.link-click.test.tsx b/src/renderer/src/components/sidebar/CommentMarkdown.link-click.test.tsx index 575d4343c3b..a3f7c7aad1f 100644 --- a/src/renderer/src/components/sidebar/CommentMarkdown.link-click.test.tsx +++ b/src/renderer/src/components/sidebar/CommentMarkdown.link-click.test.tsx @@ -3,6 +3,10 @@ import { act } from 'react' import { createRoot, type Root } from 'react-dom/client' import { afterEach, describe, expect, it, vi } from 'vitest' +import { + NATIVE_CHAT_FILE_HREF_PREFIX, + routeNativeChatHref +} from '../../../../shared/native-chat-href-routing' import CommentMarkdown from './CommentMarkdown' describe('CommentMarkdown link click handler', () => { @@ -48,6 +52,72 @@ describe('CommentMarkdown link click handler', () => { expect(event.defaultPrevented).toBe(true) }) + it('intercepts auxiliary clicks on generated native file links', () => { + const onLinkClick = vi.fn((event: React.MouseEvent) => { + event.preventDefault() + }) + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const anchor = container.querySelector('a') + const event = new window.MouseEvent('auxclick', { + bubbles: true, + cancelable: true, + button: 1 + }) + + act(() => { + anchor?.dispatchEvent(event) + }) + + expect(onLinkClick).toHaveBeenCalledWith(expect.any(Object), expect.stringMatching(/^#orca-/)) + expect(event.defaultPrevented).toBe(true) + }) + + it('does not activate generated native file links on right-click', () => { + const onLinkClick = vi.fn() + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const anchor = container.querySelector('a') + const event = new window.MouseEvent('auxclick', { + bubbles: true, + cancelable: true, + button: 2 + }) + + act(() => { + anchor?.dispatchEvent(event) + }) + + expect(onLinkClick).not.toHaveBeenCalled() + expect(event.defaultPrevented).toBe(false) + }) + it('sanitizes file URI links unless the caller opts in', () => { container = document.createElement('div') document.body.appendChild(container) @@ -146,4 +216,388 @@ describe('CommentMarkdown link click handler', () => { expect(onLinkClick).toHaveBeenCalledWith(expect.any(Object), 'assets/diagram.png') expect(event.defaultPrevented).toBe(true) }) + + it('linkifies bare POSIX and Windows document paths without an extension allowlist', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const routes = Array.from(container.querySelectorAll('a')).map((anchor) => + routeNativeChatHref(anchor.getAttribute('href')) + ) + expect(routes).toEqual([ + { kind: 'file', pathText: '/tmp/sta-6481-explainer.html', line: null }, + { kind: 'file', pathText: 'docs/review.docx', line: null }, + { kind: 'file', pathText: String.raw`C:\Reports\final.pages`, line: null }, + { kind: 'file', pathText: './scripts/release', line: null }, + { kind: 'file', pathText: 'src/release:12', line: null } + ]) + }) + + it('makes an inline-code file path clickable while preserving code styling', () => { + const onLinkClick = vi.fn((event: React.MouseEvent) => event.preventDefault()) + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const code = container.querySelector('code') + const anchor = code?.closest('a') + expect(anchor).not.toBeNull() + expect(routeNativeChatHref(anchor?.getAttribute('href'))).toEqual({ + kind: 'file', + pathText: String.raw`C:\Reports\release.docx`, + line: null + }) + + act(() => { + anchor?.dispatchEvent(new window.MouseEvent('click', { bubbles: true, cancelable: true })) + }) + expect(onLinkClick).toHaveBeenCalledOnce() + }) + + it('leaves prose-shaped slash tokens and numeric versions unlinked', () => { + const proseFalsePositives = ['and/or', 'TCP/IP', '24/7', 'N/A', 'km/h', 'A/B test'] + const inlineCodeFalsePositives = ['origin/main', 'v1.2.3', '1.0'] + const quotedFalsePositives = ['"and/or"', '"A/B test"'] + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + `\`${value}\``).join(', ')}; ${quotedFalsePositives.join(', ')}`} + onLinkClick={vi.fn()} + linkifyFilePaths + /> + ) + }) + + expect(container.querySelectorAll('a')).toHaveLength(0) + for (const value of proseFalsePositives) { + expect(container.textContent).toContain(value) + } + for (const value of quotedFalsePositives) { + expect(container.textContent).toContain(value) + } + expect(Array.from(container.querySelectorAll('code')).map((code) => code.textContent)).toEqual( + inlineCodeFalsePositives + ) + }) + + it('links each relative path separately when prose joins them', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + expect(Array.from(container.querySelectorAll('a')).map((anchor) => anchor.textContent)).toEqual( + ['src/foo.ts', 'src/bar.ts', 'docs/My Folder/notes.md'] + ) + }) + + it('links quoted spaced-first-segment paths around apostrophes', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const anchors = container.querySelectorAll('a') + expect(Array.from(anchors).map((anchor) => anchor.textContent)).toEqual([ + "Brennan's Folder/notes.md", + 'My Folder/guide.md' + ]) + expect(container.textContent).toBe( + "Don't skip \"Brennan's Folder/notes.md\"; open 'My Folder/guide.md'." + ) + expect( + Array.from(anchors).map((anchor) => routeNativeChatHref(anchor.getAttribute('href'))) + ).toEqual([ + { kind: 'file', pathText: "Brennan's Folder/notes.md", line: null }, + { kind: 'file', pathText: 'My Folder/guide.md', line: null } + ]) + }) + + it('links a spaced-first-segment relative path when inline code disambiguates it', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const anchor = container.querySelector('a') + expect(anchor?.textContent).toBe('My Folder/notes.md') + expect(routeNativeChatHref(anchor?.getAttribute('href'))).toEqual({ + kind: 'file', + pathText: 'My Folder/notes.md', + line: null + }) + }) + + it('requires path shape before a spaced line suffix can make a link', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + expect(container.querySelectorAll('a')).toHaveLength(0) + expect(container.querySelector('code')?.textContent).toBe('aspect 16:9') + expect(container.textContent).toContain('"John 3:16"') + }) + + it('preserves line suffixes on valid spaced path shapes', () => { + const content = + 'Open "My Folder/notes:12", `My Notes.md:7`, and "C:\\My Folder\\notes.txt:12:3".' + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const anchors = Array.from(container.querySelectorAll('a')) + expect(anchors.map((anchor) => anchor.textContent)).toEqual([ + 'My Folder/notes:12', + 'My Notes.md:7', + String.raw`C:\My Folder\notes.txt:12:3` + ]) + expect(anchors.map((anchor) => routeNativeChatHref(anchor.getAttribute('href')))).toEqual([ + { kind: 'file', pathText: 'My Folder/notes:12', line: null }, + { kind: 'file', pathText: 'My Notes.md:7', line: null }, + { kind: 'file', pathText: String.raw`C:\My Folder\notes.txt:12:3`, line: null } + ]) + }) + + it('links complete Unicode paths and extensions that begin with a digit', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + expect(Array.from(container.querySelectorAll('a')).map((anchor) => anchor.textContent)).toEqual( + ['/tmp/报告.html', 'docs/报告/file.html', 'docs/café/report.pdf', 'docs/archive.7z'] + ) + }) + + it('never links an ASCII suffix inside a path containing an unsupported character', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + expect(container.querySelectorAll('a')).toHaveLength(0) + expect(container.textContent).toContain('/tmp/$draft/report.html') + }) + + it('links paths before common sentence punctuation', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + expect(Array.from(container.querySelectorAll('a')).map((anchor) => anchor.textContent)).toEqual( + ['src/foo.ts', 'docs/guide.md', 'assets/report.pdf'] + ) + }) + + it('links paths after CLI assignment and before Unicode sentence punctuation', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + expect(Array.from(container.querySelectorAll('a')).map((anchor) => anchor.textContent)).toEqual( + ['./config.yaml', 'docs/指南.md', 'docs/报告.pdf'] + ) + }) + + it('does not link partial paths across unsupported punctuation', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + expect(container.querySelectorAll('a')).toHaveLength(0) + }) + + it('prevents the default action for an unresolved internal file href', () => { + const onLinkClick = vi.fn() + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const anchor = container.querySelector('a') + expect(anchor?.getAttribute('href')).toMatch(new RegExp(`^${NATIVE_CHAT_FILE_HREF_PREFIX}`)) + const event = new window.MouseEvent('click', { bubbles: true, cancelable: true }) + + act(() => { + anchor?.dispatchEvent(event) + }) + + expect(onLinkClick).toHaveBeenCalledOnce() + expect(event.defaultPrevented).toBe(true) + }) + + it('normalizes Windows markdown hrefs but leaves fenced paths as source text', () => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + + act(() => { + root?.render( + + ) + }) + + const anchors = container.querySelectorAll('a') + expect(anchors).toHaveLength(1) + expect(routeNativeChatHref(anchors[0]?.getAttribute('href'))).toEqual({ + kind: 'file', + pathText: String.raw`C:\Reports\summary.pdf`, + line: null + }) + expect(container.querySelector('pre')?.textContent).toContain('/tmp/not-a-link.html') + }) }) diff --git a/src/renderer/src/components/sidebar/CommentMarkdown.tsx b/src/renderer/src/components/sidebar/CommentMarkdown.tsx index 8673baedefa..0999c23c8c6 100644 --- a/src/renderer/src/components/sidebar/CommentMarkdown.tsx +++ b/src/renderer/src/components/sidebar/CommentMarkdown.tsx @@ -13,6 +13,7 @@ import { isTrustedCompactImageSrc, type CommentMarkdownLinkClickHandler } from './comment-markdown-element-renderers' +import { remarkNativeChatFileLinks } from './comment-markdown-native-chat-file-links' export type { CommentMarkdownLinkClickHandler } from './comment-markdown-element-renderers' @@ -185,6 +186,7 @@ type CommentMarkdownProps = React.ComponentPropsWithoutRef<'div'> & { githubRepo?: GitHubRepoReference | null onLinkClick?: CommentMarkdownLinkClickHandler allowFileUriLinks?: boolean + linkifyFilePaths?: boolean expandImages?: boolean } @@ -200,6 +202,7 @@ const CommentMarkdown = React.memo( githubRepo, onLinkClick, allowFileUriLinks = false, + linkifyFilePaths = false, expandImages = false, ...rest }, @@ -217,10 +220,12 @@ const CommentMarkdown = React.memo( ? createDocumentCommentMarkdownComponents(onLinkClick) : createCompactCommentMarkdownComponents(onLinkClick, expandImages) }, [expandImages, variant, onLinkClick]) - const activeRemarkPlugins = React.useMemo( - () => (githubRepo ? [...remarkPlugins, remarkGitHubReferences(githubRepo)] : remarkPlugins), - [githubRepo] - ) + const activeRemarkPlugins = React.useMemo(() => { + const plugins = linkifyFilePaths + ? [...remarkPlugins, remarkNativeChatFileLinks] + : remarkPlugins + return githubRepo ? [...plugins, remarkGitHubReferences(githubRepo)] : plugins + }, [githubRepo, linkifyFilePaths]) return (
s.activeModal) @@ -101,8 +102,11 @@ const NonGitFolderDialog = React.memo(function NonGitFolderDialog() { ...(launch.startup ? { startup: launch.startup } : {}), ...(launch.route === 'structured-native-chat' ? { providesInitialSurface: true } : {}) }) - if (launch.route === 'structured-native-chat' && launch.agent === 'codex') { - const structured = startStructuredCodexLaunch(folderWorktree.id) + if ( + launch.route === 'structured-native-chat' && + isAgentSessionHandleProvider(launch.agent) + ) { + const structured = startStructuredAgentLaunch(folderWorktree.id, launch.agent) const fallback = structured.claimDefinitiveRefusalFallback(() => { activateAndRevealWorktree(folderWorktree.id, { sidebarRevealBehavior: 'auto', diff --git a/src/renderer/src/components/sidebar/WorktreeList.card-memo-stability.test.tsx b/src/renderer/src/components/sidebar/WorktreeList.card-memo-stability.test.tsx index 8278eb3482b..978ce0e180e 100644 --- a/src/renderer/src/components/sidebar/WorktreeList.card-memo-stability.test.tsx +++ b/src/renderer/src/components/sidebar/WorktreeList.card-memo-stability.test.tsx @@ -1,5 +1,9 @@ // @vitest-environment happy-dom +vi.mock('@/components/confirmation-dialog-context', () => ({ + useConfirmationDialog: () => vi.fn().mockResolvedValue(false) +})) + import { act, type ReactNode } from 'react' import { createRoot, type Root } from 'react-dom/client' import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' diff --git a/src/renderer/src/components/sidebar/WorktreeList.lineage-agent-expansion-coupling.test.tsx b/src/renderer/src/components/sidebar/WorktreeList.lineage-agent-expansion-coupling.test.tsx index ab4d4a605d4..a68908e5618 100644 --- a/src/renderer/src/components/sidebar/WorktreeList.lineage-agent-expansion-coupling.test.tsx +++ b/src/renderer/src/components/sidebar/WorktreeList.lineage-agent-expansion-coupling.test.tsx @@ -1,5 +1,9 @@ // @vitest-environment happy-dom +vi.mock('@/components/confirmation-dialog-context', () => ({ + useConfirmationDialog: () => vi.fn().mockResolvedValue(false) +})) + // Regression test for the child-worktrees <-> agent-list expansion coupling: // in a worktree card that shows BOTH inline agent rows (with orchestration // lineage) AND a "N children" child-worktrees chip, toggling the child-worktrees diff --git a/src/renderer/src/components/sidebar/WorktreeList.lineage-child-real-card.test.tsx b/src/renderer/src/components/sidebar/WorktreeList.lineage-child-real-card.test.tsx index a14d69a7267..a3f7e14b818 100644 --- a/src/renderer/src/components/sidebar/WorktreeList.lineage-child-real-card.test.tsx +++ b/src/renderer/src/components/sidebar/WorktreeList.lineage-child-real-card.test.tsx @@ -1,5 +1,9 @@ // @vitest-environment happy-dom +vi.mock('@/components/confirmation-dialog-context', () => ({ + useConfirmationDialog: () => vi.fn().mockResolvedValue(false) +})) + import { act, type ReactNode } from 'react' import { createRoot, type Root } from 'react-dom/client' import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' diff --git a/src/renderer/src/components/sidebar/WorktreeList.status-lane-lineage-drop.test.tsx b/src/renderer/src/components/sidebar/WorktreeList.status-lane-lineage-drop.test.tsx index 4e726dca60c..35919acae1d 100644 --- a/src/renderer/src/components/sidebar/WorktreeList.status-lane-lineage-drop.test.tsx +++ b/src/renderer/src/components/sidebar/WorktreeList.status-lane-lineage-drop.test.tsx @@ -1,5 +1,9 @@ // @vitest-environment happy-dom +vi.mock('@/components/confirmation-dialog-context', () => ({ + useConfirmationDialog: () => vi.fn().mockResolvedValue(false) +})) + import { act, type ReactNode } from 'react' import { createRoot, type Root } from 'react-dom/client' import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' diff --git a/src/renderer/src/components/sidebar/comment-markdown-element-renderers.tsx b/src/renderer/src/components/sidebar/comment-markdown-element-renderers.tsx index 18cef393c56..bb82a47356b 100644 --- a/src/renderer/src/components/sidebar/comment-markdown-element-renderers.tsx +++ b/src/renderer/src/components/sidebar/comment-markdown-element-renderers.tsx @@ -1,5 +1,6 @@ import React from 'react' import type { Components } from 'react-markdown' +import { NATIVE_CHAT_FILE_HREF_PREFIX } from '../../../../shared/native-chat-href-routing' import { isMermaidFence, isMermaidPre, renderMermaidFence } from './comment-mermaid-fence' import { GitHubUserAttachmentImage, @@ -32,12 +33,26 @@ function handleMarkdownAnchorClick( // Why: link clicks should not also trigger an outer row/card click handler; // images only claim the click when an image handler is wired below. event.stopPropagation() - if (href?.trim().toLowerCase().startsWith('file:')) { + const trimmedHref = href?.trim() + if ( + trimmedHref?.toLowerCase().startsWith('file:') || + trimmedHref?.startsWith(NATIVE_CHAT_FILE_HREF_PREFIX) + ) { event.preventDefault() } onLinkClick?.(event, href) } +function handleMarkdownAnchorAuxClick( + event: React.MouseEvent, + href: string | undefined, + onLinkClick: CommentMarkdownLinkClickHandler | undefined +): void { + if (event.button === 1) { + handleMarkdownAnchorClick(event, href, onLinkClick) + } +} + function handleMarkdownImageClick( event: React.MouseEvent, src: string | undefined, @@ -65,6 +80,7 @@ export function createCompactCommentMarkdownComponents( rel="noreferrer" className="underline underline-offset-2 text-foreground/80 hover:text-foreground" onClick={(e) => handleMarkdownAnchorClick(e, href, onLinkClick)} + onAuxClick={(e) => handleMarkdownAnchorAuxClick(e, href, onLinkClick)} > {children} @@ -154,6 +170,7 @@ export function createCompactCommentMarkdownComponents( rel="noreferrer" className="underline underline-offset-2 text-foreground/80 hover:text-foreground" onClick={(e) => handleMarkdownAnchorClick(e, src, onLinkClick)} + onAuxClick={(e) => handleMarkdownAnchorAuxClick(e, src, onLinkClick)} > {alt || src} @@ -184,6 +201,7 @@ export function createCompactCommentMarkdownComponents( target="_blank" rel="noreferrer" onClick={(e) => handleMarkdownAnchorClick(e, src, onLinkClick)} + onAuxClick={(e) => handleMarkdownAnchorAuxClick(e, src, onLinkClick)} > {image} @@ -221,6 +239,7 @@ export function createDocumentCommentMarkdownComponents( rel="noreferrer" className="break-all text-primary underline underline-offset-2 hover:text-primary/80" onClick={(e) => handleMarkdownAnchorClick(e, href, onLinkClick)} + onAuxClick={(e) => handleMarkdownAnchorAuxClick(e, href, onLinkClick)} > {children} diff --git a/src/renderer/src/components/sidebar/comment-markdown-native-chat-file-links.ts b/src/renderer/src/components/sidebar/comment-markdown-native-chat-file-links.ts new file mode 100644 index 00000000000..ee81b859dcc --- /dev/null +++ b/src/renderer/src/components/sidebar/comment-markdown-native-chat-file-links.ts @@ -0,0 +1,243 @@ +import { + createNativeChatFileHref, + routeNativeChatHref +} from '../../../../shared/native-chat-href-routing' +import { parseFileLinkLocation } from '../../../../shared/file-link-location' +import { extractTerminalFileLinks, type ParsedTerminalFileLink } from '@/lib/terminal-links' + +type MarkdownNode = { + type: string + value?: string + url?: string + children?: MarkdownNode[] +} + +const ROOTED_PATH_PREFIX_PATTERN = /^(?:~[\\/]|\.{1,2}[\\/]|[\\/]|[A-Za-z]:[\\/])/ + +function isLinkifiableFile(link: ParsedTerminalFileLink, requireSeparator: boolean): boolean { + const hasRootedPrefix = ROOTED_PATH_PREFIX_PATTERN.test(link.pathText) + const hasLineSuffix = link.line !== null || link.column !== null + const hasAlphabeticExtension = /\.[\p{L}][\p{L}\p{N}\p{M}_+-]*$/u.test(link.pathText) + const hasPathExtension = /\.[\p{L}\p{N}][\p{L}\p{N}\p{M}_+-]*$/u.test(link.pathText) + return ( + (!requireSeparator || /[\\/]/.test(link.pathText)) && + (hasRootedPrefix || + hasLineSuffix || + (requireSeparator ? hasPathExtension : hasAlphabeticExtension)) && + routeNativeChatHref(link.displayText).kind === 'file' + ) +} + +const SAFE_LEADING_BOUNDARY_PATTERN = /[\s([{'",;=]/ +const SAFE_TRAILING_BOUNDARY_PATTERN = /[\s)\]}>'",;.:。!?,、;:]/ +const SENTENCE_PATH_PUNCTUATION_PATTERN = + /\.[\p{L}\p{N}][\p{L}\p{N}\p{M}_+-]*([!?—。!?,、;:])/gu +const QUOTED_TEXT_PATTERN = /"([^"\r\n]+)"|'([^"'\r\n]+)'/gu +const MAX_DASHED_PROSE_WORD_LENGTH = 32 + +function hasBoundedProseAfterDash(value: string, startIndex: number): boolean { + const endIndex = Math.min(value.length, startIndex + MAX_DASHED_PROSE_WORD_LENGTH) + for (let index = startIndex; index < endIndex; index += 1) { + const char = value[index] + if (!char || SAFE_TRAILING_BOUNDARY_PATTERN.test(char)) { + return true + } + if (char === '/' || char === '\\') { + return false + } + } + return endIndex === value.length +} + +function isSafeTrailingBoundary(value: string, endIndex: number): boolean { + const boundary = value[endIndex] + if (boundary === undefined || SAFE_TRAILING_BOUNDARY_PATTERN.test(boundary)) { + return true + } + if (boundary === '!' || boundary === '?') { + const next = value[endIndex + 1] + return next === undefined || SAFE_TRAILING_BOUNDARY_PATTERN.test(next) + } + if (boundary === '—') { + return hasBoundedProseAfterDash(value, endIndex + 1) + } + return false +} + +function hasPartialPathBoundary(value: string, link: ParsedTerminalFileLink): boolean { + const before = value[link.startIndex - 1] + return ( + (before !== undefined && !SAFE_LEADING_BOUNDARY_PATTERN.test(before)) || + !isSafeTrailingBoundary(value, link.endIndex) + ) +} + +function createFileLinkNode(value: string, child: MarkdownNode): MarkdownNode { + return { + type: 'link', + url: createNativeChatFileHref(value), + children: [child] + } +} + +// Why: the terminal extractor spans "src/a.ts and src/b.ts" as one spaced path. +// An unrooted span holding a bare word or several linkable tokens is prose +// joining paths, so link the tokens on their own; a spaced folder name keeps +// every token path-shaped and stays one link. +function splitProseJoinedLinks(link: ParsedTerminalFileLink): ParsedTerminalFileLink[] { + if (ROOTED_PATH_PREFIX_PATTERN.test(link.pathText)) { + return [link] + } + const tokens = Array.from(link.displayText.matchAll(/\S+/g)) + const tokenLinks: ParsedTerminalFileLink[] = [] + for (const match of tokens) { + const token = match[0] + const exactLink = extractTerminalFileLinks(token).find( + (candidate) => candidate.startIndex === 0 && candidate.endIndex === token.length + ) + if (exactLink && isLinkifiableFile(exactLink, true)) { + const startIndex = link.startIndex + (match.index ?? 0) + tokenLinks.push({ ...exactLink, startIndex, endIndex: startIndex + token.length }) + } + } + const hasBareWord = tokens.some((match) => !/[\\/.]/.test(match[0])) + return hasBareWord || tokenLinks.length > 1 ? tokenLinks : [link] +} + +function splitTextSegment(value: string): MarkdownNode[] { + const links = extractTerminalFileLinks(value) + .filter((link) => !hasPartialPathBoundary(value, link)) + .filter((link) => isLinkifiableFile(link, true)) + .flatMap(splitProseJoinedLinks) + if (links.length === 0) { + return [{ type: 'text', value }] + } + + const children: MarkdownNode[] = [] + let cursor = 0 + for (const link of links) { + if (link.startIndex < cursor) { + continue + } + if (link.startIndex > cursor) { + children.push({ type: 'text', value: value.slice(cursor, link.startIndex) }) + } + children.push(createFileLinkNode(link.displayText, { type: 'text', value: link.displayText })) + cursor = link.endIndex + } + if (cursor < value.length) { + children.push({ type: 'text', value: value.slice(cursor) }) + } + return children +} + +function splitUnquotedText(value: string): MarkdownNode[] { + const children: MarkdownNode[] = [] + let cursor = 0 + for (const match of value.matchAll(SENTENCE_PATH_PUNCTUATION_PATTERN)) { + const punctuationIndex = (match.index ?? 0) + match[0].length - 1 + if (!isSafeTrailingBoundary(value, punctuationIndex)) { + continue + } + children.push(...splitTextSegment(value.slice(cursor, punctuationIndex))) + children.push({ type: 'text', value: value[punctuationIndex] }) + cursor = punctuationIndex + 1 + } + if (cursor === 0) { + return splitTextSegment(value) + } + children.push(...splitTextSegment(value.slice(cursor))) + return children +} + +function exactFileLink(value: string, allowSpacedRelative: boolean): ParsedTerminalFileLink | null { + const exactLink = extractTerminalFileLinks(value).find( + (link) => link.startIndex === 0 && link.endIndex === value.length + ) + if (exactLink && isLinkifiableFile(exactLink, false)) { + return exactLink + } + if (!allowSpacedRelative || !/\s/.test(value)) { + return null + } + const parsed = parseFileLinkLocation(value) + if (!parsed) { + return null + } + const hasPathShape = + ROOTED_PATH_PREFIX_PATTERN.test(parsed.pathText) || + /[\\/]/.test(parsed.pathText) || + /\.[\p{L}][\p{L}\p{N}\p{M}_+-]*$/u.test(parsed.pathText) + if (!hasPathShape) { + return null + } + const explicitLink = { + ...parsed, + startIndex: 0, + endIndex: value.length, + displayText: value + } + return isLinkifiableFile(explicitLink, false) ? explicitLink : null +} + +function splitTextNode(value: string): MarkdownNode[] { + const children: MarkdownNode[] = [] + let cursor = 0 + for (const match of value.matchAll(QUOTED_TEXT_PATTERN)) { + const content = match[1] ?? match[2] + if (!content || !exactFileLink(content, true)) { + continue + } + const matchIndex = match.index ?? 0 + const quote = match[0][0] + children.push(...splitUnquotedText(value.slice(cursor, matchIndex))) + children.push({ type: 'text', value: quote }) + children.push(createFileLinkNode(content, { type: 'text', value: content })) + children.push({ type: 'text', value: quote }) + cursor = matchIndex + match[0].length + } + if (cursor === 0) { + return splitUnquotedText(value) + } + children.push(...splitUnquotedText(value.slice(cursor))) + return children +} + +function inlineCodeFileLink(node: MarkdownNode): MarkdownNode | null { + const value = node.value?.trim() + if (!value) { + return null + } + return exactFileLink(value, true) ? createFileLinkNode(value, node) : null +} + +function transformFileLinks(node: MarkdownNode): void { + if (node.type === 'link') { + if (node.url && routeNativeChatHref(node.url).kind === 'file') { + node.url = createNativeChatFileHref(node.url) + } + return + } + if (!node.children || node.type === 'image') { + return + } + + const children: MarkdownNode[] = [] + for (const child of node.children) { + if (child.type === 'text' && child.value !== undefined) { + children.push(...splitTextNode(child.value)) + continue + } + if (child.type === 'inlineCode') { + children.push(inlineCodeFileLink(child) ?? child) + continue + } + transformFileLinks(child) + children.push(child) + } + node.children = children +} + +export function remarkNativeChatFileLinks(): (tree: MarkdownNode) => void { + return (tree) => transformFileLinks(tree) +} diff --git a/src/renderer/src/components/sidebar/folder-workspace-composer-submit.ts b/src/renderer/src/components/sidebar/folder-workspace-composer-submit.ts index 661bf623880..b6f1a1654f3 100644 --- a/src/renderer/src/components/sidebar/folder-workspace-composer-submit.ts +++ b/src/renderer/src/components/sidebar/folder-workspace-composer-submit.ts @@ -28,8 +28,9 @@ import { resolveAgentLaunchRoute } from '@/lib/agent-launch-routing' import { readLocalRuntimeCapabilities } from '@/runtime/local-runtime-capabilities' -import { startStructuredCodexLaunch } from '@/lib/structured-agent-session-launch' -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { startStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' +import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' import { useAppStore } from '@/store' import { buildFolderWorkspaceLinkedStartupPlan, @@ -232,8 +233,8 @@ export async function submitFolderWorkspaceCreate({ runtimeEnvironmentId }) let structuredLaunchAccepted = structuredLaunch - if (structuredLaunch && quickAgent === 'codex') { - const launch = startStructuredCodexLaunch(folderWorkspaceKey(workspace.id), { + if (structuredLaunch && isAgentSessionHandleProvider(quickAgent)) { + const launch = startStructuredAgentLaunch(folderWorkspaceKey(workspace.id), quickAgent, { prompt: launchDraftPrompt ?? note }) const refusalFallback = launch.claimDefinitiveRefusalFallback(async () => { diff --git a/src/renderer/src/components/sidebar/use-workspace-kanban-search.ts b/src/renderer/src/components/sidebar/use-workspace-kanban-search.ts index e782b419ab3..8c2735b5259 100644 --- a/src/renderer/src/components/sidebar/use-workspace-kanban-search.ts +++ b/src/renderer/src/components/sidebar/use-workspace-kanban-search.ts @@ -2,7 +2,10 @@ import { useCallback, useDeferredValue, useMemo, useState } from 'react' import { isWorktreePaletteQueryTooLarge } from '@/lib/worktree-palette-query-bounds' import type { Repo } from '../../../../shared/repo-types' import type { Worktree } from '../../../../shared/worktree/types' -import { matchWorkspaceBoardWorktrees } from './workspace-kanban-search' +import { + buildWorkspaceBoardPaletteDocuments, + matchWorkspaceBoardWorktrees +} from './workspace-kanban-search' function areWorktreeIdSetsEqual(a: ReadonlySet, b: ReadonlySet): boolean { if (a.size !== b.size) { @@ -48,14 +51,27 @@ export function useWorkspaceKanbanSearch(args: { // lets React interrupt the board re-render. const deferredQuery = useDeferredValue(query) + // Why split from the match memo: the index only depends on the worktrees and repos, so a + // keystroke reruns the search over an already-built index instead of rebuilding every + // worktree's normalized fields. + const documents = useMemo( + () => + buildWorkspaceBoardPaletteDocuments({ + worktrees: args.worktrees, + repoMap: args.repoMap + }), + [args.repoMap, args.worktrees] + ) + const matched = useMemo( () => matchWorkspaceBoardWorktrees({ worktrees: args.worktrees, query: deferredQuery, - repoMap: args.repoMap + repoMap: args.repoMap, + documents }), - [args.repoMap, args.worktrees, deferredQuery] + [args.repoMap, args.worktrees, deferredQuery, documents] ) const matchingWorktreeIds = diff --git a/src/renderer/src/components/sidebar/workspace-kanban-search.ts b/src/renderer/src/components/sidebar/workspace-kanban-search.ts index a47320e0bc9..1d7b88c34e1 100644 --- a/src/renderer/src/components/sidebar/workspace-kanban-search.ts +++ b/src/renderer/src/components/sidebar/workspace-kanban-search.ts @@ -1,5 +1,7 @@ import { isWorktreePaletteQueryTooLarge } from '@/lib/worktree-palette-query-bounds' -import { searchWorktrees } from '@/lib/worktree-palette-search' +import { searchWorktreeDocuments } from '@/lib/worktree-palette-search' +import { buildWorktreePaletteDocuments } from '@/lib/worktree-palette-document' +import type { PaletteDocument } from '@/lib/palette-match/palette-document' import type { Repo } from '../../../../shared/repo-types' import type { WorkspaceStatus, Worktree } from '../../../../shared/worktree/types' import { @@ -12,8 +14,25 @@ export type WorkspaceKanbanLaneView = { totalCount: number } -// Why: the board is a drag surface for named workspaces, so a card may only be -// hidden by fields the user can read on it. PR/issue/port matches are palette-only. +/** + * Builds the board's palette index once per worktree/repo identity. + * + * Why separate from the match: the index is identical across keystrokes, and building it inline + * meant normalizing and segmenting every indexed field of every worktree on every character — + * and again on every agent-status tick, which churns board identities while a query is active. + */ +export function buildWorkspaceBoardPaletteDocuments(args: { + worktrees: readonly Worktree[] + repoMap: ReadonlyMap +}): Map { + // Why the board policy (#15170): the board is a drag surface for named workspaces, so a card + // may only be hidden by text printed on it. Ports, reviews and automation runs are palette-only. + return buildWorktreePaletteDocuments(args.worktrees, { + repoMap: args.repoMap, + evidencePolicy: 'board' + }) +} + /** * Returns `null` when no filtering is active — distinct from an empty set, which * means a real query matched nothing. @@ -22,6 +41,7 @@ export function matchWorkspaceBoardWorktrees(args: { worktrees: Worktree[] query: string repoMap: Map + documents?: ReadonlyMap }): ReadonlySet | null { if (!args.query.trim()) { return null @@ -33,10 +53,14 @@ export function matchWorkspaceBoardWorktrees(args: { } const matched = new Set() - // Why the board policy (#15170): a card may only be hidden by text printed on it, so - // palette-only evidence such as ports, reviews and automation runs is excluded. - for (const result of searchWorktrees(args.worktrees, args.query, args.repoMap, { - evidencePolicy: 'board' + const documents = + args.documents ?? + buildWorkspaceBoardPaletteDocuments({ worktrees: args.worktrees, repoMap: args.repoMap }) + for (const result of searchWorktreeDocuments({ + worktrees: args.worktrees, + query: args.query, + documents, + repoMap: args.repoMap })) { if (result.matchedFields.length) { // Why (STA-4343): two hosts can publish the same id, and a board filter keyed on the diff --git a/src/renderer/src/components/sidebar/worktree-header-section-boundaries.test.ts b/src/renderer/src/components/sidebar/worktree-header-section-boundaries.test.ts new file mode 100644 index 00000000000..5b7fec3bb75 --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-header-section-boundaries.test.ts @@ -0,0 +1,95 @@ +import { describe, expect, it } from 'vitest' + +import { + getProjectGroupHeaderSectionEndByGroupId, + getRepoHeaderSectionEndByRepoId +} from './worktree-header-section-boundaries' +import type { RenderRow } from './worktree-list/listing/render-row' + +const repoHeader = (id: string): RenderRow => + ({ type: 'header', key: `repo:${id}`, label: id, count: 1, tone: '', repo: { id } }) as RenderRow +const groupHeader = (id: string): RenderRow => + ({ + type: 'header', + key: `group:${id}`, + label: id, + count: 1, + tone: '', + projectGroup: { id }, + projectGroupDepth: 0 + }) as RenderRow +const item = { type: 'item' } as RenderRow + +// Estimated starts: first header 28, later headers 32, items 116. +const rows = [repoHeader('a'), item, repoHeader('b'), item, repoHeader('c'), item] +const startOfB = 28 + 116 +const startOfC = startOfB + 32 + 116 + +describe('getRepoHeaderSectionEndByRepoId', () => { + it('ends a section at the successor from the header’s own bucket', () => { + const ends = getRepoHeaderSectionEndByRepoId({ + rows, + firstHeaderIndex: 0, + sidebarRepoHeaderIdsByBucket: new Map([ + ['group:one', ['a', 'b']], + ['group:two', ['a', 'c']] + ]), + repoHeaderBucketByRepoId: new Map([ + ['a', 'group:two'], + ['b', 'group:one'], + ['c', 'group:two'] + ]) + }) + + // Regression: a successor index keyed on id alone picked `b` from the first bucket. + expect(ends.get('a')).toBe(startOfC) + expect(ends.get('b')).toBe(startOfC) + }) + + it('falls back to the next header when a bucket has no successor', () => { + const ends = getRepoHeaderSectionEndByRepoId({ + rows, + firstHeaderIndex: 0, + sidebarRepoHeaderIdsByBucket: new Map([['ungrouped', ['a']]]), + repoHeaderBucketByRepoId: new Map([['a', 'ungrouped']]) + }) + + expect(ends.get('a')).toBe(startOfB) + }) + + it('resolves a header id that renders twice to its first row, matching findIndex', () => { + const ends = getRepoHeaderSectionEndByRepoId({ + rows: [repoHeader('a'), item, repoHeader('b'), item, repoHeader('b')], + firstHeaderIndex: 0, + sidebarRepoHeaderIdsByBucket: new Map([['ungrouped', ['a', 'b', 'b']]]), + repoHeaderBucketByRepoId: new Map([ + ['a', 'ungrouped'], + ['b', 'ungrouped'] + ]) + }) + + expect(ends.get('a')).toBe(startOfB) + // The first `b` succeeds itself; a last-wins index would jump to the second `b` row. + expect(ends.get('b')).toBe(startOfB) + }) +}) + +describe('getProjectGroupHeaderSectionEndByGroupId', () => { + it('ends a section at the successor from the group’s own bucket', () => { + const ends = getProjectGroupHeaderSectionEndByGroupId({ + rows: [groupHeader('a'), item, groupHeader('b'), item, groupHeader('c'), item], + firstHeaderIndex: 0, + sidebarProjectGroupHeaderIdsByBucket: new Map([ + ['root', ['a', 'b']], + ['parent:x', ['a', 'c']] + ]), + projectGroupHeaderBucketByGroupId: new Map([ + ['a', 'parent:x'], + ['b', 'root'], + ['c', 'parent:x'] + ]) + }) + + expect(ends.get('a')).toBe(startOfC) + }) +}) diff --git a/src/renderer/src/components/sidebar/worktree-header-section-boundaries.ts b/src/renderer/src/components/sidebar/worktree-header-section-boundaries.ts index d6b97601295..417f710d071 100644 --- a/src/renderer/src/components/sidebar/worktree-header-section-boundaries.ts +++ b/src/renderer/src/components/sidebar/worktree-header-section-boundaries.ts @@ -1,5 +1,6 @@ import { estimateRenderRowSize } from './worktree-list/viewport/virtual-rows' import type { RenderRow } from './worktree-list/listing/render-row' +import type { GroupHeaderRow } from './worktree-list/grouping/row-types' function getEstimatedRenderRowStarts( rows: readonly RenderRow[], @@ -15,18 +16,39 @@ function getEstimatedRenderRowStarts( return starts } -function findRepoHeaderRenderRowIndex(rows: readonly RenderRow[], repoId: string): number { - return rows.findIndex((row) => row.type === 'header' && row.repo?.id === repoId) +// Why indexed once instead of a findIndex per header: both boundary passes ran a full row scan +// for every header row, so the sidebar row model cost O(headers x rows) on every rebuild — and it +// rebuilds on agent-status ticks, not just on drag. First match wins, matching findIndex. +function indexHeaderRenderRows( + rows: readonly RenderRow[], + keyOf: (row: GroupHeaderRow) => string | null | undefined +): Map { + const indexByKey = new Map() + rows.forEach((row, index) => { + const key = row.type === 'header' ? keyOf(row) : undefined + if (typeof key === 'string' && !indexByKey.has(key)) { + indexByKey.set(key, index) + } + }) + return indexByKey } -function findProjectGroupHeaderRenderRowIndex(rows: readonly RenderRow[], groupId: string): number { - return rows.findIndex( - (row) => - row.type === 'header' && - !row.repo && - typeof row.projectGroup?.id === 'string' && - row.projectGroup.id === groupId - ) +// Kept per bucket: an id can sit in more than one bucket, and the caller's bucket lookup decides +// which ordering applies. First occurrence wins within a bucket, matching indexOf. +function indexBucketSuccessors( + idsByBucket: ReadonlyMap +): Map> { + const successorsByBucket = new Map>() + for (const [bucketKey, ids] of idsByBucket) { + const successorById = new Map() + ids.forEach((id, index) => { + if (!successorById.has(id)) { + successorById.set(id, ids[index + 1]) + } + }) + successorsByBucket.set(bucketKey, successorById) + } + return successorsByBucket } function findNextHeaderRenderRowIndex(rows: readonly RenderRow[], startIndex: number): number { @@ -70,6 +92,8 @@ export function getRepoHeaderSectionEndByRepoId(args: { repoHeaderBucketByRepoId: ReadonlyMap }): Map { const rowStarts = getEstimatedRenderRowStarts(args.rows, args.firstHeaderIndex) + const repoHeaderIndexByRepoId = indexHeaderRenderRows(args.rows, (row) => row.repo?.id) + const repoSuccessorsByBucket = indexBucketSuccessors(args.sidebarRepoHeaderIdsByBucket) const sectionEndByRepoId = new Map() for (let index = 0; index < args.rows.length; index++) { const row = args.rows[index] @@ -78,11 +102,9 @@ export function getRepoHeaderSectionEndByRepoId(args: { continue } const bucketKey = args.repoHeaderBucketByRepoId.get(repoId) - const bucketRepoIds = bucketKey ? args.sidebarRepoHeaderIdsByBucket.get(bucketKey) : undefined - const bucketIndex = bucketRepoIds?.indexOf(repoId) ?? -1 - const nextRepoId = bucketIndex >= 0 ? bucketRepoIds?.[bucketIndex + 1] : undefined + const nextRepoId = bucketKey ? repoSuccessorsByBucket.get(bucketKey)?.get(repoId) : undefined const endIndex = nextRepoId - ? findRepoHeaderRenderRowIndex(args.rows, nextRepoId) + ? (repoHeaderIndexByRepoId.get(nextRepoId) ?? -1) : findNextHeaderRenderRowIndex(args.rows, index + 1) sectionEndByRepoId.set( repoId, @@ -99,6 +121,10 @@ export function getProjectGroupHeaderSectionEndByGroupId(args: { projectGroupHeaderBucketByGroupId: ReadonlyMap }): Map { const rowStarts = getEstimatedRenderRowStarts(args.rows, args.firstHeaderIndex) + const projectGroupHeaderIndexByGroupId = indexHeaderRenderRows(args.rows, (row) => + row.repo ? undefined : row.projectGroup?.id + ) + const groupSuccessorsByBucket = indexBucketSuccessors(args.sidebarProjectGroupHeaderIdsByBucket) const sectionEndByGroupId = new Map() for (let index = 0; index < args.rows.length; index++) { const row = args.rows[index] @@ -114,14 +140,10 @@ export function getProjectGroupHeaderSectionEndByGroupId(args: { continue } const bucketKey = args.projectGroupHeaderBucketByGroupId.get(groupId) - const bucketGroupIds = bucketKey - ? args.sidebarProjectGroupHeaderIdsByBucket.get(bucketKey) - : undefined - const bucketIndex = bucketGroupIds?.indexOf(groupId) ?? -1 - const nextGroupId = bucketIndex >= 0 ? bucketGroupIds?.[bucketIndex + 1] : undefined + const nextGroupId = bucketKey ? groupSuccessorsByBucket.get(bucketKey)?.get(groupId) : undefined const depth = projectGroupHeader.row.projectGroupDepth ?? 0 const endIndex = nextGroupId - ? findProjectGroupHeaderRenderRowIndex(args.rows, nextGroupId) + ? (projectGroupHeaderIndexByGroupId.get(nextGroupId) ?? -1) : findProjectGroupSectionEndIndex(args.rows, index + 1, depth) sectionEndByGroupId.set( groupId, diff --git a/src/renderer/src/components/sidebar/worktree-keyboard-cycle.test.ts b/src/renderer/src/components/sidebar/worktree-keyboard-cycle.test.ts index c4c4bda6da1..430736c92be 100644 --- a/src/renderer/src/components/sidebar/worktree-keyboard-cycle.test.ts +++ b/src/renderer/src/components/sidebar/worktree-keyboard-cycle.test.ts @@ -165,8 +165,11 @@ describe('WorktreeList keyboard cycling', () => { // Why: a second buildRows call drifts from the rendered layout (host sections, // pinned placement); cycling must read the same rows the viewport renders. - expect(navigateWorktree).toContain('getCyclableWorktrees(rows, pinnedDisplayPolicy)') - expect(navigateWorktree).toContain('getWorktreeHostIdentity') + expect(navigateWorktree).toContain('getCyclableWorktreeRows(rows, pinnedDisplayPolicy)') + expect(navigateWorktree).toContain('getCyclableRowIdentity') + // Why: the active host is stored resolved while a local row is unqualified; comparing raw identities wraps to the top. + expect(navigateWorktree).toContain('resolveActiveCycleIdentity') + expect(navigateWorktree).not.toContain('composeWorktreeHostIdentity') expect(navigateWorktree).toContain('executionHostId: nextWorktree.hostId') expect(navigateWorktree).toContain('resolveCycledWorktreeId') expect(navigateWorktree).not.toContain('buildRows(') diff --git a/src/renderer/src/components/sidebar/worktree-keyboard-cycle.ts b/src/renderer/src/components/sidebar/worktree-keyboard-cycle.ts index 0a2bf23e905..4c50c78b5f5 100644 --- a/src/renderer/src/components/sidebar/worktree-keyboard-cycle.ts +++ b/src/renderer/src/components/sidebar/worktree-keyboard-cycle.ts @@ -1,9 +1,49 @@ import type { HostSectionRow } from './host-section-rows' import type { Worktree } from '../../../../shared/worktree/types' -import { getWorktreeHostIdentity } from '../../../../shared/worktree/host-qualified-identity' +import { composeWorktreeHostIdentity } from '../../../../shared/worktree/host-qualified-identity' +import { getWorktreeExecutionHostId, type ExecutionHostId } from '../../../../shared/execution-host' import type { PinnedWorktreeDisplayPolicy, WorktreeRow } from './worktree-list/grouping/row-types' import { getPreferredWorktreeRows } from './worktree-sidebar-row-preference' +/** Host-resolved identity for a cyclable row. + * + * Why resolved rather than `getWorktreeHostIdentity`: a local worktree carries no + * `hostId` (`withRepoHostOwnership` leaves it unqualified), but every activation + * path stores the host it resolved to, so raw and resolved identities never match. + */ +export function getCyclableRowIdentity(row: Pick): string { + return composeWorktreeHostIdentity( + getWorktreeExecutionHostId(row.worktree, row.repo), + row.worktree.id + ) +} + +export function getCyclableWorktreeRows( + rows: readonly HostSectionRow[], + pinnedDisplayPolicy: PinnedWorktreeDisplayPolicy +): WorktreeRow[] { + const itemRows = rows.filter((row): row is WorktreeRow => row.type === 'item') + return getPreferredWorktreeRows(itemRows, pinnedDisplayPolicy) +} + +/** Identity that locates the active workspace among the cyclable rows. */ +export function resolveActiveCycleIdentity(args: { + rows: readonly WorktreeRow[] + activeWorktreeId: string | null + activeWorkspaceExecutionHostId: ExecutionHostId | null +}): string | null { + const { rows, activeWorktreeId, activeWorkspaceExecutionHostId } = args + if (!activeWorktreeId) { + return null + } + if (activeWorkspaceExecutionHostId) { + return composeWorktreeHostIdentity(activeWorkspaceExecutionHostId, activeWorktreeId) + } + // Host-unqualified activation names no host; the row it landed on does. + const row = rows.find((candidate) => candidate.worktree.id === activeWorktreeId) + return row ? getCyclableRowIdentity(row) : null +} + /** Worktree ids in sidebar order, taken from the rows the sidebar actually * rendered, so collapsed groups and collapsed host sections drop out on their own. */ export function getCyclableWorktreeIds( @@ -12,11 +52,10 @@ export function getCyclableWorktreeIds( ): string[] { // Why item-only: folder workspaces render as their own row type and are not // activatable through activateAndRevealWorktree, so cycling has never included them. - const itemRows = rows.filter((row): row is WorktreeRow => row.type === 'item') const ids: string[] = [] const seen = new Set() - for (const row of getPreferredWorktreeRows(itemRows, pinnedDisplayPolicy)) { - const identity = getWorktreeHostIdentity(row.worktree) + for (const row of getCyclableWorktreeRows(rows, pinnedDisplayPolicy)) { + const identity = getCyclableRowIdentity(row) if (seen.has(identity)) { continue } @@ -30,8 +69,7 @@ export function getCyclableWorktrees( rows: readonly HostSectionRow[], pinnedDisplayPolicy: PinnedWorktreeDisplayPolicy ): Worktree[] { - const itemRows = rows.filter((row): row is WorktreeRow => row.type === 'item') - return getPreferredWorktreeRows(itemRows, pinnedDisplayPolicy).map((row) => row.worktree) + return getCyclableWorktreeRows(rows, pinnedDisplayPolicy).map((row) => row.worktree) } /** Pick the worktree that `worktree.navigateUp` / `worktree.navigateDown` moves diff --git a/src/renderer/src/components/sidebar/worktree-list-lineage-card-test-harness.ts b/src/renderer/src/components/sidebar/worktree-list-lineage-card-test-harness.ts index 46e44d6e836..ef47c1fac63 100644 --- a/src/renderer/src/components/sidebar/worktree-list-lineage-card-test-harness.ts +++ b/src/renderer/src/components/sidebar/worktree-list-lineage-card-test-harness.ts @@ -1,6 +1,7 @@ import React from 'react' import { renderToStaticMarkup } from 'react-dom/server' import { vi } from 'vitest' +import { ConfirmationDialogContext } from '@/components/confirmation-dialog-context' import type { Repo } from '../../../../shared/repo-types' import type { Worktree } from '../../../../shared/worktree/types' @@ -20,10 +21,14 @@ export async function loadWorktreeList(): Promise { export async function renderWorktreeListMarkup(): Promise { return renderToStaticMarkup( - React.createElement(WorktreeList!, { - scrollOffsetRef: { current: 0 }, - scrollAnchorRef: { current: null } - }) + React.createElement( + ConfirmationDialogContext.Provider, + { value: async () => false }, + React.createElement(WorktreeList!, { + scrollOffsetRef: { current: 0 }, + scrollAnchorRef: { current: null } + }) + ) ) } diff --git a/src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.host-identity.test.tsx b/src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.host-identity.test.tsx new file mode 100644 index 00000000000..efac59a9c07 --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.host-identity.test.tsx @@ -0,0 +1,122 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { Worktree } from '../../../../../../shared/worktree/types' +import type { HostSectionRow } from '../../host-section-rows' +import type { RenderRow } from '../listing/render-row' +import { getShortcutPlatform } from '@/lib/shortcut-platform' + +const activateAndRevealWorktree = vi.fn() + +vi.mock('@/lib/worktree-activation', () => ({ + activateAndRevealWorktree: (...args: unknown[]) => activateAndRevealWorktree(...args) +})) + +vi.mock('@/store', () => ({ + useAppStore: (selector: (state: { keybindings: undefined }) => unknown) => + selector({ keybindings: undefined }) +})) + +const { useWorktreeListKeyboardNavigation } = await import('./use-keyboard') + +;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true + +const repo = { id: 'repo-1', path: '/repo-1', displayName: 'Repo 1' } + +// Local worktrees carry no `hostId` — `withRepoHostOwnership` leaves them unqualified. +function localRow(id: string): HostSectionRow & { type: 'item' } { + return { + type: 'item', + rowKey: `row:${id}`, + sectionKey: 'repo:repo-1', + worktree: { id, repoId: repo.id } as unknown as Worktree, + repo: repo as never, + depth: 0, + groupDepth: 0, + lineageTrail: [], + isLastLineageChild: false, + lineageChildCount: 0 + } +} + +const rows: HostSectionRow[] = [localRow('a'), localRow('b'), localRow('c')] +const renderRows = rows as unknown as RenderRow[] + +let container: HTMLDivElement +let root: Root + +function press(direction: 'up' | 'down'): void { + const mod = getShortcutPlatform() === 'darwin' ? { metaKey: true } : { ctrlKey: true } + act(() => { + window.dispatchEvent( + new KeyboardEvent('keydown', { + key: direction === 'down' ? 'ArrowDown' : 'ArrowUp', + code: direction === 'down' ? 'ArrowDown' : 'ArrowUp', + shiftKey: true, + bubbles: true, + cancelable: true, + ...mod + }) + ) + }) +} + +function renderProbe(activeWorktreeId: string, activeHostId: 'local' | null): void { + function Probe(): null { + useWorktreeListKeyboardNavigation({ + rows, + renderRows, + activeWorktreeId, + activeWorkspaceExecutionHostId: activeHostId, + pinnedDisplayPolicy: 'single-location', + virtualizer: { scrollToIndex: () => {} } as never, + scrollRef: { current: null }, + activeModal: 'none', + markDirectScrollInput: () => {} + }) + return null + } + act(() => root.render()) +} + +beforeEach(() => { + activateAndRevealWorktree.mockClear() + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) +}) + +afterEach(() => { + act(() => root.unmount()) + container.remove() +}) + +describe('worktree keyboard cycling with a resolved active host', () => { + it('steps to the next row when the active host resolved to local but rows are unqualified', () => { + // Why: a sidebar click activates with the repo-resolved host (`local`), while + // local rows carry no hostId; a raw identity compare misses and wraps to the top. + renderProbe('b', 'local') + + press('down') + + expect(activateAndRevealWorktree).toHaveBeenCalledWith('c', {}) + }) + + it('steps to the previous row when the active host resolved to local', () => { + renderProbe('b', 'local') + + press('up') + + expect(activateAndRevealWorktree).toHaveBeenCalledWith('a', {}) + }) + + it('still steps normally when the active host is unqualified', () => { + renderProbe('b', null) + + press('down') + + expect(activateAndRevealWorktree).toHaveBeenCalledWith('c', {}) + }) +}) diff --git a/src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.ts b/src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.ts index 5746e16bf2a..c8d3ec5d80d 100644 --- a/src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.ts +++ b/src/renderer/src/components/sidebar/worktree-list/navigation/use-keyboard.ts @@ -4,16 +4,17 @@ import type { Virtualizer } from '@tanstack/react-virtual' import { useAppStore } from '@/store' import { activateAndRevealWorktree } from '@/lib/worktree-activation' import type { ExecutionHostId } from '../../../../../../shared/execution-host' -import { - composeWorktreeHostIdentity, - getWorktreeHostIdentity -} from '../../../../../../shared/worktree/host-qualified-identity' import { getShortcutPlatform } from '@/lib/shortcut-platform' import { keybindingMatchesAction } from '../../../../../../shared/keybindings' import type { HostSectionRow } from '../../host-section-rows' import type { PinnedWorktreeDisplayPolicy } from '../grouping/row-types' import type { RenderRow } from '../listing/render-row' -import { getCyclableWorktrees, resolveCycledWorktreeId } from '../../worktree-keyboard-cycle' +import { + getCyclableRowIdentity, + getCyclableWorktreeRows, + resolveActiveCycleIdentity, + resolveCycledWorktreeId +} from '../../worktree-keyboard-cycle' import { findPreferredRenderRowIndexForWorktreeIdentity } from './render-row-lookup' function isEditableTarget(target: EventTarget | null): boolean { @@ -65,24 +66,22 @@ export function useWorktreeListKeyboardNavigation(args: { // Why: cycle over the rows the sidebar actually rendered — collapsing a group // means "not now", and a rebuilt near-copy would drift from what is on screen // (host sections, pinned placement, folder workspaces). - const worktrees = getCyclableWorktrees(rows, pinnedDisplayPolicy) - const worktreeIdentities = worktrees.map(getWorktreeHostIdentity) + const worktreeRows = getCyclableWorktreeRows(rows, pinnedDisplayPolicy) const nextWorktreeIdentity = resolveCycledWorktreeId({ - worktreeIds: worktreeIdentities, - activeWorktreeId: activeWorktreeId - ? composeWorktreeHostIdentity( - activeWorkspaceExecutionHostId ?? undefined, - activeWorktreeId - ) - : null, + worktreeIds: worktreeRows.map(getCyclableRowIdentity), + activeWorktreeId: resolveActiveCycleIdentity({ + rows: worktreeRows, + activeWorktreeId, + activeWorkspaceExecutionHostId + }), direction }) if (nextWorktreeIdentity === null) { return } - const nextWorktree = worktrees.find( - (worktree) => getWorktreeHostIdentity(worktree) === nextWorktreeIdentity - ) + const nextWorktree = worktreeRows.find( + (row) => getCyclableRowIdentity(row) === nextWorktreeIdentity + )?.worktree if (!nextWorktree) { return } diff --git a/src/renderer/src/components/sidebar/worktree-list/navigation/use-reveal-requests.test.tsx b/src/renderer/src/components/sidebar/worktree-list/navigation/use-reveal-requests.test.tsx new file mode 100644 index 00000000000..7f78d3d011a --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-list/navigation/use-reveal-requests.test.tsx @@ -0,0 +1,191 @@ +// @vitest-environment happy-dom +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { ConfirmationDialogProvider } from '@/components/confirmation-dialog' +import { + requestScrollToCurrentWorkspaceReveal, + requestScrollToCurrentWorkspaceRevealAndRename +} from '@/lib/scroll-to-current-workspace-status' +import { folderWorkspaceKey } from '../../../../../../shared/workspace-scope' +import { useSidebarRevealRequests } from './use-reveal-requests' +import type { Worktree } from '../../../../../../shared/worktree/types' + +globalThis.IS_REACT_ACT_ENVIRONMENT = true + +const state = vi.hoisted(() => ({ + setGroupBy: vi.fn(), + pendingRevealSidebarRow: null, + revealSidebarRow: vi.fn(), + revealWorktreeInSidebar: vi.fn(), + setContextualToursBlockingSurfaceVisible: vi.fn() +})) +vi.mock('@/store', () => ({ + useAppStore: (selector: (value: typeof state) => unknown) => selector(state) +})) + +type Args = Parameters[0] +function Host({ args }: { args: Args }): null { + useSidebarRevealRequests(args) + return null +} + +let root: Root +let container: HTMLDivElement +let args: Args + +async function render(): Promise { + await act(async () => { + root.render( + + + + ) + }) +} + +async function click(label: string): Promise { + const button = Array.from(document.querySelectorAll('button')).find( + (candidate) => candidate.textContent === label + ) + expect(button).toBeDefined() + await act(async () => button!.click()) +} + +beforeEach(() => { + vi.clearAllMocks() + container = document.createElement('div') + document.body.append(container) + root = createRoot(container) + const worktree: Worktree = { + id: 'wt-1', + hostId: 'ssh:dev', + repoId: 'repo-1', + path: '/repo/feature', + displayName: 'Feature', + branch: 'feature', + head: 'abc123', + isBare: false, + isMainWorktree: false, + comment: '', + linkedIssue: null, + linkedPR: null, + linkedLinearIssue: null, + linkedGitLabMR: null, + linkedGitLabIssue: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 1, + lastActivityAt: 1 + } + args = { + groupBy: 'repo', + renderedSidebarRowKeys: new Set(), + renderedWorktreeIdentities: [], + currentSidebarWorktreeId: worktree.id, + currentSidebarExecutionHostId: 'ssh:dev', + worktreeMap: new Map([[worktree.id, worktree]]), + worktrees: [worktree], + folderWorkspaces: [], + hasFilters: true, + clearFilters: vi.fn() + } +}) + +afterEach(async () => { + await act(async () => root.unmount()) + container.remove() +}) + +describe('revealing a filtered workspace', () => { + it('explains the filter reset and leaves filters intact when dismissed', async () => { + await render() + await act(async () => requestScrollToCurrentWorkspaceReveal()) + expect(document.body.textContent).toContain('Revealing it will clear your sidebar filters.') + expect(args.clearFilters).not.toHaveBeenCalled() + expect(state.revealWorktreeInSidebar).not.toHaveBeenCalled() + await click('Keep filters') + expect(args.clearFilters).not.toHaveBeenCalled() + expect(state.revealWorktreeInSidebar).not.toHaveBeenCalled() + }) + + it('clears filters and reveals on the original execution host only after confirmation', async () => { + await render() + await act(async () => { + requestScrollToCurrentWorkspaceReveal() + requestScrollToCurrentWorkspaceReveal() + }) + await click('Clear filters and reveal') + expect(args.clearFilters).toHaveBeenCalledTimes(1) + expect(state.revealWorktreeInSidebar).toHaveBeenCalledWith('wt-1', { + behavior: 'smooth', + highlight: true, + beginRename: false, + executionHostId: 'ssh:dev' + }) + expect(document.querySelector('[role="dialog"]')).toBeNull() + }) + + it.each([true, false])( + 'reveals immediately when clearing filters is unnecessary (%s)', + async (visible) => { + args = { + ...args, + hasFilters: visible, + renderedWorktreeIdentities: visible ? ['ssh:dev|wt-1'] : [] + } + await render() + await act(async () => requestScrollToCurrentWorkspaceReveal()) + expect(document.querySelector('[role="dialog"]')).toBeNull() + expect(args.clearFilters).not.toHaveBeenCalled() + expect(state.revealWorktreeInSidebar).toHaveBeenCalledTimes(1) + } + ) + + it('does not apply a stale confirmation after switching workspaces', async () => { + await render() + await act(async () => requestScrollToCurrentWorkspaceReveal()) + args = { ...args, currentSidebarWorktreeId: 'wt-2' } + await render() + await click('Clear filters and reveal') + expect(args.clearFilters).not.toHaveBeenCalled() + expect(state.revealWorktreeInSidebar).not.toHaveBeenCalled() + }) + + it('confirms filtered folder workspaces and preserves the rename request', async () => { + args = { + ...args, + currentSidebarWorktreeId: folderWorkspaceKey('folder-1'), + currentSidebarExecutionHostId: null, + folderWorkspaces: [ + { + id: 'folder-1', + projectGroupId: 'project-1', + name: 'Notes', + folderPath: '/notes', + linkedTask: null, + comment: '', + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 1, + lastActivityAt: 1, + createdAt: 1, + updatedAt: 1 + } + ] + } + await render() + await act(async () => requestScrollToCurrentWorkspaceRevealAndRename()) + expect(args.clearFilters).not.toHaveBeenCalled() + await click('Clear filters and reveal') + expect(args.clearFilters).toHaveBeenCalledTimes(1) + expect(state.revealWorktreeInSidebar).toHaveBeenCalledWith(folderWorkspaceKey('folder-1'), { + behavior: 'smooth', + highlight: true, + beginRename: true, + executionHostId: undefined + }) + }) +}) diff --git a/src/renderer/src/components/sidebar/worktree-list/navigation/use-reveal-requests.ts b/src/renderer/src/components/sidebar/worktree-list/navigation/use-reveal-requests.ts index bbe9d57e0cb..27a8ff5c3fa 100644 --- a/src/renderer/src/components/sidebar/worktree-list/navigation/use-reveal-requests.ts +++ b/src/renderer/src/components/sidebar/worktree-list/navigation/use-reveal-requests.ts @@ -1,4 +1,7 @@ -import { useCallback, useEffect } from 'react' +import { useCallback, useEffect, useLayoutEffect, useRef } from 'react' +import { Crosshair } from 'lucide-react' +import { useConfirmationDialog } from '@/components/confirmation-dialog-context' +import { translate } from '@/i18n/i18n' import { useAppStore } from '@/store' import { SCROLL_TO_CURRENT_WORKSPACE_REVEAL_REQUEST_EVENT, @@ -41,6 +44,12 @@ export function useSidebarRevealRequests(args: { const pendingRevealSidebarRow = useAppStore((s) => s.pendingRevealSidebarRow) const revealSidebarRow = useAppStore((s) => s.revealSidebarRow) const revealWorktreeInSidebar = useAppStore((s) => s.revealWorktreeInSidebar) + const confirm = useConfirmationDialog() + const confirmationPending = useRef(false) + const latestArgs = useRef(args) + useLayoutEffect(() => { + latestArgs.current = args + }) useEffect(() => { if (!pendingRevealSidebarRow) { @@ -68,7 +77,7 @@ export function useSidebarRevealRequests(args: { ]) const handleRevealCurrentWorkspaceRequest = useCallback( - (event: Event) => { + async (event: Event) => { const detail = event instanceof CustomEvent ? (event.detail as ScrollToCurrentWorkspaceRevealRequestDetail | undefined) @@ -101,9 +110,40 @@ export function useSidebarRevealRequests(args: { currentSidebarExecutionHostId ?? undefined, currentSidebarWorktreeId ) - if (!renderedWorktreeIdentities.includes(currentIdentity)) { - // Why: the reveal action must show the current workspace, so relax filters that hide it first. - clearFilters() + if (hasFilters && !renderedWorktreeIdentities.includes(currentIdentity)) { + if (confirmationPending.current) { + return + } + confirmationPending.current = true + let confirmed: boolean + try { + confirmed = await confirm({ + icon: Crosshair, + initialFocus: 'confirm', + cancelVariant: 'ghost', + title: translate('sidebar.revealFiltered.title', 'Reveal hidden workspace?'), + description: translate( + 'sidebar.revealFiltered.description', + 'The active workspace is hidden in the sidebar. Revealing it will clear your sidebar filters.' + ), + confirmLabel: translate('sidebar.revealFiltered.confirm', 'Clear filters and reveal'), + cancelLabel: translate('sidebar.revealFiltered.cancel', 'Keep filters') + }) + } finally { + confirmationPending.current = false + } + const latest = latestArgs.current + // A workspace switch while the dialog is open must not clear filters for a stale target. + if ( + !confirmed || + latest.currentSidebarWorktreeId !== currentSidebarWorktreeId || + latest.currentSidebarExecutionHostId !== currentSidebarExecutionHostId + ) { + return + } + if (latest.hasFilters && !latest.renderedWorktreeIdentities.includes(currentIdentity)) { + latest.clearFilters() + } } revealWorktreeInSidebar(currentSidebarWorktreeId, { behavior: 'smooth', @@ -113,7 +153,8 @@ export function useSidebarRevealRequests(args: { }) }, [ - clearFilters, + confirm, + hasFilters, currentSidebarWorktreeId, currentSidebarExecutionHostId, folderWorkspaces, diff --git a/src/renderer/src/components/sidebar/worktree-list/rows/RepoScanUnavailableIndicator.test.tsx b/src/renderer/src/components/sidebar/worktree-list/rows/RepoScanUnavailableIndicator.test.tsx new file mode 100644 index 00000000000..5ce0c7a950b --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-list/rows/RepoScanUnavailableIndicator.test.tsx @@ -0,0 +1,102 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { TooltipProvider } from '@/components/ui/tooltip' +import { useAppStore } from '@/store' +import type { Repo } from '../../../../../../shared/repo-types' +import { makeDetectedResult } from '@/store/slices/worktrees-detected-listing-fixtures' +import { RepoScanUnavailableIndicator } from './RepoScanUnavailableIndicator' + +const repo = { + id: 'repo-1', + path: 'C:\\repo', + displayName: 'repo', + badgeColor: '#000', + addedAt: 0 +} as Repo + +const initialState = useAppStore.getInitialState() +const roots: Root[] = [] + +async function render(): Promise { + const container = document.createElement('div') + document.body.appendChild(container) + const root = createRoot(container) + roots.push(root) + await act(async () => { + root.render( + + + + ) + }) + return container +} + +describe('RepoScanUnavailableIndicator', () => { + beforeEach(() => { + globalThis.IS_REACT_ACT_ENVIRONMENT = true + useAppStore.setState(initialState, true) + }) + + afterEach(async () => { + for (const root of roots.splice(0)) { + await act(async () => root.unmount()) + } + document.body.innerHTML = '' + useAppStore.setState(initialState, true) + }) + + it('renders nothing for an authoritative listing', async () => { + useAppStore.setState({ + detectedWorktreesByRepo: { [repo.id]: makeDetectedResult(repo.id, []) } + }) + + const container = await render() + + expect(container.querySelector('button')).toBeNull() + }) + + // Why: a non-authoritative listing without a reason is the disconnected-SSH shape, which the + // host header already explains; this marker is only for a scan that failed with a cause. + it('renders nothing for a non-authoritative listing that carries no reason', async () => { + useAppStore.setState({ + detectedWorktreesByRepo: { + [repo.id]: makeDetectedResult(repo.id, [], { + authoritative: false, + source: 'metadata-fallback' + }) + } + }) + + const container = await render() + + expect(container.querySelector('button')).toBeNull() + }) + + it('marks a failed scan and re-runs it on click', async () => { + const fetchWorktrees = vi.fn(async () => true) + useAppStore.setState({ + fetchWorktrees: fetchWorktrees as never, + detectedWorktreesByRepo: { + [repo.id]: makeDetectedResult(repo.id, [], { + authoritative: false, + source: 'metadata-fallback', + unavailableReason: 'wsl.exe host failure (distro "kali-linux"): WSL_E_DISTRO_NOT_FOUND' + }) + } + }) + + const container = await render() + const button = container.querySelector('button') + + expect(button?.getAttribute('aria-label')).toContain('Worktree scan failed for repo') + expect(button?.className).toContain('text-destructive') + await act(async () => { + button?.click() + }) + expect(fetchWorktrees).toHaveBeenCalledWith(repo.id, { executionHostId: 'local' }) + }) +}) diff --git a/src/renderer/src/components/sidebar/worktree-list/rows/RepoScanUnavailableIndicator.tsx b/src/renderer/src/components/sidebar/worktree-list/rows/RepoScanUnavailableIndicator.tsx new file mode 100644 index 00000000000..a956598140e --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-list/rows/RepoScanUnavailableIndicator.tsx @@ -0,0 +1,75 @@ +import React from 'react' +import { TriangleAlert } from 'lucide-react' +import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' +import { cn } from '@/lib/utils' +import { translate } from '@/i18n/i18n' +import { useAppStore } from '@/store' +import type { Repo } from '../../../../../../shared/repo-types' +import { getRepoExecutionHostId } from '../../../../../../shared/execution-host' +import { + handleRepoHeaderActionPointerDown, + stopRepoHeaderKeyboardToggle +} from './header-event-guards' + +/** + * Marks a repo whose worktree scan failed, so its rows are retained but cannot be trusted. + * Click re-runs the scan: the failure is otherwise re-tried only by the next incidental refresh. + */ +export function RepoScanUnavailableIndicator({ repo }: { repo: Repo }): React.JSX.Element | null { + const detected = useAppStore((s) => s.detectedWorktreesByRepo[repo.id]) + const fetchWorktrees = useAppStore((s) => s.fetchWorktrees) + const [pending, setPending] = React.useState(false) + if (!detected || detected.authoritative || !detected.unavailableReason) { + return null + } + const title = translate( + 'auto.components.sidebar.RepoScanUnavailableIndicator.title', + 'Worktree scan failed for {{value0}}', + { value0: repo.displayName } + ) + const retryLabel = translate( + 'auto.components.sidebar.RepoScanUnavailableIndicator.retry', + 'Retry scan' + ) + return ( + + + + + +
+
{title}
+
{detected.unavailableReason}
+
+ {translate( + 'auto.components.sidebar.RepoScanUnavailableIndicator.retained', + 'Existing worktrees are kept until a scan succeeds. Click to retry.' + )} +
+
+
+
+ ) +} diff --git a/src/renderer/src/components/sidebar/worktree-list/rows/SectionHeader.tsx b/src/renderer/src/components/sidebar/worktree-list/rows/SectionHeader.tsx index b4ed149712a..db0c7ec1371 100644 --- a/src/renderer/src/components/sidebar/worktree-list/rows/SectionHeader.tsx +++ b/src/renderer/src/components/sidebar/worktree-list/rows/SectionHeader.tsx @@ -26,6 +26,7 @@ import { WORKTREE_SECTION_HEADER_PADDING_LEFT } from './indentation' import { FolderPathStatusIndicator } from './FolderPathStatusIndicator' +import { RepoScanUnavailableIndicator } from './RepoScanUnavailableIndicator' import { ProjectGroupCreateWorkspaceButton, ProjectGroupHeaderMenu @@ -334,6 +335,7 @@ export function renderWorktreeSectionHeaderRow(args: {
+ {isRepoHeader ? : null} diff --git a/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx b/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx index 85d978e3bf6..6d5b523c2af 100644 --- a/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx +++ b/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx @@ -8,6 +8,7 @@ import { useAgentDetectionTargetForWorktree } from '@/hooks/useAgentDetectionTar import { useDetectedAgents } from '@/hooks/useDetectedAgents' import { useOptionalShortcutLabel } from '@/hooks/useShortcutLabel' import { launchAgentInNewTab } from '@/lib/launch-agent-in-new-tab' +import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' import type { TuiAgent } from '../../../../shared/tui-agent' import type { LaunchSource } from '../../../../shared/telemetry-events' import { @@ -15,7 +16,7 @@ import { filterEnabledTuiAgents } from '../../../../shared/tui-agent-selection' import { translate } from '@/i18n/i18n' -import { useStructuredCodexLaunchStatus } from '@/lib/structured-agent-session-launch' +import { useStructuredAgentLaunchStatus } from '@/lib/structured-agent-session-launch' export type QuickLaunchAgentMenuItemsProps = { worktreeId: string @@ -117,7 +118,12 @@ function QuickLaunchAgentMenuItemsInner({ const openSettingsPage = useAppStore((s) => s.openSettingsPage) const openSettingsTarget = useAppStore((s) => s.openSettingsTarget) const newAgentShortcut = useOptionalShortcutLabel('tab.newAgent') - const structuredCodexLaunchStatus = useStructuredCodexLaunchStatus(worktreeId) + // One hook per structured provider: the launch registry is keyed by agent, and hooks cannot run + // inside the agent list's render loop. + const structuredLaunchStatusByAgent = { + claude: useStructuredAgentLaunchStatus(worktreeId, 'claude'), + codex: useStructuredAgentLaunchStatus(worktreeId, 'codex') + } const openAgentSettings = useCallback(() => { openSettingsTarget({ pane: 'agents', repoId: null }) @@ -199,26 +205,33 @@ function QuickLaunchAgentMenuItemsInner({ {agents.map((agent) => { const entry = getCatalogEntry(agent) const label = entry?.label ?? agent - const isStructuredCodexPending = - agent === 'codex' && structuredCodexLaunchStatus === 'pending' - const menuLabel = isStructuredCodexPending ? 'Starting Codex chat…' : label + const isStructuredLaunchPending = + isAgentSessionHandleProvider(agent) && structuredLaunchStatusByAgent[agent] === 'pending' + const pendingLabel = translate( + 'components.native-chat.structuredSessionLaunchPending', + 'Starting {{value0}} chat…', + { value0: label } + ) + const menuLabel = isStructuredLaunchPending ? pendingLabel : label const showsDefaultAgentShortcut = newAgentShortcut !== null && defaultAgent !== 'blank' && agent === defaultAgent return ( runLaunch(agent)} className="gap-2 rounded-[7px] px-2 py-1.5 text-[12px] leading-5 font-medium" - title={translate( - 'auto.components.tab.bar.QuickLaunchButton.ec2adf093e', - isStructuredCodexPending - ? 'Starting Codex chat…' - : 'Launch {{value0}} in a new terminal', - isStructuredCodexPending ? undefined : { value0: label } - )} + title={ + isStructuredLaunchPending + ? pendingLabel + : translate( + 'auto.components.tab.bar.QuickLaunchButton.ec2adf093e', + 'Launch {{value0}} in a new terminal', + { value0: label } + ) + } > - {isStructuredCodexPending ? ( + {isStructuredLaunchPending ? ( ))} diff --git a/src/renderer/src/components/tab-bar/TerminalTabSplitMenuSection.tsx b/src/renderer/src/components/tab-bar/TerminalTabSplitMenuSection.tsx index 1b620e35dbe..a2226036f27 100644 --- a/src/renderer/src/components/tab-bar/TerminalTabSplitMenuSection.tsx +++ b/src/renderer/src/components/tab-bar/TerminalTabSplitMenuSection.tsx @@ -21,6 +21,7 @@ export function TerminalTabSplitMenuSection({ onActivate, splitRightShortcut, splitDownShortcut, + showTerminalSplit = true, trailingSeparator = false }: { unifiedTabId: string @@ -30,6 +31,7 @@ export function TerminalTabSplitMenuSection({ onActivate: (tabId: string) => void splitRightShortcut: string splitDownShortcut: string + showTerminalSplit?: boolean trailingSeparator?: boolean }): React.JSX.Element { const splitActiveTerminalPane = (direction: 'vertical' | 'horizontal'): void => { @@ -42,33 +44,37 @@ export function TerminalTabSplitMenuSection({ return ( <> - - - - {translate( - 'auto.components.tab.bar.TerminalTabSplitMenuSection.splitTerminal', - 'Split terminal' - )} - - - splitActiveTerminalPane('vertical')}> - + {showTerminalSplit ? ( + + + {translate( - 'auto.components.tab.bar.SortableTabContextMenu.splitTerminalRight', - 'Split terminal right' + 'auto.components.tab.bar.TerminalTabSplitMenuSection.splitTerminal', + 'Split terminal' )} - {splitRightShortcut} - - splitActiveTerminalPane('horizontal')}> - - {translate( - 'auto.components.tab.bar.SortableTabContextMenu.splitTerminalDown', - 'Split terminal down' - )} - {splitDownShortcut} - - - + + + splitActiveTerminalPane('vertical')}> + + {translate( + 'auto.components.tab.bar.SortableTabContextMenu.splitTerminalRight', + 'Split terminal right' + )} + {splitRightShortcut} + + splitActiveTerminalPane('horizontal')}> + + {translate( + 'auto.components.tab.bar.SortableTabContextMenu.splitTerminalDown', + 'Split terminal down' + )} + {splitDownShortcut} + + + + ) : null} {trailingSeparator ? : null} ) diff --git a/src/renderer/src/components/tab-bar/tab-bar-item-surface.tsx b/src/renderer/src/components/tab-bar/tab-bar-item-surface.tsx index 24114f42a16..95e9a98e7bb 100644 --- a/src/renderer/src/components/tab-bar/tab-bar-item-surface.tsx +++ b/src/renderer/src/components/tab-bar/tab-bar-item-surface.tsx @@ -266,6 +266,7 @@ export function renderTabBarItems({ onSetTabColor={onSetTabColor} onTogglePin={() => togglePinned(item)} onToggleExpand={() => {}} + canSplitTerminal={false} dragData={dragData} dropIndicator={dropIndicatorByVisibleId.get(item.id) ?? null} includeTopTabBorder={includeTopTabBorder} diff --git a/src/renderer/src/components/tab-group/useTabGroupItemProjections.ts b/src/renderer/src/components/tab-group/useTabGroupItemProjections.ts index 921176604b0..782af51595a 100644 --- a/src/renderer/src/components/tab-group/useTabGroupItemProjections.ts +++ b/src/renderer/src/components/tab-group/useTabGroupItemProjections.ts @@ -50,6 +50,18 @@ export function useTabGroupItemProjections({ () => new Map(worktreeState.terminalTabs.map((item) => [item.id, item])), [worktreeState.terminalTabs] ) + // Why indexed like the terminal tabs above: `openFiles` is the global list across every + // worktree and `tabOrder` is as long as the group, so the per-tab `.find` scans below were + // quadratic in tab count on a path that reruns whenever any unified tab is written. + const openFileById = useMemo( + () => new Map(worktreeState.openFiles.map((item) => [item.id, item])), + [worktreeState.openFiles] + ) + const browserTabById = useMemo( + () => new Map(worktreeState.browserTabs.map((item) => [item.id, item])), + [worktreeState.browserTabs] + ) + const groupTabById = useMemo(() => new Map(groupTabs.map((item) => [item.id, item])), [groupTabs]) const terminalTabs = useMemo( () => @@ -100,11 +112,11 @@ export function useTabGroupItemProjections({ item.contentType === 'check-details' ) .map((item) => { - const file = worktreeState.openFiles.find((candidate) => candidate.id === item.entityId) + const file = openFileById.get(item.entityId) return file ? { ...file, tabId: item.id } : null }) .filter((item): item is GroupEditorItem => item !== null), - [groupTabs, worktreeState.openFiles] + [groupTabs, openFileById] ) const browserItems = useMemo( @@ -112,11 +124,11 @@ export function useTabGroupItemProjections({ groupTabs .filter((item) => item.contentType === 'browser') .map((item) => { - const bt = worktreeState.browserTabs.find((candidate) => candidate.id === item.entityId) + const bt = browserTabById.get(item.entityId) return bt ? { ...bt, tabId: item.id } : null }) .filter((item): item is GroupBrowserItem => item !== null), - [groupTabs, worktreeState.browserTabs] + [browserTabById, groupTabs] ) const agentSessionItems = useMemo( @@ -130,7 +142,7 @@ export function useTabGroupItemProjections({ const tabBarOrder = useMemo( () => (group?.tabOrder ?? []).map((itemId) => { - const item = groupTabs.find((candidate) => candidate.id === itemId) + const item = groupTabById.get(itemId) if (!item) { return itemId } @@ -138,7 +150,7 @@ export function useTabGroupItemProjections({ ? item.entityId : item.id }), - [group, groupTabs] + [group, groupTabById] ) return { diff --git a/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.structured-session.test.ts b/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.structured-session.test.ts index 20791dd3dde..d1feb923255 100644 --- a/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.structured-session.test.ts +++ b/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.structured-session.test.ts @@ -3,7 +3,7 @@ import type * as ReactModule from 'react' const mocks = vi.hoisted(() => ({ callRuntimeRpc: vi.fn(), - cancelStructuredCodexLaunch: vi.fn(), + cancelStructuredAgentLaunch: vi.fn(), closeBrowserTab: vi.fn(), closeFile: vi.fn(), closeStructuredAgentSession: vi.fn(), @@ -73,7 +73,7 @@ vi.mock('@/runtime/structured-agent-session-close', () => ({ })) vi.mock('@/lib/structured-agent-session-launch', () => ({ - cancelStructuredCodexLaunch: mocks.cancelStructuredCodexLaunch + cancelStructuredAgentLaunch: mocks.cancelStructuredAgentLaunch })) vi.mock('@/runtime/runtime-worktree-selector', () => ({ @@ -129,7 +129,7 @@ describe('structured agent-session close ordering', () => { closeItem(AGENT_TAB.id) await vi.waitFor(() => expect(order).toEqual(['agent-close', 'tab-close', 'local-remove'])) - expect(mocks.cancelStructuredCodexLaunch).toHaveBeenCalledWith('wt-1', 'session-1') + expect(mocks.cancelStructuredAgentLaunch).toHaveBeenCalledWith('wt-1', 'session-1') }) it('keeps the tab available when owner disposal fails, so close can be retried', async () => { @@ -154,7 +154,7 @@ describe('structured agent-session close ordering', () => { closeMany([AGENT_TAB.id]) - expect(mocks.cancelStructuredCodexLaunch).toHaveBeenCalledWith('wt-1', 'session-1') + expect(mocks.cancelStructuredAgentLaunch).toHaveBeenCalledWith('wt-1', 'session-1') await vi.waitFor(() => expect(mocks.closeUnifiedTab).toHaveBeenCalledWith(AGENT_TAB.id)) }) }) diff --git a/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.ts b/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.ts index 0c8c71d6d7c..a5a9a65455d 100644 --- a/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.ts +++ b/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.ts @@ -10,7 +10,7 @@ import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner import { closeBrowserWorkspaceTabOnHosts } from '@/runtime/browser-workspace-tab-close' import { callRuntimeRpc, getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client' import { closeStructuredAgentSession } from '@/runtime/structured-agent-session-close' -import { cancelStructuredCodexLaunch } from '@/lib/structured-agent-session-launch' +import { cancelStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' import { toRuntimeWorktreeSelector } from '@/runtime/runtime-worktree-selector' import { translate } from '@/i18n/i18n' @@ -18,7 +18,7 @@ function reportStructuredSessionCloseError(error: unknown): void { toast.error( translate( 'components.native-chat.structuredSessionCloseFailed', - 'Could not close this Codex chat' + 'Could not close this chat session' ), { description: error instanceof Error ? error.message : String(error) } ) @@ -121,12 +121,12 @@ export function useTabGroupTabCloseCommands({ worktreeId ) if (item.contentType === 'agent-session') { - cancelStructuredCodexLaunch(worktreeId, item.entityId) + cancelStructuredAgentLaunch(worktreeId, item.entityId) // Why: the structured session lives on the host, so the local tab close must also // retire the host's canonical row or it reappears on the next sync. // Cancel a still-reconciling create before closing its owner; otherwise a missing // post-create snapshot is mistaken for an unknown outcome and retried after close. - cancelStructuredCodexLaunch(worktreeId, item.entityId) + cancelStructuredAgentLaunch(worktreeId, item.entityId) const target = getActiveRuntimeTarget({ activeRuntimeEnvironmentId: runtimeEnvironmentId }) @@ -198,7 +198,7 @@ export function useTabGroupTabCloseCommands({ worktreeId ) if (item.contentType === 'agent-session') { - cancelStructuredCodexLaunch(worktreeId, item.entityId) + cancelStructuredAgentLaunch(worktreeId, item.entityId) const target = getActiveRuntimeTarget({ activeRuntimeEnvironmentId: runtimeEnvironmentId }) diff --git a/src/renderer/src/components/terminal-pane/StructuredAgentSessionTerminalReturnButton.tsx b/src/renderer/src/components/terminal-pane/StructuredAgentSessionTerminalReturnButton.tsx new file mode 100644 index 00000000000..bba08bd914c --- /dev/null +++ b/src/renderer/src/components/terminal-pane/StructuredAgentSessionTerminalReturnButton.tsx @@ -0,0 +1,25 @@ +import { Button } from '@/components/ui/button' +import { translate } from '@/i18n/i18n' + +export function StructuredAgentSessionTerminalReturnButton(props: { + enabled: boolean + onReturn?: () => void +}): React.JSX.Element | null { + if (!props.enabled) { + return null + } + return ( + + ) +} diff --git a/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx b/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx index 1b5e94b4a11..da2a3f02133 100644 --- a/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx @@ -41,6 +41,31 @@ export function TerminalPaneNativeChatPortal({ return null } + const contextMenuActions = { + onSplitRight: () => contextMenu.runForPane(chatPane.id, contextMenu.onSplitRight), + onSplitDown: () => contextMenu.runForPane(chatPane.id, contextMenu.onSplitDown), + canEqualizePaneSizes: managedPanes.length > 1 && expandedPaneId === null, + onEqualizePaneSizes: () => contextMenu.runForPane(chatPane.id, contextMenu.onEqualizePaneSizes), + canExpandPane: managedPanes.length > 1, + isPaneExpanded: expandedPaneId === chatPane.id, + onToggleExpand: () => contextMenu.runForPane(chatPane.id, contextMenu.onToggleExpand), + canContinueAgentSessionInNewSession: canContinueAgentSessionInNewSession( + resolveAgentForLeaf(chatPane.leafId) + ), + onContinueAgentSessionInNewSession: () => + contextMenu.runForPane(chatPane.id, contextMenu.onContinueAgentSessionInNewSession), + onForkAgentSession: () => + void contextMenu.runForPane(chatPane.id, contextMenu.onForkAgentSession), + onSetTitle: () => contextMenu.runForPane(chatPane.id, contextMenu.onSetTitle), + onCopyTerminalId: () => void contextMenu.runForPane(chatPane.id, contextMenu.onCopyTerminalId), + onCopyPaneId: () => void contextMenu.runForPane(chatPane.id, contextMenu.onCopyPaneId), + canCopyAgentSessionId: chatPaneSessionId !== null, + onCopyAgentSessionId: () => + void contextMenu.runForPane(chatPane.id, contextMenu.onCopyAgentSessionId), + canClosePane: managedPanes.length > 1, + onClosePane: () => contextMenu.runForPane(chatPane.id, contextMenu.onClosePane) + } + return createPortal(
{structuredSessionId && structuredChatAgent ? ( @@ -51,7 +76,7 @@ export function TerminalPaneNativeChatPortal({ agent={structuredChatAgent} isVisible={isRendererVisible} target={structuredChatTarget} - allowFileUriLinks + contextMenuActions={contextMenuActions} orchestrationDispatchStatus={chatPaneDispatchStatus} /> ) : ( @@ -65,32 +90,7 @@ export function TerminalPaneNativeChatPortal({ ownsTabWideLaunchDraft={chatPaneOwnsTabWideLaunchDraft} onSwitchToTerminal={switchNativeChatToTerminal} readTerminalScreen={readNativeChatTerminalScreen} - contextMenuActions={{ - onSplitRight: () => contextMenu.runForPane(chatPane.id, contextMenu.onSplitRight), - onSplitDown: () => contextMenu.runForPane(chatPane.id, contextMenu.onSplitDown), - canEqualizePaneSizes: managedPanes.length > 1 && expandedPaneId === null, - onEqualizePaneSizes: () => - contextMenu.runForPane(chatPane.id, contextMenu.onEqualizePaneSizes), - canExpandPane: managedPanes.length > 1, - isPaneExpanded: expandedPaneId === chatPane.id, - onToggleExpand: () => contextMenu.runForPane(chatPane.id, contextMenu.onToggleExpand), - canContinueAgentSessionInNewSession: canContinueAgentSessionInNewSession( - resolveAgentForLeaf(chatPane.leafId) - ), - onContinueAgentSessionInNewSession: () => - contextMenu.runForPane(chatPane.id, contextMenu.onContinueAgentSessionInNewSession), - onForkAgentSession: () => - void contextMenu.runForPane(chatPane.id, contextMenu.onForkAgentSession), - onSetTitle: () => contextMenu.runForPane(chatPane.id, contextMenu.onSetTitle), - onCopyTerminalId: () => - void contextMenu.runForPane(chatPane.id, contextMenu.onCopyTerminalId), - onCopyPaneId: () => void contextMenu.runForPane(chatPane.id, contextMenu.onCopyPaneId), - canCopyAgentSessionId: chatPaneSessionId !== null, - onCopyAgentSessionId: () => - void contextMenu.runForPane(chatPane.id, contextMenu.onCopyAgentSessionId), - canClosePane: managedPanes.length > 1, - onClosePane: () => contextMenu.runForPane(chatPane.id, contextMenu.onClosePane) - }} + contextMenuActions={contextMenuActions} orchestrationDispatchStatus={chatPaneDispatchStatus} /> )} diff --git a/src/renderer/src/components/terminal-pane/agent-completion-coordinator-types.ts b/src/renderer/src/components/terminal-pane/agent-completion-coordinator-types.ts index 7d7c53f6bb8..7bf7e90f7f9 100644 --- a/src/renderer/src/components/terminal-pane/agent-completion-coordinator-types.ts +++ b/src/renderer/src/components/terminal-pane/agent-completion-coordinator-types.ts @@ -32,7 +32,7 @@ export type AgentCompletionCoordinatorOptions = { inspectProcess: ( settings: Pick | null | undefined, ptyId: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; steadyState?: boolean } ) => Promise dispatchCompletion: (title: string, meta?: AgentCompletionDispatchMeta) => void dispatchAttention?: (title: string, meta: AgentAttentionDispatchMeta) => void diff --git a/src/renderer/src/components/terminal-pane/agent-completion-poll-interval.test.ts b/src/renderer/src/components/terminal-pane/agent-completion-poll-interval.test.ts new file mode 100644 index 00000000000..fffbb94de90 --- /dev/null +++ b/src/renderer/src/components/terminal-pane/agent-completion-poll-interval.test.ts @@ -0,0 +1,72 @@ +import { describe, expect, it } from 'vitest' +import { PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS } from '../../../../shared/process-table-snapshot-reader' +import { POLL_TIER_INTERVAL_MS } from './agent-completion-poll-cadence' +import { nextCadenceInspectionDelayMs } from './agent-completion-poll-interval' + +const IDLE_MS = POLL_TIER_INTERVAL_MS.idle + +describe('nextCadenceInspectionDelayMs', () => { + const alignedDelay = (now: number): number => + nextCadenceInspectionDelayMs({ + baseMs: IDLE_MS, + hasConsecutiveErrors: false, + alignToSharedGrid: true, + now + }) + + it('walks panes that scheduled at different moments onto one shared deadline', () => { + // Why this matters: the inspection queue collapses shared-observation tasks enqueued in the + // same tick onto one process-table capture, so a shared deadline is one `ps` for all panes. + const clocks = [0, 137, 999, 1_501].map((offset) => 1_700_000_000_000 + offset) + // Each pane may only be pulled forward by the snapshot TTL per step, so convergence takes + // at most IDLE_MS / TTL steps. + for (let step = 0; step < IDLE_MS / PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS; step += 1) { + for (let pane = 0; pane < clocks.length; pane += 1) { + clocks[pane] += alignedDelay(clocks[pane]!) + } + } + + expect(new Set(clocks).size).toBe(1) + }) + + it('never waits longer than the tier interval, nor more than the snapshot TTL less', () => { + for (let offset = 0; offset < IDLE_MS * 3; offset += 1) { + const delay = alignedDelay(1_700_000_000_000 + offset) + expect(delay).toBeGreaterThanOrEqual(IDLE_MS - PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS) + expect(delay).toBeLessThanOrEqual(IDLE_MS) + } + }) + + it('keeps jitter while backing off, so a failing host is not retried by every pane at once', () => { + const lowJitter = nextCadenceInspectionDelayMs({ + baseMs: IDLE_MS, + hasConsecutiveErrors: true, + alignToSharedGrid: true, + now: 1_700_000_000_000, + random: () => 0 + }) + const highJitter = nextCadenceInspectionDelayMs({ + baseMs: IDLE_MS, + hasConsecutiveErrors: true, + alignToSharedGrid: true, + now: 1_700_000_000_000, + random: () => 1 + }) + + expect(lowJitter).toBe(Math.round(IDLE_MS * 0.9)) + expect(highJitter).toBe(Math.round(IDLE_MS * 1.1)) + }) + + it('degrades safely on a non-positive interval', () => { + for (const baseMs of [0, -1, Number.NaN, Number.POSITIVE_INFINITY]) { + expect( + nextCadenceInspectionDelayMs({ + baseMs, + hasConsecutiveErrors: false, + alignToSharedGrid: true, + now: 1_700_000_000_000 + }) + ).toBe(0) + } + }) +}) diff --git a/src/renderer/src/components/terminal-pane/agent-completion-poll-interval.ts b/src/renderer/src/components/terminal-pane/agent-completion-poll-interval.ts new file mode 100644 index 00000000000..0de8632f552 --- /dev/null +++ b/src/renderer/src/components/terminal-pane/agent-completion-poll-interval.ts @@ -0,0 +1,43 @@ +// Why not the sibling reader that re-exports this: it imports `node:child_process`, which the +// renderer cannot load — reaching it blanks the window at module evaluation. +import { PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS } from '../../../../shared/process-table-snapshot' + +/** + * Picks the delay until a pane's next cadence inspection. + * + * Local panes all resolve out of one TTL-deduped process-table snapshot, and the inspection + * queue collapses every shared-observation task enqueued in the same tick onto a single host + * capture. Independent per-pane jitter defeated that: panes drifted apart, each landing in its + * own tick and forking its own `ps`. Snapping to a grid anchored at the epoch puts same-tier + * panes back in one tick, so N panes cost one capture instead of N. + * + * The pull-forward is clamped to the process-table snapshot TTL, so a pane never polls more than + * that early and never later than its tier interval. A pane off the grid therefore walks onto it + * in at most `baseMs / TTL` steps, costing at most one extra inspection in total, and no + * inspection is ever delayed. + * + * Alignment is scoped to genuinely idle panes: no foreground agent and no pane activity inside + * the hot window. A pane that just produced output keeps its exact interval, so the bounded + * post-activity cadence is unchanged, and the error-backoff path keeps its jitter — spreading + * retries across panes is the point when a host has just failed. + */ + +export function nextCadenceInspectionDelayMs(args: { + baseMs: number + hasConsecutiveErrors: boolean + alignToSharedGrid: boolean + now: number + random?: () => number +}): number { + const { alignToSharedGrid, baseMs, hasConsecutiveErrors, now } = args + if (!Number.isFinite(baseMs) || baseMs <= 0) { + return 0 + } + if (hasConsecutiveErrors || !alignToSharedGrid) { + const random = args.random ?? Math.random + return Math.round(baseMs * (1 + (random() * 0.2 - 0.1))) + } + const deadline = Math.floor((now + baseMs) / baseMs) * baseMs + const earliest = Math.max(1, baseMs - PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS) + return Math.min(baseMs, Math.max(earliest, deadline - now)) +} diff --git a/src/renderer/src/components/terminal-pane/agent-completion-poll-scheduler.ts b/src/renderer/src/components/terminal-pane/agent-completion-poll-scheduler.ts index 8894edbd9c6..30a7b2e817e 100644 --- a/src/renderer/src/components/terminal-pane/agent-completion-poll-scheduler.ts +++ b/src/renderer/src/components/terminal-pane/agent-completion-poll-scheduler.ts @@ -7,6 +7,7 @@ import { POLL_TIER_INTERVAL_MS, type PollCadenceTier } from './agent-completion-poll-cadence' +import { nextCadenceInspectionDelayMs } from './agent-completion-poll-interval' export function createAgentCompletionPollScheduler(args: { options: AgentCompletionCoordinatorOptions @@ -88,7 +89,19 @@ export function createAgentCompletionPollScheduler(args: { state.consecutiveInspectionErrors > 0 ? Math.min(Math.max(10_000, base), base * 2 ** state.consecutiveInspectionErrors) : base - const interval = Math.round(backoff * (1 + (Math.random() * 0.2 - 0.1))) + const now = Date.now() + // Only genuinely idle panes share a deadline: a pane with a foreground agent, or one still + // inside the post-activity hot window, keeps its exact interval and its own phase. + const isIdlePane = + state.lastForegroundAgent === null && + (state.lastPaneActivityAt === null || + now - state.lastPaneActivityAt >= NO_EVIDENCE_ACTIVITY_HOT_WINDOW_MS) + const interval = nextCadenceInspectionDelayMs({ + baseMs: backoff, + hasConsecutiveErrors: state.consecutiveInspectionErrors > 0, + alignToSharedGrid: isIdlePane, + now + }) state.pollTimerTier = tier state.pollTimer = setTimeout(() => { state.pollTimer = null diff --git a/src/renderer/src/components/terminal-pane/agent-completion-process-monitor.ts b/src/renderer/src/components/terminal-pane/agent-completion-process-monitor.ts index f34ceffa97e..ca2456aedbb 100644 --- a/src/renderer/src/components/terminal-pane/agent-completion-process-monitor.ts +++ b/src/renderer/src/components/terminal-pane/agent-completion-process-monitor.ts @@ -108,10 +108,18 @@ export function createAgentCompletionProcessMonitor({ let inspectedRecognizedAgent = false let inspectionSucceeded = false try { - const result = await (expectedIncarnationIdAtRequest - ? options.inspectProcess(options.getSettings(), ptyId, { - expectedIncarnationId: expectedIncarnationIdAtRequest - }) + // Only a cadence tick on a local pane reads nothing but the name; every other read + // (pending-title, remote) needs the full capture and must not ask for the cheap one. + const inspectOptions = { + ...(expectedIncarnationIdAtRequest + ? { expectedIncarnationId: expectedIncarnationIdAtRequest } + : {}), + ...(priority === 'cadence' && options.isRemotePtyId?.(ptyId) !== true + ? { steadyState: true } + : {}) + } + const result = await (Object.keys(inspectOptions).length > 0 + ? options.inspectProcess(options.getSettings(), ptyId, inspectOptions) : options.inspectProcess(options.getSettings(), ptyId)) if ( !state.disposed && diff --git a/src/renderer/src/components/terminal-pane/agent-completion-stale-evidence-backoff.test.ts b/src/renderer/src/components/terminal-pane/agent-completion-stale-evidence-backoff.test.ts new file mode 100644 index 00000000000..82714edce6e --- /dev/null +++ b/src/renderer/src/components/terminal-pane/agent-completion-stale-evidence-backoff.test.ts @@ -0,0 +1,136 @@ +// The second consumer of the capture budget, at the point where a user feels it. +// +// A whole-machine `ps` costs seconds on a large or loaded host: 2.5-9.0s on an idle 2,002-process +// laptop, 4.0-18.6s at load 46. Publishing that as a truthful-but-late `live` record does not help +// this pane. `admitRemoteForegroundEvidence` refuses it, every refusal increments +// `consecutiveInspectionErrors`, and the poll scheduler's backoff then stretches the cadence to its +// 10s floor -- so agent-completion detection degrades on exactly the hosts where a capture is +// slowest, which are the hosts where agents take longest to finish. +// +// Giving up on the capture and publishing a prompt `unverifiable` instead costs one poll and +// nothing else: the record is admitted, so no error is counted. +import { describe, expect, it, vi } from 'vitest' +import { handleAgentCompletionInspectionResult } from './agent-completion-inspection-result' +import type { RemoteInspectionState } from './agent-completion-inspection-result' +import type { ProcessMonitorState } from './agent-completion-process-types' +import type { AgentCompletionCoordinatorOptions } from './agent-completion-coordinator-types' +import type { RuntimeTerminalProcessInspection } from '@/runtime/runtime-terminal-inspection' +import { toAppSshPtyId } from '../../../../shared/ssh-pty-id' +import { REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS } from '../../../../shared/remote-foreground-evidence-admission' + +const SSH_PTY_ID = toAppSshPtyId('target-1', 'pty-1') +const INCARNATION = 'inc-1' + +function liveRecord(capturedAgeMs: number): RuntimeTerminalProcessInspection { + return { + foregroundProcess: 'claude', + hasChildProcesses: true, + foregroundProcessEvidence: { + verdict: 'live', + processName: 'claude', + authorityGeneration: 'gen-1', + observationEpoch: 1, + capturedAgeMs, + ptyId: 'pty-1', + ptyIncarnationId: INCARNATION, + fence: { + platform: 'posix', + shellPid: 10, + shellStartTime: '100', + tty: '/dev/pts/2', + foregroundPgid: 11, + process: { pid: 11, startTime: '101' } + } + } + } +} + +/** What both relay call sites publish when the capture misses its budget. */ +function unreadableTableRecord(): RuntimeTerminalProcessInspection { + return { + foregroundProcess: 'claude', + hasChildProcesses: true, + foregroundProcessEvidence: { + verdict: 'unverifiable', + reason: 'process_table_unreadable', + authorityGeneration: 'gen-1', + observationEpoch: 1, + capturedAgeMs: 0, + ptyId: 'pty-1', + ptyIncarnationId: INCARNATION + } + } +} + +function inspect(result: RuntimeTerminalProcessInspection, roundTripMs = 20): ProcessMonitorState { + const state: ProcessMonitorState = { + disposed: false, + inspectionInFlight: false, + inspectionGeneration: 0, + consecutiveInspectionErrors: 0, + pollTrackingStarted: true, + pollTimer: null, + pollTimerTier: null, + lastPaneActivityAt: null, + hasAgentRunEvidence: false, + pendingProcessExitAgent: null, + lastForegroundAgent: null, + processSession: 1 + } + const remoteInspection: RemoteInspectionState = { + authorityGeneration: null, + observationEpoch: -1, + bindingKey: null, + knownAuthorityGenerations: new Set() + } + const started = performance.now() + vi.spyOn(performance, 'now').mockReturnValue(started + roundTripMs) + handleAgentCompletionInspectionResult({ + result, + requestStartedAtMonotonic: started, + options: { + paneKey: 'tab-1:leaf-1', + getPtyId: () => SSH_PTY_ID, + getSettings: () => null, + isRemotePtyId: () => true, + getExpectedIncarnationId: () => INCARNATION + } as unknown as AgentCompletionCoordinatorOptions, + state, + identityScope: {} as never, + clearAgentRunEvidence: vi.fn(), + hasPendingHookDone: () => false, + hasPendingCodexAttention: () => false, + scheduleNextPoll: vi.fn(), + handleRecognizedProcess: vi.fn(), + dispatchCompletion: vi.fn(), + remoteInspection + }) + vi.restoreAllMocks() + return state +} + +describe('agent completion polling under a host capture it cannot use', () => { + it('counts no error for the prompt unverifiable a capture over budget produces', () => { + expect(inspect(unreadableTableRecord()).consecutiveInspectionErrors).toBe(0) + }) + + it('counts an error for the late live record the same capture would have produced', () => { + // One of the measured captures. Every poll refusing this way is what drives the cadence to + // its 10s backoff floor and stops completion detection for the pane. + expect(inspect(liveRecord(6_140)).consecutiveInspectionErrors).toBe(1) + }) + + it.each([ + ['at the ceiling', REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS, 0], + ['one step past it', REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS + 1, 1] + ])('counts %s as %s errors', (_label, capturedAgeMs, errors) => { + expect(inspect(liveRecord(capturedAgeMs), 0).consecutiveInspectionErrors).toBe(errors) + }) + + it('admits a capture that lands inside the evidence budget', () => { + // The reason the budget is 1,200ms rather than lower: a capture inside it must still clear the + // 2,000ms ceiling once its duration is counted once instead of twice. Under the double count + // this same record was refused, because 1,200 + a 1,300ms round trip read as 2,500. + expect(inspect(liveRecord(1_200), 1_300).consecutiveInspectionErrors).toBe(0) + }) +}) diff --git a/src/renderer/src/components/terminal-pane/agent-completion-steady-state-opt-in.test.ts b/src/renderer/src/components/terminal-pane/agent-completion-steady-state-opt-in.test.ts new file mode 100644 index 00000000000..4f2286c7eb7 --- /dev/null +++ b/src/renderer/src/components/terminal-pane/agent-completion-steady-state-opt-in.test.ts @@ -0,0 +1,84 @@ +import { describe, expect, it, vi } from 'vitest' +import { createAgentCompletionCoordinator } from './agent-completion-coordinator' +import { + flushAsyncTicks, + processResult, + useAgentCompletionCoordinatorLifecycle +} from './agent-completion-coordinator-test-harness' + +// The renderer opts a read into the cheap tier ONLY when it is a self-correcting cadence poll on a +// local pane. Pending-title reads decide a completion once and remote reads consume evidence, so +// neither may ask for a capture that omits evidence. +describe('agent completion steadyState opt-in', () => { + useAgentCompletionCoordinatorLifecycle() + + const optionsOf = (call: unknown[]): unknown => call[2] + + it('marks cadence polls on a local pane as steadyState', async () => { + const inspectProcess = vi.fn(async () => processResult('codex')) + const coordinator = createAgentCompletionCoordinator({ + paneKey: 'tab-1:leaf-1', + getPtyId: () => 'pty-1', + getSettings: () => null, + inspectProcess, + dispatchCompletion: vi.fn(), + isLive: () => true, + shouldPollProcessCadence: () => true + }) + coordinator.startProcessTracking() + coordinator.observeTitle('Codex working') + vi.advanceTimersByTime(3_000) + await flushAsyncTicks() + expect(inspectProcess).toHaveBeenCalled() + for (const call of inspectProcess.mock.calls) { + expect(optionsOf(call as unknown[])).toEqual({ steadyState: true }) + } + coordinator.dispose() + }) + + it('a pending-title read on a local pane is NOT steadyState: it decides a completion once', async () => { + const inspectProcess = vi.fn(async () => processResult('codex')) + const coordinator = createAgentCompletionCoordinator({ + paneKey: 'tab-1:leaf-1', + getPtyId: () => 'pty-1', + getSettings: () => null, + inspectProcess, + dispatchCompletion: vi.fn(), + isLive: () => true, + shouldPollProcessCadence: () => false + }) + coordinator.observeTitle('Codex working') + coordinator.observeTitle('/tmp/orca-e2e-repo') + await flushAsyncTicks() + expect(inspectProcess).toHaveBeenCalled() + for (const call of inspectProcess.mock.calls) { + expect(optionsOf(call as unknown[])).toBeUndefined() + } + coordinator.dispose() + }) + + it('never marks a remote pane as steadyState: remote identity needs evidence', async () => { + const inspectProcess = vi.fn(async () => processResult('codex')) + const coordinator = createAgentCompletionCoordinator({ + paneKey: 'tab-1:leaf-1', + getPtyId: () => 'remote:pty-1', + isRemotePtyId: () => true, + getExpectedIncarnationId: () => 'inc-1', + getSettings: () => null, + inspectProcess, + dispatchCompletion: vi.fn(), + isLive: () => true, + shouldPollProcessCadence: () => true + }) + coordinator.startProcessTracking() + coordinator.observeTitle('Codex working') + coordinator.observeTitle('/tmp/orca-e2e-repo') + vi.advanceTimersByTime(3_000) + await flushAsyncTicks() + expect(inspectProcess).toHaveBeenCalled() + for (const call of inspectProcess.mock.calls) { + expect(optionsOf(call as unknown[])).toEqual({ expectedIncarnationId: 'inc-1' }) + } + coordinator.dispose() + }) +}) diff --git a/src/renderer/src/components/terminal/terminal-tab-actions.ts b/src/renderer/src/components/terminal/terminal-tab-actions.ts index 49bc3d11fef..92eb85b49e5 100644 --- a/src/renderer/src/components/terminal/terminal-tab-actions.ts +++ b/src/renderer/src/components/terminal/terminal-tab-actions.ts @@ -157,7 +157,7 @@ export function closeTerminalTab( toast.error( translate( 'components.native-chat.structuredSessionCloseFailed', - 'Could not close this Codex chat' + 'Could not close this chat session' ), { description: translate( diff --git a/src/renderer/src/components/use-worktree-jump-palette-quick-actions.ts b/src/renderer/src/components/use-worktree-jump-palette-quick-actions.ts index 0783a7e5d23..32cddf92a17 100644 --- a/src/renderer/src/components/use-worktree-jump-palette-quick-actions.ts +++ b/src/renderer/src/components/use-worktree-jump-palette-quick-actions.ts @@ -56,6 +56,7 @@ export function useWorktreeJumpPaletteQuickActions({ sshConnectionStates, activeGroupIdByWorktree, groupsByWorktree, + unifiedTabsByWorktree, isLoading, settings, runtimeStatusByEnvironmentId, @@ -129,6 +130,7 @@ export function useWorktreeJumpPaletteQuickActions({ void sshConnectionStates void activeGroupIdByWorktree void groupsByWorktree + void unifiedTabsByWorktree void isLoading void settings?.activeRuntimeEnvironmentId void runtimeStatusByEnvironmentId @@ -144,6 +146,7 @@ export function useWorktreeJumpPaletteQuickActions({ sshConnectionStates, activeGroupIdByWorktree, groupsByWorktree, + unifiedTabsByWorktree, isLoading, settings?.activeRuntimeEnvironmentId, runtimeStatusByEnvironmentId diff --git a/src/renderer/src/hooks/composer-state/composer-submit-orchestration.ts b/src/renderer/src/hooks/composer-state/composer-submit-orchestration.ts index e968e499afa..707b9916c5c 100644 --- a/src/renderer/src/hooks/composer-state/composer-submit-orchestration.ts +++ b/src/renderer/src/hooks/composer-state/composer-submit-orchestration.ts @@ -140,27 +140,14 @@ export function useComposerSubmitOrchestration( workspaceSeedName: target.derivedComposerState.workspaceSeedName }) const multipleCreateReset = useMultipleCreateReset({ + handleClearSmartNameSelection: source.issueSourceActions.handleClearSmartNameSelection, lastAutoNameRef: target.asyncComposerState.lastAutoNameRef, nameInputRef: target.asyncComposerState.nameInputRef, setAgentPrompt: target.sourceContextState.setAgentPrompt, setAttachmentPaths: target.sourceContextState.setAttachmentPaths, - setBranchNameOverride: target.workspaceIdentityState.setBranchNameOverride, - setBranchNameOverridePreservesNameEdits: - target.workspaceIdentityState.setBranchNameOverridePreservesNameEdits, - setCompareBaseRef: target.workspaceIdentityState.setCompareBaseRef, setCreateError: target.asyncComposerState.setCreateError, - setForkPushWarning: target.workspaceIdentityState.setForkPushWarning, - setLinkedGitLabIssue: target.workspaceIdentityState.setLinkedGitLabIssue, - setLinkedGitLabMR: target.workspaceIdentityState.setLinkedGitLabMR, - setLinkedIssue: target.workspaceIdentityState.setLinkedIssue, - setLinkedPR: target.workspaceIdentityState.setLinkedPR, - setLinkedTaskSourceContext: target.sourceContextState.setLinkedTaskSourceContext, - setLinkedWorkItem: target.sourceContextState.setLinkedWorkItem, setName: target.sourceContextState.setName, - setNote: target.sourceContextState.setNote, - setPushTarget: target.workspaceIdentityState.setPushTarget, - setReuseSelectedBranch: target.workspaceIdentityState.setReuseSelectedBranch, - setStartFromResetHint: target.workspaceIdentityState.setStartFromResetHint + setNote: target.sourceContextState.setNote }) const quickSubmitSourcePreparation = useQuickSubmitSourcePreparation({ baseBranch: target.workspaceIdentityState.baseBranch, diff --git a/src/renderer/src/hooks/composer-state/full-creation-structured-launch.test.ts b/src/renderer/src/hooks/composer-state/full-creation-structured-launch.test.ts index 2176a1863ed..77c94652a52 100644 --- a/src/renderer/src/hooks/composer-state/full-creation-structured-launch.test.ts +++ b/src/renderer/src/hooks/composer-state/full-creation-structured-launch.test.ts @@ -1,23 +1,23 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' const mocks = vi.hoisted(() => ({ - startStructuredCodexLaunch: vi.fn(), + startStructuredAgentLaunch: vi.fn(), activateStructuredAgentSessionById: vi.fn() })) vi.mock('@/lib/structured-agent-session-launch', () => ({ - startStructuredCodexLaunch: mocks.startStructuredCodexLaunch + startStructuredAgentLaunch: mocks.startStructuredAgentLaunch })) vi.mock('@/lib/structured-agent-session-tab-activation', () => ({ activateStructuredAgentSessionById: mocks.activateStructuredAgentSessionById })) -vi.mock('@/lib/launch-structured-codex-session', () => ({ +vi.mock('@/lib/launch-structured-agent-session', () => ({ StructuredAgentSessionCreateRefusalError: class extends Error {} })) -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' import { settleFullCreationStructuredLaunch } from './full-creation-structured-launch' describe('settleFullCreationStructuredLaunch', () => { @@ -26,7 +26,7 @@ describe('settleFullCreationStructuredLaunch', () => { it('runs the legacy terminal fallback after a definitive refusal', async () => { const fallbackActivation = { primaryTabId: 'fallback-tab' } const onDefinitiveRefusal = vi.fn().mockResolvedValue(fallbackActivation) - mocks.startStructuredCodexLaunch.mockReturnValue({ + mocks.startStructuredAgentLaunch.mockReturnValue({ launchResult: Promise.reject(new StructuredAgentSessionCreateRefusalError('unsupported')), isVisibilityUnknown: () => false, claimDefinitiveRefusalFallback: (fallback: () => Promise) => @@ -54,7 +54,7 @@ describe('settleFullCreationStructuredLaunch', () => { it('reports an unknown outcome without starting a fallback terminal', async () => { const onDefinitiveRefusal = vi.fn() - mocks.startStructuredCodexLaunch.mockReturnValue({ + mocks.startStructuredAgentLaunch.mockReturnValue({ launchResult: Promise.reject(new Error('connection lost')), isVisibilityUnknown: () => true, claimDefinitiveRefusalFallback: vi.fn(() => Promise.resolve(false)) diff --git a/src/renderer/src/hooks/composer-state/full-creation-structured-launch.ts b/src/renderer/src/hooks/composer-state/full-creation-structured-launch.ts index f352724d841..90ecb7c6722 100644 --- a/src/renderer/src/hooks/composer-state/full-creation-structured-launch.ts +++ b/src/renderer/src/hooks/composer-state/full-creation-structured-launch.ts @@ -1,7 +1,8 @@ +import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' import type { TuiAgent } from '../../../../shared/tui-agent' import type { ActivateAndRevealResult } from '@/lib/worktree-activation' -import { startStructuredCodexLaunch } from '@/lib/structured-agent-session-launch' -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { startStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' import { activateStructuredAgentSessionById } from '@/lib/structured-agent-session-tab-activation' type Activation = ActivateAndRevealResult | false @@ -20,11 +21,11 @@ export async function settleFullCreationStructuredLaunch(args: { }> { let activation = args.initialActivation let structuredLaunchAccepted = args.structuredLaunch - if (!args.structuredLaunch || args.agent !== 'codex') { + if (!args.structuredLaunch || !isAgentSessionHandleProvider(args.agent)) { return { structuredLaunchAccepted, visibilityUnknown: false, activation } } - const launch = startStructuredCodexLaunch(args.worktreeId, { prompt: args.prompt }) + const launch = startStructuredAgentLaunch(args.worktreeId, args.agent, { prompt: args.prompt }) const refusalFallback = launch.claimDefinitiveRefusalFallback(async () => { structuredLaunchAccepted = false activation = await args.onDefinitiveRefusal() diff --git a/src/renderer/src/hooks/composer-state/multiple-create-reset.test.ts b/src/renderer/src/hooks/composer-state/multiple-create-reset.test.ts new file mode 100644 index 00000000000..910143cace8 --- /dev/null +++ b/src/renderer/src/hooks/composer-state/multiple-create-reset.test.ts @@ -0,0 +1,168 @@ +// @vitest-environment happy-dom + +import { useRef, useState } from 'react' +import { act, renderHook } from '@testing-library/react' +import { describe, expect, it, vi } from 'vitest' +import type { LinkedWorkItemSummary } from '@/lib/new-workspace' +import { useIssueSourceActions } from './issue-source-actions' +import { useMultipleCreateReset } from './multiple-create-reset' +import type { SmartGitHubPrStartPointSelection } from './source-selection-decisions' + +const sources: LinkedWorkItemSummary[] = [ + { + provider: 'github', + type: 'pr', + number: 42, + title: 'Fix checkout', + url: 'https://github.com/acme/app/pull/42' + }, + { + provider: 'github', + type: 'issue', + number: 43, + title: 'Fix checkout', + url: 'https://github.com/acme/app/issues/43' + }, + { + provider: 'gitlab', + type: 'mr', + number: 44, + title: 'Fix checkout', + url: 'https://gitlab.com/acme/app/-/merge_requests/44' + } +] + +function useSelectedSourceReset( + initialItem: LinkedWorkItemSummary | null, + isProjectGroupTarget = false, + initialBaseBranch: string | undefined = '1234567890abcdef1234567890abcdef12345678' +) { + const [linkedWorkItem, setLinkedWorkItem] = useState(initialItem) + const [baseBranch, setBaseBranch] = useState(initialBaseBranch) + const [name, setName] = useState('fix-checkout') + const [note, setNote] = useState('User note') + const lastAutoNameRef = useRef(name) + const branchAutoNameRef = useRef('fix-checkout') + const lastAutoNoteRef = useRef('Generated note') + const smartGitHubPrStartPointSelectionRef = useRef( + initialItem?.provider === 'github' && initialItem.type === 'pr' + ? { + repoId: 'repo-1', + item: { + ...initialItem, + type: 'pr', + id: 'pr-42', + repoId: 'repo-1', + state: 'open', + labels: [], + updatedAt: '2026-09-01T00:00:00Z', + author: null + } + } + : null + ) + const source = useIssueSourceActions({ + baseBranch, + branchAutoNameRef, + isProjectGroupTarget, + lastAutoNameRef, + lastAutoNoteRef, + linkedWorkItem, + name, + noteRef: useRef(note), + setBaseBranch, + setBranchNameOverride: vi.fn(), + setBranchNameOverridePreservesNameEdits: vi.fn(), + setCompareBaseRef: vi.fn(), + setForkPushWarning: vi.fn(), + setLinkedGitLabIssue: vi.fn(), + setLinkedGitLabMR: vi.fn(), + setLinkedIssue: vi.fn(), + setLinkedPR: vi.fn(), + setLinkedTaskSourceContext: vi.fn(), + setLinkedWorkItem, + setName, + setNote, + setPushTarget: vi.fn(), + setReuseEligibleBranch: vi.fn(), + setReuseSelectedBranch: vi.fn(), + setStartFromResetHint: vi.fn(), + smartGitHubPrStartPointSelectionRef + }) + const reset = useMultipleCreateReset({ + handleClearSmartNameSelection: source.handleClearSmartNameSelection, + lastAutoNameRef, + nameInputRef: useRef(null), + setAgentPrompt: vi.fn(), + setAttachmentPaths: vi.fn(), + setCreateError: vi.fn(), + setName, + setNote + }) + return { + ...reset, + selection: source.smartNameSelection, + linkedWorkItem, + baseBranch, + name, + note, + branchAutoNameRef, + smartGitHubPrStartPointSelectionRef + } +} + +describe('create more source reset', () => { + it.each(sources)( + 'clears $provider $type and its checkout source before the next create', + (item) => { + const { result } = renderHook(() => useSelectedSourceReset(item)) + expect(result.current.selection?.label).toContain('Fix checkout') + + if (item.provider === 'github' && item.type === 'pr') { + expect(result.current.smartGitHubPrStartPointSelectionRef.current).not.toBeNull() + } + + act(() => result.current.resetForNextCreate()) + + expect(result.current.smartGitHubPrStartPointSelectionRef.current).toBeNull() + expect(result.current.selection).toBeNull() + expect(result.current.linkedWorkItem).toBeNull() + expect(result.current.baseBranch).toBeUndefined() + expect(result.current.name).toBe('') + expect(result.current.note).toBe('') + expect(result.current.branchAutoNameRef.current).toBe('') + } + ) + + it.each(['linear', 'jira'] as const)('clears a %s task on a folder target', (provider) => { + const item: LinkedWorkItemSummary = { + provider, + type: 'issue', + number: 0, + title: 'Fix checkout', + url: + provider === 'linear' + ? 'https://linear.app/acme/issue/APP-45' + : 'https://acme.atlassian.net/browse/APP-45' + } + const { result } = renderHook(() => useSelectedSourceReset(item, true)) + expect(result.current.selection?.kind).toBe(provider) + + act(() => result.current.resetForNextCreate()) + + expect(result.current.selection).toBeNull() + expect(result.current.linkedWorkItem).toBeNull() + expect(result.current.name).toBe('') + expect(result.current.note).toBe('') + }) + + it('clears a plain branch selection before the next create', () => { + const { result } = renderHook(() => useSelectedSourceReset(null, false, 'feature/checkout')) + expect(result.current.selection).toEqual({ kind: 'branch', label: 'feature/checkout' }) + + act(() => result.current.resetForNextCreate()) + + expect(result.current.selection).toBeNull() + expect(result.current.baseBranch).toBeUndefined() + }) +}) diff --git a/src/renderer/src/hooks/composer-state/multiple-create-reset.ts b/src/renderer/src/hooks/composer-state/multiple-create-reset.ts index b1b77720934..e078d692717 100644 --- a/src/renderer/src/hooks/composer-state/multiple-create-reset.ts +++ b/src/renderer/src/hooks/composer-state/multiple-create-reset.ts @@ -2,96 +2,48 @@ import type { ComposerModel } from './composer-model' type MultipleCreateResetInput = Pick< ComposerModel, + | 'handleClearSmartNameSelection' | 'lastAutoNameRef' | 'nameInputRef' | 'setAgentPrompt' | 'setAttachmentPaths' - | 'setBranchNameOverride' - | 'setBranchNameOverridePreservesNameEdits' - | 'setCompareBaseRef' | 'setCreateError' - | 'setForkPushWarning' - | 'setLinkedGitLabIssue' - | 'setLinkedGitLabMR' - | 'setLinkedIssue' - | 'setLinkedPR' - | 'setLinkedTaskSourceContext' - | 'setLinkedWorkItem' | 'setName' | 'setNote' - | 'setPushTarget' - | 'setReuseSelectedBranch' - | 'setStartFromResetHint' > import { useCallback } from 'react' export function useMultipleCreateReset(input: MultipleCreateResetInput) { const { + handleClearSmartNameSelection, lastAutoNameRef, nameInputRef, setAgentPrompt, setAttachmentPaths, - setBranchNameOverride, - setBranchNameOverridePreservesNameEdits, - setCompareBaseRef, setCreateError, - setForkPushWarning, - setLinkedGitLabIssue, - setLinkedGitLabMR, - setLinkedIssue, - setLinkedPR, - setLinkedTaskSourceContext, - setLinkedWorkItem, setName, - setNote, - setPushTarget, - setReuseSelectedBranch, - setStartFromResetHint + setNote } = input const resetForNextCreate = useCallback(() => { - // Why: clear identity fields derived from a PR pick while retaining repo, base, agent, and group context for sequential creates. + // Clear the checkout source too, so a PR's resolved SHA cannot become the next selection. + handleClearSmartNameSelection() setName('') lastAutoNameRef.current = '' setAgentPrompt('') setNote('') setAttachmentPaths([]) - setLinkedWorkItem(null) - setLinkedTaskSourceContext(null) - setLinkedIssue('') - setLinkedPR(null) - setLinkedGitLabIssue(null) - setLinkedGitLabMR(null) - setBranchNameOverride(undefined) - setBranchNameOverridePreservesNameEdits(false) - setCompareBaseRef(undefined) - setPushTarget(undefined) - setReuseSelectedBranch(false) - setStartFromResetHint(null) - setForkPushWarning(null) setCreateError(null) requestAnimationFrame(() => nameInputRef.current?.focus()) }, [ + handleClearSmartNameSelection, lastAutoNameRef, nameInputRef, setAgentPrompt, setAttachmentPaths, - setBranchNameOverride, - setBranchNameOverridePreservesNameEdits, - setCompareBaseRef, setCreateError, - setForkPushWarning, - setLinkedGitLabIssue, - setLinkedGitLabMR, - setLinkedIssue, - setLinkedPR, - setLinkedTaskSourceContext, - setLinkedWorkItem, setName, - setNote, - setPushTarget, - setReuseSelectedBranch, - setStartFromResetHint + setNote ]) return { diff --git a/src/renderer/src/i18n/en-runtime-required.json b/src/renderer/src/i18n/en-runtime-required.json index 6f3c49a21e3..84a51c52468 100644 --- a/src/renderer/src/i18n/en-runtime-required.json +++ b/src/renderer/src/i18n/en-runtime-required.json @@ -3665,18 +3665,29 @@ } }, "status": { - "working": "Working…" + "responding": "Agent is responding", + "thinking": "Thinking", + "toggleDetails": "Toggle turn details", + "workedFor": "Worked for {{value0}}", + "working": "Working…", + "workingFor": "Working for {{value0}}" }, "toggle": { "showChat": "Show chat view", "showTerminal": "Show terminal" }, "tool": { + "countN": "{{value0}} tool calls", + "countOne": "1 tool call", "ranCommandManyToolsSummary": "Ran {{commandCount}} command and used {{toolCount}} tools", "ranCommandOneToolSummary": "Ran {{commandCount}} command and used {{toolCount}} tool", "ranCommandsManyToolsSummary": "Ran {{commandCount}} commands and used {{toolCount}} tools", "ranCommandsOneToolSummary": "Ran {{commandCount}} commands and used {{toolCount}} tool", "running": "Running…", + "runningCommand": "Running command", + "runningNamed": "Running {{toolName}}", + "runningNamedPreview": "Running {{toolName}} {{preview}}", + "runningPreview": "Running {{preview}}", "usedManySummary": "Used {{toolCount}} tools", "usedOneSummary": "Used 1 tool" } diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index e8bc924f64e..4f9919db5c4 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -1,4 +1,12 @@ { + "sidebar": { + "revealFiltered": { + "title": "Reveal hidden workspace?", + "description": "The active workspace is hidden in the sidebar. Revealing it will clear your sidebar filters.", + "confirm": "Clear filters and reveal", + "cancel": "Keep filters" + } + }, "app": { "recoverableError": { "rootTitle": "Orca hit a renderer error.", @@ -6231,6 +6239,11 @@ "projectOnly": "Added in this project only.", "useGlobalFor": "Use global for {{value0}}", "useGlobal": "Use global" + }, + "RepoScanUnavailableIndicator": { + "title": "Worktree scan failed for {{value0}}", + "retry": "Retry scan", + "retained": "Existing worktrees are kept until a scan succeeds. Click to retry." } }, "shared": { @@ -6964,8 +6977,8 @@ "defaultViewTerminal": "Terminal chat", "defaultViewNative": "Chat UI", "structuredTitle": "Use updated structured native chat", - "structuredCopy": "Opt in to the host-owned structured Codex runtime. Off keeps the existing terminal-backed chat path.", - "structuredScope": "Local macOS and Linux sessions only for now. Windows, WSL, and remote execution hosts (including SSH) continue to use terminal chat.", + "structuredCopy": "Opt in to the host-owned structured chat runtime for Codex and Claude. Off keeps the existing terminal-backed chat path.", + "structuredScope": "Local sessions only for now. WSL and remote execution hosts (including SSH) continue to use terminal chat, and Windows falls back to it unless Orca can read process start times.", "structuredToggleLabel": "Toggle updated structured native chat" }, "agentDashboard": { @@ -15299,8 +15312,18 @@ "removeWorktree": "remove worktree", "trashWorktree": "trash worktree", "addQuickCommand": "add quick command", - "newQuickCommand": "new quick command" - } + "newQuickCommand": "new quick command", + "splitChatRight": "split chat right", + "moveChatRight": "move chat right", + "chatPaneRight": "chat pane right", + "splitChatDown": "split chat down", + "moveChatDown": "move chat down", + "chatPaneBelow": "chat pane below" + }, + "splitChatRight": "Split Chat Right", + "splitChatRightDescription": "Open the active chat in a split pane to the right.", + "splitChatDown": "Split Chat Down", + "splitChatDownDescription": "Open the active chat in a split pane below." } }, "palette": { @@ -16943,7 +16966,14 @@ "ranCommandsOneToolSummary": "Ran {{commandCount}} commands and used {{toolCount}} tool", "ranCommandsManyToolsSummary": "Ran {{commandCount}} commands and used {{toolCount}} tools", "usedOneSummary": "Used 1 tool", - "usedManySummary": "Used {{toolCount}} tools" + "usedManySummary": "Used {{toolCount}} tools", + "editedFile": "Edited file", + "addedFile": "Added file", + "deletedFile": "Deleted file", + "renamedFile": "Renamed file", + "copyDiff": "Copy diff", + "diffGap": "Lines not shown", + "diffTruncated": "Diff truncated" }, "providerFrame": { "byteLength": "{{value0}} bytes" @@ -16952,10 +16982,21 @@ "responding": "Agent is responding", "working": "Working…", "thinking": "Thinking", - "workingFor": "Working for {{value0}} seconds", - "workedFor": "Worked for {{value0}} seconds", + "workingFor": "Working for {{value0}}", + "workedFor": "Worked for {{value0}}", "toggleDetails": "Toggle turn details" }, + "backgroundTasks": { + "monitoring": "Monitoring background tasks", + "stop": "Stop", + "agent": "Background agent", + "workflow": "Background workflow", + "command": "Background command", + "monitor": "Background monitor", + "task": "Background task", + "runningList": "Running background tasks", + "detailsUnavailable": "Task details are unavailable for this session." + }, "jumpToLatest": "Jump to latest", "toggle": { "showTerminal": "Show terminal", @@ -17009,9 +17050,45 @@ "message": "Structured Chat blocks terminal prompts and sends. Orchestration messages remain queued; switch to Terminal, then check the Orca inbox with", "command": "orca orchestration check" }, - "structuredSessionCloseFailed": "Could not close this Codex chat", - "structuredSessionLaunchFailed": "Could not open Codex chat", - "structuredSessionCloseFailedDescription": "The terminal stayed open so the provider remains recoverable." + "structuredSessionCloseFailed": "Could not close this chat session", + "structuredSessionLaunchFailed": "Could not open {{value0}} chat", + "structuredSessionLaunchPending": "Starting {{value0}} chat…", + "structuredSessionCloseFailedDescription": "The terminal stayed open so the provider remains recoverable.", + "handoff": { + "stage": { + "finishingChat": "Finishing chat session…", + "finishingTerminal": "Finishing agent terminal…", + "openingTerminal": "Opening agent terminal…", + "resumingChat": "Resuming chat session…", + "verifyingTerminal": "Verifying agent terminal…", + "verifyingChat": "Verifying chat session…", + "recovering": "Recovering agent session…", + "manualRecovery": "Agent session needs recovery" + }, + "switchingOwner": "Switching session owner…", + "mode": { + "switching": "Switching", + "terminal": "Terminal", + "chat": "Chat" + }, + "switchingAfterTurn": "Switching after this turn", + "returningAfterTurn": "Returning after this turn", + "cancel": "Cancel", + "switchAfterTurn": "Switch after this turn", + "stopTurnAndSwitch": "Stop turn and switch", + "openAgentTui": "Open agent TUI", + "returnAfterTurn": "Return after this turn", + "returnToChat": "Return to chat", + "agentOpenOnHost": "Agent is open in terminal on {{value0}}.", + "agentOpen": "Agent is open in terminal.", + "exitTerminal": "Exit the agent terminal to continue in chat.", + "retryProof": "Retry proof", + "retry": "Retry", + "details": "Details" + }, + "structuredSessionFellBackToTerminal": "Structured chat isn't available", + "structuredSessionFellBackToTerminalDescription": "Orca tried to open a {{value0}} terminal instead.", + "structuredSessionLaunchFailedDescription": "Orca could not open a structured {{value0}} chat. See the logs for details." }, "tab": { "bar": { @@ -17535,5 +17612,17 @@ "action": "Try Agents", "hiddenToast": "Agents tab hidden. Re-enable it in Settings → Experimental." } + }, + "rendererRecovery": { + "reload": "Reload", + "copyCommands": "Copy Commands", + "quit": "Quit", + "stalledDetail": "Orca reloaded the window after a crash, but it never finished loading.", + "crashLoopDetail": "Orca tried to recover {{recoveryCount}} times in a row without success.", + "driverFallback": "If that does not help, the cause is usually a graphics driver.", + "genericDetail": "This is often a graphics-driver or installation problem. Reload to try again, or quit and relaunch Orca.", + "title": "Orca keeps failing to load", + "stalledMessage": "The app window stopped responding while reloading after a crash.", + "crashLoopMessage": "The app window crashed repeatedly and stopped reloading automatically." } } diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index 4dc81b9201d..a703b7ad614 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -14561,6 +14561,41 @@ } }, "components": { + "onboarding": { + "flow": { + "stepTooltip": { + "agent": "デフォルトの Agent", + "theme": "外観", + "windowsTerminal": "Windows ターミナル", + "notifications": "通知", + "integrations": "連携" + }, + "actions": { + "addFirstProject": "最初のプロジェクトを追加", + "continue": "続行", + "openingAddProject": "[プロジェクトを追加] を開いています…" + } + }, + "skipConfirmation": { + "skip": "スキップ", + "keepGoing": "いいえ、続ける" + }, + "theme": { + "hints": { + "system": "OS に合わせる", + "dark": "目にやさしい", + "light": "明るく鮮明" + } + }, + "integrations": { + "capabilities": { + "startWorkspaceFromIssue": "GitHub の Issue や PR から、タイトルとコンテキストがあらかじめ入力されたワークスペースを開始", + "browseIssues": "Orca から離れずに、「タスク」ページで GitHub の Issue と PR を閲覧", + "reviewStatus": "すべてのワークツリーで Issue の状態、レビュー状況、CI チェックを確認", + "managePullRequests": "Orca から離れずに、PR の閲覧、コメント、マージ" + } + } + }, "native-chat": { "composer": { "imageUnsupported": "この Agent では画像の貼り付けはサポートされていません。", diff --git a/src/renderer/src/lib/agent-launch-routing.test.ts b/src/renderer/src/lib/agent-launch-routing.test.ts index a16ed1d5ff2..dd33a8357d9 100644 --- a/src/renderer/src/lib/agent-launch-routing.test.ts +++ b/src/renderer/src/lib/agent-launch-routing.test.ts @@ -27,6 +27,68 @@ function route(overrides: Partial[0]> } describe('resolveAgentLaunchRoute', () => { + it.each(['claude', 'codex'] as const)( + 'routes a supported local %s launch to structured native chat', + (agent) => { + expect(route({ agent })).toBe('structured-native-chat') + expect( + route({ agent, launchText: 'explain this change', promptDelivery: 'auto-submit' }) + ).toBe('structured-native-chat') + } + ) + + /** Boundary guard between this lane and the one that owns Windows Codex. Codex's win32 refusal is + * deliberate, so it is asserted against whatever currently lets Claude through rather than + * against one host answer — a future gate swap must not be able to flip Codex on quietly. */ + describe("Codex's Windows refusal", () => { + it('holds in the exact situation that routes Claude to structured', () => { + const onWindows = { platform: 'win32' } as const + expect(route({ ...onWindows, agent: 'claude' })).toBe('structured-native-chat') + expect(route({ ...onWindows, agent: 'codex' })).toBe('legacy-native-chat') + }) + + it('holds for every host capability set, including ones that carry extra gates', () => { + for (const hostCapabilities of [ + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, 'agent-session.structured.claude.v1'], + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, 'agent-session.structured.hold.v1'] + ]) { + expect(route({ agent: 'codex', platform: 'win32', hostCapabilities })).toBe( + 'legacy-native-chat' + ) + } + }) + + it('holds for prompted and folder-workspace launches too', () => { + expect( + route({ + agent: 'codex', + platform: 'win32', + launchText: 'go', + promptDelivery: 'auto-submit' + }) + ).toBe('legacy-native-chat') + expect(route({ agent: 'codex', platform: 'win32', workspaceKind: 'folder' })).toBe( + 'legacy-native-chat' + ) + }) + }) + + /** Pins Codex's whole platform answer, not just win32, so no platform silently changes here. */ + it.each([ + ['darwin', 'structured-native-chat'], + ['linux', 'structured-native-chat'], + ['win32', 'legacy-native-chat'] + ] as const)('leaves Codex routing on %s unchanged', (platform, expected) => { + expect(route({ agent: 'codex', platform })).toBe(expected) + }) + + /** Claude's Windows answer is not a client-side platform guess: the route lets it through and the + * executing host settles it with agentSession.createSupport at create time. */ + it('lets a Windows Claude launch reach the host-measured create support check', () => { + expect(route({ agent: 'claude', platform: 'win32' })).toBe('structured-native-chat') + }) + it('routes a supported local Codex launch to structured native chat', () => { expect(route()).toBe('structured-native-chat') expect(route({ launchText: 'explain this change', promptDelivery: 'auto-submit' })).toBe( @@ -52,7 +114,9 @@ describe('resolveAgentLaunchRoute', () => { it('fails closed for missing capability, unsupported providers, and explicit TUI options', () => { expect(route({ hostCapabilities: [] })).toBe('legacy-native-chat') - expect(route({ agent: 'claude' })).toBe('legacy-native-chat') + // openclaude and grok render native chat but have no structured adapter. + expect(route({ agent: 'openclaude' })).toBe('legacy-native-chat') + expect(route({ agent: 'grok' })).toBe('legacy-native-chat') expect(route({ requiresTuiLaunchCustomization: true })).toBe('legacy-native-chat') expect(route({ initialSessionOptions: { model: 'gpt-5.6-sol' } })).toBe('legacy-native-chat') }) @@ -71,9 +135,11 @@ describe('resolveAgentLaunchRoute', () => { } ) - it('keeps floating, Windows, WSL, and repair-required launches terminal-backed', () => { + it('keeps floating, WSL, and repair-required launches terminal-backed', () => { expect(route({ workspaceKind: 'floating' })).toBe('legacy-native-chat') - expect(route({ platform: 'win32' })).toBe('legacy-native-chat') + expect(route({ agent: 'claude', workspaceKind: 'floating', platform: 'win32' })).toBe( + 'legacy-native-chat' + ) expect( route({ projectRuntime: { diff --git a/src/renderer/src/lib/agent-launch-routing.ts b/src/renderer/src/lib/agent-launch-routing.ts index dd72cb29794..243788f8117 100644 --- a/src/renderer/src/lib/agent-launch-routing.ts +++ b/src/renderer/src/lib/agent-launch-routing.ts @@ -1,5 +1,6 @@ import type { GlobalSettings } from '../../../shared/global-settings-types' import type { ProjectExecutionRuntimeResolution } from '../../../shared/project-execution-runtime' +import { isAgentSessionHandleProvider } from '../../../shared/agent-session-provider-handle' import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../shared/protocol-version' import type { TuiAgent } from '../../../shared/tui-agent' import { @@ -92,13 +93,16 @@ export function resolveAgentLaunchRoute(input: AgentLaunchRoutingInput): AgentLa input.initialSessionOptions && Object.keys(input.initialSessionOptions).length > 0 ) const structuredSupported = - input.agent === 'codex' && + isAgentSessionHandleProvider(input.agent) && input.promptDelivery !== 'draft' && input.workspaceKind !== 'floating' && input.requiresTuiLaunchCustomization !== true && !hasInitialSessionOptions && input.executionHostId === 'local' && - input.platform !== 'win32' && + // Codex's Windows refusal is deliberate and settled elsewhere, so it stays a client-side + // answer. Claude's is measured by the executing host at create time (agentSession.createSupport) + // because only that host knows whether it can read a provider child's start time. + (input.agent !== 'codex' || input.platform !== 'win32') && !runtimeRefused && input.hostCapabilities.includes(STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) diff --git a/src/renderer/src/lib/launch-agent-background-session-remote.test.ts b/src/renderer/src/lib/launch-agent-background-session-remote.test.ts index 6cf8e9b717d..938534f5729 100644 --- a/src/renderer/src/lib/launch-agent-background-session-remote.test.ts +++ b/src/renderer/src/lib/launch-agent-background-session-remote.test.ts @@ -219,6 +219,54 @@ describe('launchAgentBackgroundSession remote runtime and SSH startup delivery', } }) + // #18767: a plain Codex launch carries no shell-ready hint, but the remote host + // still arms the marker for it, so writing early would display the launch twice. + it('waits for shell-ready for a promptless SSH background Codex launch', async () => { + vi.useFakeTimers() + try { + state.repos = [{ id: 'repo-1', connectionId: 'ssh-1', path: '/repo' }] + const { launchAgentBackgroundSession } = await import('./launch-agent-background-session') + + await launchAgentBackgroundSession({ agent: 'codex', worktreeId: 'wt-1' }) + const dataSidecar = mockSubscribeToPtyData.mock.calls[0]?.[1] as (data: string) => void + dataSidecar('user@remote repo % ') + vi.advanceTimersByTime(1_400) + + expect(mockWrite).not.toHaveBeenCalled() + + dataSidecar('\x1b]777;orca-shell-ready\x07user@remote repo % ') + vi.advanceTimersByTime(50) + + expect(mockWrite).toHaveBeenCalledWith( + 'pty-1', + "codex '--dangerously-bypass-approvals-and-sandbox'\r" + ) + } finally { + vi.useRealTimers() + } + }) + + it('skips the shell-ready wait when the host reports it did not arm the marker', async () => { + vi.useFakeTimers() + try { + state.repos = [{ id: 'repo-1', connectionId: 'ssh-1', path: '/repo' }] + mockSpawn.mockResolvedValue({ id: 'pty-1', shellReadyArmed: false }) + const { launchAgentBackgroundSession } = await import('./launch-agent-background-session') + + await launchAgentBackgroundSession({ agent: 'codex', worktreeId: 'wt-1' }) + const dataSidecar = mockSubscribeToPtyData.mock.calls[0]?.[1] as (data: string) => void + dataSidecar('user@remote repo % ') + vi.advanceTimersByTime(50) + + expect(mockWrite).toHaveBeenCalledWith( + 'pty-1', + "codex '--dangerously-bypass-approvals-and-sandbox'\r" + ) + } finally { + vi.useRealTimers() + } + }) + it('falls back when an SSH shell produces no observable startup data', async () => { vi.useFakeTimers() try { diff --git a/src/renderer/src/lib/launch-agent-background-session.ts b/src/renderer/src/lib/launch-agent-background-session.ts index 6bc6c65e420..9ef4bcfae4a 100644 --- a/src/renderer/src/lib/launch-agent-background-session.ts +++ b/src/renderer/src/lib/launch-agent-background-session.ts @@ -28,8 +28,10 @@ import { subscribeToRuntimeTerminalData, toRemoteRuntimePtyId } from '@/runtime/runtime-terminal-stream' -import { createSshBackgroundStartupDelivery } from '@/lib/ssh-background-startup-delivery' -import { shouldUseShellReadyStartupDelivery } from '../../../shared/codex-startup-delivery' +import { + createSshBackgroundStartupDelivery, + sshBackgroundLaunchWaitsForShellReady +} from '@/lib/ssh-background-startup-delivery' import { isMainTerminalSideEffectAuthorityForPty } from '@/components/terminal-pane/terminal-side-effect-facts-handler' import { resolveLocalWindowsAgentStartupShell } from '../../../shared/windows-terminal-shell' import { runBestEffortAgentBackgroundCleanups } from '@/lib/agent-background-session-cleanup' @@ -115,11 +117,7 @@ export async function launchAgentBackgroundSession( const sshStartupDelivery = createSshBackgroundStartupDelivery({ command: sshConnectionId ? startupPlan.launchCommand : null, waitForShellReady: - Boolean(sshConnectionId) && - shouldUseShellReadyStartupDelivery({ - command: startupPlan.launchCommand, - startupCommandDelivery: startupPlan.startupCommandDelivery - }), + Boolean(sshConnectionId) && sshBackgroundLaunchWaitsForShellReady(startupPlan), write: (ptyId, data) => window.api.pty.write(ptyId, data) }) // Route by the worktree's owner host, not the focused runtime. @@ -223,6 +221,7 @@ export async function launchAgentBackgroundSession( }) ptyId = result.id spawned = result + sshStartupDelivery.applyHostShellReadyArmed(result.shellReadyArmed) } const adopted = await adoptAgentBackgroundSessionTab({ store, diff --git a/src/renderer/src/lib/launch-agent-in-new-tab.ts b/src/renderer/src/lib/launch-agent-in-new-tab.ts index 5259a054bfe..118fcdbeda8 100644 --- a/src/renderer/src/lib/launch-agent-in-new-tab.ts +++ b/src/renderer/src/lib/launch-agent-in-new-tab.ts @@ -31,7 +31,8 @@ import type { LaunchSource } from '../../../shared/telemetry-events' import { getConnectionIdFromState } from '@/lib/connection-context' import { resolveInitialNativeChatSessionOptions } from '@/components/native-chat/native-chat-launch-session-options' import { seedNativeChatAppliedSessionOptions } from '@/components/native-chat/native-chat-session-option-cache' -import { startStructuredCodexLaunch } from '@/lib/structured-agent-session-launch' +import { startStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' +import { isAgentSessionHandleProvider } from '../../../shared/agent-session-provider-handle' import { hasExplicitTuiLaunchCustomization, hasExplicitTuiAgentArgs, @@ -226,8 +227,8 @@ function launchAgentInNewTabInternal( hasExplicitTuiLaunchCustomization(store.settings, agent), initialSessionOptions: startupPlan.sessionOptions }) - if (launchRoute === 'structured-native-chat' && agent === 'codex') { - const structuredLaunch = startStructuredCodexLaunch(worktreeId, { + if (launchRoute === 'structured-native-chat' && isAgentSessionHandleProvider(agent)) { + const structuredLaunch = startStructuredAgentLaunch(worktreeId, agent, { prompt: trimmedPrompt, ...(promptDelivery === 'submit-after-ready' ? { promptDelivery } : {}), onPromptDelivered diff --git a/src/renderer/src/lib/launch-agent-structured-chat-guard.test.ts b/src/renderer/src/lib/launch-agent-structured-chat-guard.test.ts index 35fdcc1d261..5753651c051 100644 --- a/src/renderer/src/lib/launch-agent-structured-chat-guard.test.ts +++ b/src/renderer/src/lib/launch-agent-structured-chat-guard.test.ts @@ -13,6 +13,8 @@ const mockLaunchStructuredCodexSession = vi.fn() const mockRefreshLocalStructuredSessionTabs = vi.fn() const mockToastError = vi.fn() const mockCallStructuredAgentSession = vi.fn() +const STRUCTURED_HOST_CAPABILITIES = ['agent-session.structured.v1'] +let hostCapabilities: readonly string[] = STRUCTURED_HOST_CAPABILITIES function structuredLaunchIntent(worktreeId: string, sessionId = 'codex-session-1') { return { @@ -91,12 +93,12 @@ vi.mock('@/runtime/web-runtime-session', () => ({ isWebRuntimeSessionActive: vi.fn(() => false), isWebTerminalSurfaceTabId: vi.fn(() => false) })) -vi.mock('@/lib/launch-structured-codex-session', () => { +vi.mock('@/lib/launch-structured-agent-session', () => { class StructuredAgentSessionCreateRefusalError extends Error {} return { - createStructuredCodexSessionLaunchIntent: mockCreateStructuredCodexSessionLaunchIntent, + createStructuredAgentSessionLaunchIntent: mockCreateStructuredCodexSessionLaunchIntent, abandonStructuredAgentSessionLaunchIntent: mockAbandonStructuredAgentSessionLaunchIntent, - launchStructuredCodexSession: mockLaunchStructuredCodexSession, + launchStructuredAgentSession: mockLaunchStructuredCodexSession, StructuredAgentSessionCreateRefusalError } }) @@ -105,7 +107,7 @@ vi.mock('@/runtime/local-structured-session-tabs-sync', () => ({ LOCAL_STRUCTURED_SESSION_OWNER: 'local-structured-session' })) vi.mock('@/runtime/local-runtime-capabilities', () => ({ - readLocalRuntimeCapabilities: () => ['agent-session.structured.v1'] + readLocalRuntimeCapabilities: () => hostCapabilities })) vi.mock('@/lib/worktree-runtime-owner', () => ({ getExecutionHostIdForWorktree: () => @@ -149,6 +151,7 @@ describe('structured chat adoption guard on the launch path', () => { } ]) mockToastError.mockReset() + hostCapabilities = STRUCTURED_HOST_CAPABILITIES store.settings.openAgentTabsInChatByDefault = true }) @@ -164,7 +167,7 @@ describe('structured chat adoption guard on the launch path', () => { focusAfterMenuClose: 'structured-session' }) expect(shouldQueueTerminalFocusAfterMenuClose(result!)).toBe(false) - expect(mockCreateStructuredCodexSessionLaunchIntent).toHaveBeenCalledWith('wt-1') + expect(mockCreateStructuredCodexSessionLaunchIntent).toHaveBeenCalledWith('wt-1', 'codex') expect(mockLaunchStructuredCodexSession).toHaveBeenCalledWith( expect.objectContaining({ worktreeId: 'wt-1' }) ) @@ -172,6 +175,51 @@ describe('structured chat adoption guard on the launch path', () => { expect(mockWaitForAgentReady).not.toHaveBeenCalled() }) + it('takes the structured path for Claude, naming Claude as the create provider', async () => { + const { launchAgentInNewTab } = await import('./launch-agent-in-new-tab') + + const result = launchAgentInNewTab({ agent: 'claude', worktreeId: 'wt-1' }) + + expect(result).toMatchObject({ tabId: null, focusAfterMenuClose: 'structured-session' }) + expect(mockCreateStructuredCodexSessionLaunchIntent).toHaveBeenCalledWith('wt-1', 'claude') + expect(mockCreateTab).not.toHaveBeenCalled() + }) + + it('keeps a native-chat agent with no structured adapter on the terminal-backed path', async () => { + const { launchAgentInNewTab } = await import('./launch-agent-in-new-tab') + + launchAgentInNewTab({ agent: 'openclaude', worktreeId: 'wt-1' }) + + expect(mockCreateStructuredCodexSessionLaunchIntent).not.toHaveBeenCalled() + expect(mockCreateTab).toHaveBeenCalled() + }) + + it('fails a Claude launch closed to the terminal when the host declines create support', async () => { + const { StructuredAgentSessionCreateRefusalError } = + await import('./launch-structured-agent-session') + mockLaunchStructuredCodexSession.mockRejectedValueOnce( + new StructuredAgentSessionCreateRefusalError('structured_agent_session_unsupported') + ) + const { launchAgentInNewTab } = await import('./launch-agent-in-new-tab') + + const result = launchAgentInNewTab({ agent: 'claude', worktreeId: 'wt-1' }) + + expect(result).toMatchObject({ tabId: null }) + await vi.waitFor(() => expect(mockCreateTab).toHaveBeenCalledOnce()) + expect(mockToastError).not.toHaveBeenCalled() + }) + + it('routes every structured launch through the shared host capability gate', async () => { + hostCapabilities = [] + const { launchAgentInNewTab } = await import('./launch-agent-in-new-tab') + + launchAgentInNewTab({ agent: 'claude', worktreeId: 'wt-1' }) + launchAgentInNewTab({ agent: 'codex', worktreeId: 'wt-1' }) + + expect(mockCreateStructuredCodexSessionLaunchIntent).not.toHaveBeenCalled() + expect(mockCreateTab).toHaveBeenCalledTimes(2) + }) + /** The toggle is hidden under Terminal chat but its persisted value survives, so the launch * path must re-check the default view rather than trust a stale opt-in. */ it('ignores a stale structured opt-in while the default view is Terminal chat', async () => { @@ -192,7 +240,7 @@ describe('structured chat adoption guard on the launch path', () => { it('falls back to the preserved terminal launch on a definitive refusal', async () => { const { StructuredAgentSessionCreateRefusalError } = - await import('./launch-structured-codex-session') + await import('./launch-structured-agent-session') mockLaunchStructuredCodexSession.mockRejectedValueOnce( new StructuredAgentSessionCreateRefusalError('provider unavailable') ) @@ -207,7 +255,7 @@ describe('structured chat adoption guard on the launch path', () => { it('reports prompt delivery from the definitive-refusal terminal fallback', async () => { const { StructuredAgentSessionCreateRefusalError } = - await import('./launch-structured-codex-session') + await import('./launch-structured-agent-session') mockLaunchStructuredCodexSession.mockRejectedValueOnce( new StructuredAgentSessionCreateRefusalError('provider unavailable') ) diff --git a/src/renderer/src/lib/launch-structured-agent-session.test.ts b/src/renderer/src/lib/launch-structured-agent-session.test.ts new file mode 100644 index 00000000000..e9f65f3477b --- /dev/null +++ b/src/renderer/src/lib/launch-structured-agent-session.test.ts @@ -0,0 +1,308 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { structuredAgentSessionPayloadFingerprint } from '../../../shared/structured-agent-session-mutation' +import { callStructuredAgentSession } from '@/runtime/structured-agent-session-client' +import { + createStructuredAgentSessionLaunchIntent, + isDefinitiveStructuredAgentSessionCreateError, + launchStructuredAgentSession, + StructuredAgentSessionCreateRefusalError +} from './launch-structured-agent-session' + +vi.mock('@/runtime/structured-agent-session-client', () => ({ + callStructuredAgentSession: vi.fn() +})) + +describe('structured agent session launch', () => { + beforeEach(() => { + vi.mocked(callStructuredAgentSession).mockReset() + }) + + it('creates a native session with a host-verifiable launch intent', async () => { + vi.mocked(callStructuredAgentSession).mockImplementation(async (_target, _method, params) => ({ + ok: true, + replayed: false, + fence: 1, + cursor: { epoch: 'epoch-1', sequence: 0 }, + value: { + sessionId: (params as { envelope: { sessionId: string } }).envelope.sessionId, + fence: 1, + page: { + sessionId: 'session-1', + epoch: 'epoch-1', + direction: 'tail', + items: [], + removedItemIds: [], + submissions: [], + window: { + oldest: null, + newest: null, + nextCursor: { epoch: 'epoch-1', sequence: 0 } + }, + liveCursor: { epoch: 'epoch-1', sequence: 0 }, + hasOlder: false, + hasNewer: false + }, + unconfirmedClientMessageIds: [] + } + })) + + const intent = createStructuredAgentSessionLaunchIntent('workspace-1', 'codex') + const receipt = await launchStructuredAgentSession(intent) + const params = vi.mocked(callStructuredAgentSession).mock.calls[0]?.[2] as { + envelope: { sessionId: string; payloadFingerprint: string } + worktree: string + agent: 'codex' + } + + expect(receipt).toEqual({ + sessionId: expect.stringMatching(/^codex_[A-Za-z0-9_]{36}$/), + fence: 1 + }) + expect(callStructuredAgentSession).toHaveBeenCalledWith( + { kind: 'local' }, + 'agentSession.create', + expect.objectContaining({ worktree: 'id:workspace-1', agent: 'codex' }) + ) + expect(params.envelope.payloadFingerprint).toBe( + structuredAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: params.envelope.sessionId, + fields: { worktree: 'id:workspace-1', agent: 'codex' } + }) + ) + expect(params).toBe(intent.params) + }) + + it('names Claude as the create provider and in the session id', () => { + const intent = createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + expect(intent.sessionId).toMatch(/^claude_[A-Za-z0-9_]{36}$/) + expect(intent.params.agent).toBe('claude') + expect(intent.params.envelope.payloadFingerprint).toBe( + structuredAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: intent.sessionId, + fields: { worktree: 'id:workspace-1', agent: 'claude' } + }) + ) + }) + + it('asks the executing host for create support before creating a Claude session', async () => { + vi.mocked(callStructuredAgentSession).mockImplementation(async (_target, method) => + method === 'agentSession.createSupport' + ? { supported: true } + : { ok: true, replayed: false, value: { sessionId: 'claude_1', fence: 1 } } + ) + + const intent = createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + await launchStructuredAgentSession(intent) + + expect(vi.mocked(callStructuredAgentSession).mock.calls.map(([, method]) => method)).toEqual([ + 'agentSession.createSupport', + 'agentSession.create' + ]) + expect(callStructuredAgentSession).toHaveBeenNthCalledWith( + 1, + { kind: 'local' }, + 'agentSession.createSupport', + { worktree: 'id:workspace-1', agent: 'claude' } + ) + }) + + it('refuses a Claude launch the host says it cannot support, without creating', async () => { + vi.mocked(callStructuredAgentSession).mockResolvedValue({ supported: false, reason: 'agent' }) + + const intent = createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + + await expect(launchStructuredAgentSession(intent)).rejects.toBeInstanceOf( + StructuredAgentSessionCreateRefusalError + ) + expect(vi.mocked(callStructuredAgentSession).mock.calls.map(([, method]) => method)).toEqual([ + 'agentSession.createSupport' + ]) + }) + + it('fails closed when the create support probe cannot be answered', async () => { + vi.mocked(callStructuredAgentSession).mockRejectedValue(new Error('runtime unreachable')) + + const intent = createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + + await expect(launchStructuredAgentSession(intent)).rejects.toBeInstanceOf( + StructuredAgentSessionCreateRefusalError + ) + expect(callStructuredAgentSession).toHaveBeenCalledOnce() + }) + + /** A worktree is not resolvable for a beat after createWorktree resolves, so the probe fails with + * selector_not_found instead of answering. That is "not ready", not "no". */ + it('retries a probe the host cannot answer yet, then creates', async () => { + const notResolvableYet = Object.assign(new Error('selector_not_found'), { + code: 'selector_not_found' + }) + vi.mocked(callStructuredAgentSession) + .mockRejectedValueOnce(notResolvableYet) + .mockRejectedValueOnce(notResolvableYet) + .mockImplementation(async (_target, method) => + method === 'agentSession.createSupport' + ? { supported: true } + : { ok: true, replayed: false, value: { sessionId: 'claude_1', fence: 1 } } + ) + + const intent = createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + await expect(launchStructuredAgentSession(intent)).resolves.toMatchObject({ + sessionId: 'claude_1' + }) + expect(vi.mocked(callStructuredAgentSession).mock.calls.map(([, method]) => method)).toEqual([ + 'agentSession.createSupport', + 'agentSession.createSupport', + 'agentSession.createSupport', + 'agentSession.create' + ]) + }) + + it('refuses once the retry budget for an unresolvable selector is spent', async () => { + vi.mocked(callStructuredAgentSession).mockRejectedValue( + Object.assign(new Error('selector_not_found'), { code: 'selector_not_found' }) + ) + + const intent = createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + + await expect(launchStructuredAgentSession(intent)).rejects.toBeInstanceOf( + StructuredAgentSessionCreateRefusalError + ) + // Bounded: the first ask plus the retry delays, and never agentSession.create. + expect(vi.mocked(callStructuredAgentSession).mock.calls.map(([, method]) => method)).toEqual([ + 'agentSession.createSupport', + 'agentSession.createSupport', + 'agentSession.createSupport', + 'agentSession.createSupport' + ]) + }) + + it('does not retry a host that answered no, or an unrelated failure', async () => { + vi.mocked(callStructuredAgentSession).mockResolvedValue({ supported: false, reason: 'wsl' }) + await expect( + launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + ) + ).rejects.toBeInstanceOf(StructuredAgentSessionCreateRefusalError) + expect(callStructuredAgentSession).toHaveBeenCalledOnce() + + vi.mocked(callStructuredAgentSession).mockReset() + vi.mocked(callStructuredAgentSession).mockRejectedValue(new Error('runtime unreachable')) + await expect( + launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + ) + ).rejects.toBeInstanceOf(StructuredAgentSessionCreateRefusalError) + expect(callStructuredAgentSession).toHaveBeenCalledOnce() + }) + + /** A message-wrapped token must not be confused with prose that merely mentions it. */ + it('does not retry a failure that only mentions the token in passing', async () => { + vi.mocked(callStructuredAgentSession).mockRejectedValue( + new Error('Access denied after a prior selector_not_found') + ) + + await expect( + launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + ) + ).rejects.toBeInstanceOf(StructuredAgentSessionCreateRefusalError) + expect(callStructuredAgentSession).toHaveBeenCalledOnce() + }) + + /** Codex's support answer is settled by the launch route and owned elsewhere; this pins that the + * Claude probe did not change Codex's wire traffic. */ + it('does not probe create support for Codex', async () => { + vi.mocked(callStructuredAgentSession).mockResolvedValue({ + ok: true, + replayed: false, + value: { sessionId: 'codex_1', fence: 1 } + }) + + await launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-1', 'codex') + ) + + expect(vi.mocked(callStructuredAgentSession).mock.calls.map(([, method]) => method)).toEqual([ + 'agentSession.create' + ]) + }) + + it('replays the exact create envelope when an unknown outcome is retried', async () => { + const intent = createStructuredAgentSessionLaunchIntent('workspace-retry', 'codex') + vi.mocked(callStructuredAgentSession).mockRejectedValue(new Error('response lost')) + + await expect(launchStructuredAgentSession(intent)).rejects.toThrow('response lost') + await expect(launchStructuredAgentSession(intent)).rejects.toThrow('response lost') + + const first = vi.mocked(callStructuredAgentSession).mock.calls[0]?.[2] + const second = vi.mocked(callStructuredAgentSession).mock.calls[1]?.[2] + expect(first).toBe(intent.params) + expect(second).toBe(first) + expect(intent.params.envelope.clientOperationId).toMatch(/^\d{13}-[0-9a-f]{32}$/) + }) + + it('preserves an unknown refusal code without classifying it as fallback-safe', async () => { + vi.mocked(callStructuredAgentSession).mockResolvedValue({ + ok: false, + refusal: { + code: 'agent_session_operation_unknown', + message: 'The chat may already exist.' + } + }) + + const error = await launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-unknown', 'codex') + ).catch((caught: unknown) => caught) + + expect(error).not.toBeInstanceOf(StructuredAgentSessionCreateRefusalError) + expect(error).toMatchObject({ code: 'agent_session_operation_unknown' }) + expect(isDefinitiveStructuredAgentSessionCreateError(error)).toBe(false) + }) + + it('preserves a definitive refusal code for the fallback path', async () => { + vi.mocked(callStructuredAgentSession).mockResolvedValue({ + ok: false, + refusal: { + code: 'structured_agent_session_unsupported', + message: 'Structured chat is unavailable.' + } + }) + + const error = await launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-unsupported', 'codex') + ).catch((caught: unknown) => caught) + + expect(error).toBeInstanceOf(StructuredAgentSessionCreateRefusalError) + expect(error).toMatchObject({ code: 'structured_agent_session_unsupported' }) + expect(isDefinitiveStructuredAgentSessionCreateError(error)).toBe(true) + }) + + it.each(['method_not_found', 'structured_agent_session_unsupported'])( + 'turns an old-host %s error into a definitive transport refusal', + async (code) => { + vi.mocked(callStructuredAgentSession).mockRejectedValueOnce( + Object.assign(new Error(code), { code }) + ) + const oldHostError = await launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent(`workspace-old-host-${code}`, 'codex') + ).catch((caught: unknown) => caught) + + expect(oldHostError).toBeInstanceOf(StructuredAgentSessionCreateRefusalError) + expect(oldHostError).toMatchObject({ code }) + } + ) + + it('keeps an unclassified transport failure outcome unknown', async () => { + vi.mocked(callStructuredAgentSession).mockRejectedValueOnce( + Object.assign(new Error('Connection lost'), { code: 'runtime_error' }) + ) + const transportError = await launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-offline', 'codex') + ).catch((caught: unknown) => caught) + + expect(transportError).not.toBeInstanceOf(StructuredAgentSessionCreateRefusalError) + expect(isDefinitiveStructuredAgentSessionCreateError(transportError)).toBe(false) + }) +}) diff --git a/src/renderer/src/lib/launch-structured-agent-session.ts b/src/renderer/src/lib/launch-structured-agent-session.ts new file mode 100644 index 00000000000..6c2d1694437 --- /dev/null +++ b/src/renderer/src/lib/launch-structured-agent-session.ts @@ -0,0 +1,214 @@ +import type { AgentSessionHandleProvider } from '../../../shared/agent-session-provider-handle' +import type { + AgentSessionAttachResult, + AgentSessionMutationEnvelope, + AgentSessionMutationResult +} from '../../../shared/agent-session-wire' +import { + createStructuredAgentSessionOperationId, + structuredAgentSessionPayloadFingerprint +} from '../../../shared/structured-agent-session-mutation' +import { hasRuntimeRpcErrorCode } from '../../../shared/runtime-rpc-error-code' +import { isDefinitiveAgentSessionCreateRefusal } from '../../../shared/agent-session-definitive-refusal' +import { callStructuredAgentSession } from '@/runtime/structured-agent-session-client' +import { toRuntimeWorktreeSelector } from '@/runtime/runtime-worktree-selector' +import { useAppStore } from '@/store' +import { + clearWebSessionFocusIntentIfMatches, + recordWebSessionFocusIntent, + resolveWebSessionVisibleTabId +} from '@/runtime/web-session-focus-intent' +import { LOCAL_STRUCTURED_SESSION_OWNER } from '@/runtime/local-structured-session-tabs-sync' + +type StructuredAgentSessionCreateParams = { + envelope: AgentSessionMutationEnvelope + worktree: string + agent: AgentSessionHandleProvider +} + +export type StructuredAgentSessionLaunchIntent = { + sessionId: string + worktreeId: string + agent: AgentSessionHandleProvider + params: StructuredAgentSessionCreateParams +} + +export class StructuredAgentSessionCreateRefusalError extends Error { + constructor( + message: string, + readonly code: string = 'structured_agent_session_unsupported' + ) { + super(message) + this.name = 'StructuredAgentSessionCreateRefusalError' + } +} + +const DEFINITIVE_CREATE_FAILURE_CODES = [ + 'structured_agent_session_unsupported', + 'method_not_found' +] as const + +function definitiveStructuredAgentSessionCreateErrorCode(error: unknown): string | null { + if (error instanceof StructuredAgentSessionCreateRefusalError) { + return isDefinitiveAgentSessionCreateRefusal(error.code) ? error.code : null + } + for (const code of DEFINITIVE_CREATE_FAILURE_CODES) { + if (hasRuntimeRpcErrorCode(error, code)) { + return code + } + } + return null +} + +export function isDefinitiveStructuredAgentSessionCreateError(error: unknown): boolean { + return definitiveStructuredAgentSessionCreateErrorCode(error) !== null +} + +export function createStructuredAgentSessionLaunchIntent( + worktreeId: string, + agent: AgentSessionHandleProvider +): StructuredAgentSessionLaunchIntent { + const sessionId = `${agent}_${crypto.randomUUID().replaceAll('-', '_')}` + const fields = { worktree: toRuntimeWorktreeSelector(worktreeId), agent } + const state = useAppStore.getState() + recordWebSessionFocusIntent( + { environmentId: LOCAL_STRUCTURED_SESSION_OWNER }, + worktreeId, + `agent-session:${sessionId}`, + undefined, + resolveWebSessionVisibleTabId(state, worktreeId) + ) + return { + sessionId, + worktreeId, + agent, + params: { + envelope: { + sessionId, + clientOperationId: createStructuredAgentSessionOperationId(() => crypto.randomUUID()), + expectedRuntimeFence: null, + payloadFingerprint: structuredAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId, + fields + }) + }, + ...fields + } + } +} + +export function abandonStructuredAgentSessionLaunchIntent( + intent: StructuredAgentSessionLaunchIntent +): void { + clearWebSessionFocusIntentIfMatches( + { environmentId: LOCAL_STRUCTURED_SESSION_OWNER }, + intent.worktreeId, + `agent-session:${intent.sessionId}` + ) +} + +/** The host answers a worktree selector it cannot resolve yet with this rather than a verdict. */ +const SELECTOR_NOT_RESOLVABLE_CODE = 'selector_not_found' + +/** + * A worktree is not resolvable for a beat after `createWorktree` resolves, so a probe fired + * immediately after creation fails instead of answering. Measured window: under ~250ms. These + * delays cover it with margin and bound the wait when the selector is genuinely absent. + */ +const CREATE_SUPPORT_RETRY_DELAYS_MS: readonly number[] = [50, 150, 300] + +function delay(ms: number): Promise { + return new Promise((resolve) => setTimeout(resolve, ms)) +} + +/** + * Whether the executing host supports creating this session — retrying only while the host cannot + * yet resolve the worktree. + * + * "Could not answer" and "answered no" are different states and only the second is a verdict. + * Collapsing them sends a launch to the terminal because a selector was a beat late, which is + * indistinguishable to the user from the gate refusing them. The retry is narrowed to that one + * transient code so every other failure still refuses on the first ask. + */ +async function hostSupportsCreate(intent: StructuredAgentSessionLaunchIntent): Promise { + for (let attempt = 0; ; attempt += 1) { + try { + const support = await callStructuredAgentSession<{ supported: boolean; reason?: string }>( + { kind: 'local' }, + 'agentSession.createSupport', + { worktree: intent.params.worktree, agent: intent.agent } + ) + return support.supported === true + } catch (error) { + const retryDelayMs = CREATE_SUPPORT_RETRY_DELAYS_MS[attempt] + if ( + retryDelayMs === undefined || + !hasRuntimeRpcErrorCode(error, SELECTOR_NOT_RESOLVABLE_CODE) + ) { + // An unanswered probe is still not a yes. + return false + } + await delay(retryDelayMs) + } + } +} + +/** + * Only the host that will execute the session can answer whether it supports creating one there — + * on Windows that means reading the provider child's process start time, which a client cannot + * observe. + * + * Codex is absent on purpose: its answer is settled by the launch route and owned elsewhere, so + * probing here would change Codex's wire traffic. Note that this early return is also why the + * unresolvable-selector race above has never been able to refuse a Codex launch — the race is + * identical for Codex, nothing asks. Whoever gives Codex a probe inherits it. + */ +async function requireHostCreateSupport(intent: StructuredAgentSessionLaunchIntent): Promise { + if (intent.agent !== 'claude') { + return + } + if (!(await hostSupportsCreate(intent))) { + abandonStructuredAgentSessionLaunchIntent(intent) + throw new StructuredAgentSessionCreateRefusalError( + 'structured_agent_session_unsupported', + 'structured_agent_session_unsupported' + ) + } +} + +export async function launchStructuredAgentSession( + intent: StructuredAgentSessionLaunchIntent +): Promise> { + await requireHostCreateSupport(intent) + let result: AgentSessionMutationResult + try { + result = await callStructuredAgentSession>( + { kind: 'local' }, + 'agentSession.create', + intent.params + ) + } catch (error) { + const code = definitiveStructuredAgentSessionCreateErrorCode(error) + if (code) { + abandonStructuredAgentSessionLaunchIntent(intent) + throw new StructuredAgentSessionCreateRefusalError( + error instanceof Error ? error.message : String(error), + code + ) + } + throw error + } + if (!result.ok) { + const error = new StructuredAgentSessionCreateRefusalError( + result.refusal.message, + result.refusal.code + ) + if (isDefinitiveStructuredAgentSessionCreateError(error)) { + abandonStructuredAgentSessionLaunchIntent(intent) + throw error + } + throw Object.assign(new Error(error.message), { code: error.code }) + } + return { sessionId: result.value.sessionId, fence: result.value.fence } +} diff --git a/src/renderer/src/lib/launch-structured-codex-session.test.ts b/src/renderer/src/lib/launch-structured-codex-session.test.ts deleted file mode 100644 index ea498c81719..00000000000 --- a/src/renderer/src/lib/launch-structured-codex-session.test.ts +++ /dev/null @@ -1,87 +0,0 @@ -import { beforeEach, describe, expect, it, vi } from 'vitest' -import { structuredAgentSessionPayloadFingerprint } from '../../../shared/structured-agent-session-mutation' -import { callStructuredAgentSession } from '@/runtime/structured-agent-session-client' -import { - createStructuredCodexSessionLaunchIntent, - launchStructuredCodexSession -} from './launch-structured-codex-session' - -vi.mock('@/runtime/structured-agent-session-client', () => ({ - callStructuredAgentSession: vi.fn() -})) - -describe('structured Codex launch', () => { - beforeEach(() => { - vi.mocked(callStructuredAgentSession).mockReset() - }) - - it('creates a native session with a host-verifiable launch intent', async () => { - vi.mocked(callStructuredAgentSession).mockImplementation(async (_target, _method, params) => ({ - ok: true, - replayed: false, - fence: 1, - cursor: { epoch: 'epoch-1', sequence: 0 }, - value: { - sessionId: (params as { envelope: { sessionId: string } }).envelope.sessionId, - fence: 1, - page: { - sessionId: 'session-1', - epoch: 'epoch-1', - direction: 'tail', - items: [], - removedItemIds: [], - submissions: [], - window: { - oldest: null, - newest: null, - nextCursor: { epoch: 'epoch-1', sequence: 0 } - }, - liveCursor: { epoch: 'epoch-1', sequence: 0 }, - hasOlder: false, - hasNewer: false - }, - unconfirmedClientMessageIds: [] - } - })) - - const intent = createStructuredCodexSessionLaunchIntent('workspace-1') - const receipt = await launchStructuredCodexSession(intent) - const params = vi.mocked(callStructuredAgentSession).mock.calls[0]?.[2] as { - envelope: { sessionId: string; payloadFingerprint: string } - worktree: string - agent: 'codex' - } - - expect(receipt).toEqual({ - sessionId: expect.stringMatching(/^codex_[A-Za-z0-9_]{36}$/), - fence: 1 - }) - expect(callStructuredAgentSession).toHaveBeenCalledWith( - { kind: 'local' }, - 'agentSession.create', - expect.objectContaining({ worktree: 'id:workspace-1', agent: 'codex' }) - ) - expect(params.envelope.payloadFingerprint).toBe( - structuredAgentSessionPayloadFingerprint({ - method: 'agentSession.create', - sessionId: params.envelope.sessionId, - fields: { worktree: 'id:workspace-1', agent: 'codex' } - }) - ) - expect(params).toBe(intent.params) - }) - - it('replays the exact create envelope when an unknown outcome is retried', async () => { - const intent = createStructuredCodexSessionLaunchIntent('workspace-retry') - vi.mocked(callStructuredAgentSession).mockRejectedValue(new Error('response lost')) - - await expect(launchStructuredCodexSession(intent)).rejects.toThrow('response lost') - await expect(launchStructuredCodexSession(intent)).rejects.toThrow('response lost') - - const first = vi.mocked(callStructuredAgentSession).mock.calls[0]?.[2] - const second = vi.mocked(callStructuredAgentSession).mock.calls[1]?.[2] - expect(first).toBe(intent.params) - expect(second).toBe(first) - expect(intent.params.envelope.clientOperationId).toMatch(/^\d{13}-[0-9a-f]{32}$/) - }) -}) diff --git a/src/renderer/src/lib/launch-structured-codex-session.ts b/src/renderer/src/lib/launch-structured-codex-session.ts deleted file mode 100644 index 32a3feb19b4..00000000000 --- a/src/renderer/src/lib/launch-structured-codex-session.ts +++ /dev/null @@ -1,87 +0,0 @@ -import type { - AgentSessionAttachResult, - AgentSessionMutationEnvelope, - AgentSessionMutationResult -} from '../../../shared/agent-session-wire' -import { - createStructuredAgentSessionOperationId, - structuredAgentSessionPayloadFingerprint -} from '../../../shared/structured-agent-session-mutation' -import { callStructuredAgentSession } from '@/runtime/structured-agent-session-client' -import { toRuntimeWorktreeSelector } from '@/runtime/runtime-worktree-selector' -import { useAppStore } from '@/store' -import { - clearWebSessionFocusIntentIfMatches, - recordWebSessionFocusIntent, - resolveWebSessionVisibleTabId -} from '@/runtime/web-session-focus-intent' -import { LOCAL_STRUCTURED_SESSION_OWNER } from '@/runtime/local-structured-session-tabs-sync' - -type StructuredAgentSessionCreateParams = { - envelope: AgentSessionMutationEnvelope - worktree: string - agent: 'codex' -} - -export type StructuredAgentSessionLaunchIntent = { - sessionId: string - worktreeId: string - params: StructuredAgentSessionCreateParams -} - -export class StructuredAgentSessionCreateRefusalError extends Error {} - -export function createStructuredCodexSessionLaunchIntent( - worktreeId: string -): StructuredAgentSessionLaunchIntent { - const sessionId = `codex_${crypto.randomUUID().replaceAll('-', '_')}` - const fields = { worktree: toRuntimeWorktreeSelector(worktreeId), agent: 'codex' as const } - const state = useAppStore.getState() - recordWebSessionFocusIntent( - { environmentId: LOCAL_STRUCTURED_SESSION_OWNER }, - worktreeId, - `agent-session:${sessionId}`, - undefined, - resolveWebSessionVisibleTabId(state, worktreeId) - ) - return { - sessionId, - worktreeId, - params: { - envelope: { - sessionId, - clientOperationId: createStructuredAgentSessionOperationId(() => crypto.randomUUID()), - expectedRuntimeFence: null, - payloadFingerprint: structuredAgentSessionPayloadFingerprint({ - method: 'agentSession.create', - sessionId, - fields - }) - }, - ...fields - } - } -} - -export function abandonStructuredAgentSessionLaunchIntent( - intent: StructuredAgentSessionLaunchIntent -): void { - clearWebSessionFocusIntentIfMatches( - { environmentId: LOCAL_STRUCTURED_SESSION_OWNER }, - intent.worktreeId, - `agent-session:${intent.sessionId}` - ) -} - -export async function launchStructuredCodexSession( - intent: StructuredAgentSessionLaunchIntent -): Promise> { - const result = await callStructuredAgentSession< - AgentSessionMutationResult - >({ kind: 'local' }, 'agentSession.create', intent.params) - if (!result.ok) { - abandonStructuredAgentSessionLaunchIntent(intent) - throw new StructuredAgentSessionCreateRefusalError(result.refusal.message) - } - return { sessionId: result.value.sessionId, fence: result.value.fence } -} diff --git a/src/renderer/src/lib/launch-work-item-direct-agent-routing.test.ts b/src/renderer/src/lib/launch-work-item-direct-agent-routing.test.ts index f0e9cf5b783..7d3e6522bc8 100644 --- a/src/renderer/src/lib/launch-work-item-direct-agent-routing.test.ts +++ b/src/renderer/src/lib/launch-work-item-direct-agent-routing.test.ts @@ -1,13 +1,13 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' const mocks = vi.hoisted(() => ({ - startStructuredCodexLaunch: vi.fn(), + startStructuredAgentLaunch: vi.fn(), activateAndRevealWorktree: vi.fn(), preflightAgentTrust: vi.fn() })) vi.mock('@/lib/structured-agent-session-launch', () => ({ - startStructuredCodexLaunch: mocks.startStructuredCodexLaunch + startStructuredAgentLaunch: mocks.startStructuredAgentLaunch })) vi.mock('@/lib/worktree-activation', () => ({ @@ -18,7 +18,7 @@ vi.mock('@/lib/agent-trust-preflight', () => ({ preflightAgentTrust: mocks.preflightAgentTrust })) -vi.mock('@/lib/launch-structured-codex-session', () => ({ +vi.mock('@/lib/launch-structured-agent-session', () => ({ StructuredAgentSessionCreateRefusalError: class extends Error {} })) @@ -26,7 +26,7 @@ vi.mock('@/lib/native-chat-transcript-readability', () => ({ isNativeChatTranscriptLocalReadable: vi.fn(() => true) })) -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' import { settleDirectWorkItemStructuredLaunch } from './launch-work-item-direct-agent-routing' const baseArgs = { @@ -47,7 +47,7 @@ describe('settleDirectWorkItemStructuredLaunch', () => { it('runs the legacy terminal fallback after a definitive refusal', async () => { mocks.activateAndRevealWorktree.mockReturnValue({ primaryTabId: 'fallback-tab' }) - mocks.startStructuredCodexLaunch.mockReturnValue({ + mocks.startStructuredAgentLaunch.mockReturnValue({ launchResult: Promise.reject(new StructuredAgentSessionCreateRefusalError('unsupported')), isVisibilityUnknown: () => false, claimDefinitiveRefusalFallback: (fallback: () => Promise) => @@ -65,7 +65,7 @@ describe('settleDirectWorkItemStructuredLaunch', () => { }) it('reports an unknown outcome without starting a fallback terminal', async () => { - mocks.startStructuredCodexLaunch.mockReturnValue({ + mocks.startStructuredAgentLaunch.mockReturnValue({ launchResult: Promise.reject(new Error('connection lost')), isVisibilityUnknown: () => true, claimDefinitiveRefusalFallback: vi.fn(() => Promise.resolve(false)) diff --git a/src/renderer/src/lib/launch-work-item-direct-agent-routing.ts b/src/renderer/src/lib/launch-work-item-direct-agent-routing.ts index 43f6a6e700d..7d77b8d0d18 100644 --- a/src/renderer/src/lib/launch-work-item-direct-agent-routing.ts +++ b/src/renderer/src/lib/launch-work-item-direct-agent-routing.ts @@ -1,3 +1,4 @@ +import { isAgentSessionHandleProvider } from '../../../shared/agent-session-provider-handle' import type { TuiAgent } from '../../../shared/tui-agent' import type { AgentStartupPlan } from '@/lib/tui-agent-startup' import type { LaunchSource } from '../../../shared/telemetry-events' @@ -9,8 +10,8 @@ import { buildDirectWorkItemAgentStartupPlan, buildDirectWorkItemStartupOpts } from '@/lib/launch-work-item-direct-agent' -import { startStructuredCodexLaunch } from '@/lib/structured-agent-session-launch' -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { startStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' import { isNativeChatTranscriptLocalReadable } from '@/lib/native-chat-transcript-readability' import { resolveSourceControlLaunchPlatform } from '@/lib/source-control-launch-platform' import { preflightAgentTrust } from '@/lib/agent-trust-preflight' @@ -127,11 +128,11 @@ export async function settleDirectWorkItemStructuredLaunch(args: { primaryTabId: string | null }> { let { structuredLaunch, primaryTabId } = args - if (!structuredLaunch || args.agent !== 'codex') { + if (!structuredLaunch || !isAgentSessionHandleProvider(args.agent)) { return { completed: false, structuredLaunch, visibilityUnknown: false, primaryTabId } } - const launch = startStructuredCodexLaunch(args.worktreeId, { + const launch = startStructuredAgentLaunch(args.worktreeId, args.agent, { prompt: args.draftContent, ...(args.promptDelivery === 'submit-after-ready' ? { promptDelivery: args.promptDelivery } : {}) }) diff --git a/src/renderer/src/lib/ssh-background-startup-delivery.test.ts b/src/renderer/src/lib/ssh-background-startup-delivery.test.ts index ee5248cf2fd..427d1dbdb8f 100644 --- a/src/renderer/src/lib/ssh-background-startup-delivery.test.ts +++ b/src/renderer/src/lib/ssh-background-startup-delivery.test.ts @@ -18,6 +18,25 @@ function createDelivery(): { } } +// Bracketed paste only wraps multiline submissions, so the marker's effect on it +// is only observable through a command that carries a newline. +const MULTILINE_COMMAND = 'codex "run the\nautomation"' + +function createMultilineDelivery(waitForShellReady: boolean): { + delivery: ReturnType + write: ReturnType +} { + const write = vi.fn() + return { + delivery: createSshBackgroundStartupDelivery({ + command: MULTILINE_COMMAND, + waitForShellReady, + write + }), + write + } +} + beforeEach(() => { vi.useFakeTimers() }) @@ -89,4 +108,89 @@ describe('createSshBackgroundStartupDelivery shell-ready fallback', () => { expect(write).toHaveBeenCalledTimes(1) }) + + // #18767: the marker is what proves the host wrapped the shell and armed + // bracketed paste. A fallback release means it never did. + it('uses bracketed paste only after the marker actually arrived', () => { + const { delivery, write } = createMultilineDelivery(true) + + delivery.armFallback('pty-1') + delivery.handleData(`${SHELL_READY}user@remote repo % `) + vi.advanceTimersByTime(50) + + expect(write.mock.calls[0]?.[1]).toContain('\x1b[200~') + }) + + // The host answers in the spawn reply whether it armed the marker. `false` means + // none will ever come, so the pre-#18796 fast path applies; `true` keeps the wait; + // absent is an older host and leaves the client-side prediction alone. + describe('host shell-ready verdict', () => { + it('delivers immediately and raw when the host reports the marker was not armed', () => { + const { delivery, write } = createMultilineDelivery(true) + + delivery.applyHostShellReadyArmed(false) + delivery.armFallback('pty-1') + // The launch flow schedules on every data chunk (launch-agent-background-session). + delivery.handleData('user@remote repo % ') + delivery.schedule('pty-1') + vi.advanceTimersByTime(50) + + expect(write).toHaveBeenCalledTimes(1) + expect(write.mock.calls[0]?.[1]).not.toContain('\x1b[200~') + }) + + it('keeps the short silent-shell budget once the host says no marker is coming', () => { + const { delivery, write } = createDelivery() + + delivery.applyHostShellReadyArmed(false) + delivery.armFallback('pty-1') + vi.advanceTimersByTime(1_550) + + expect(write).toHaveBeenCalledTimes(1) + }) + + it('still waits for the marker when the host reports it armed one', () => { + const { delivery, write } = createDelivery() + + delivery.applyHostShellReadyArmed(true) + delivery.armFallback('pty-1') + delivery.handleData('user@remote repo % ') + vi.advanceTimersByTime(1_400) + + expect(write).not.toHaveBeenCalled() + + delivery.handleData(`${SHELL_READY}user@remote repo % `) + vi.advanceTimersByTime(50) + + expect(write).toHaveBeenCalledTimes(1) + }) + + it('keeps the client-side prediction when an older host omits the verdict', () => { + const { delivery, write } = createDelivery() + + delivery.applyHostShellReadyArmed(undefined) + delivery.armFallback('pty-1') + delivery.handleData('user@remote repo % ') + vi.advanceTimersByTime(1_400) + + expect(write).not.toHaveBeenCalled() + + vi.advanceTimersByTime(150) + vi.advanceTimersByTime(50) + + expect(write).toHaveBeenCalledTimes(1) + }) + }) + + it('submits raw when the wait ends at the fallback instead of the marker', () => { + const { delivery, write } = createMultilineDelivery(true) + + delivery.armFallback('pty-1') + delivery.handleData('user@remote repo % ') + vi.advanceTimersByTime(1_550) + vi.advanceTimersByTime(50) + + expect(write).toHaveBeenCalledTimes(1) + expect(write.mock.calls[0]?.[1]).not.toContain('\x1b[200~') + }) }) diff --git a/src/renderer/src/lib/ssh-background-startup-delivery.ts b/src/renderer/src/lib/ssh-background-startup-delivery.ts index 1c11254d6bb..0e1c08c70ff 100644 --- a/src/renderer/src/lib/ssh-background-startup-delivery.ts +++ b/src/renderer/src/lib/ssh-background-startup-delivery.ts @@ -2,8 +2,34 @@ import { createShellReadyMarkerScanState, scanForShellReadyMarker } from '@/components/terminal-pane/shell-ready-marker-scan' +import { + isCodexStartupCommand, + shouldUseShellReadyStartupDelivery, + type StartupCommandDelivery +} from '../../../shared/codex-startup-delivery' import { buildStartupCommandSubmission } from '../../../shared/startup-command-submission' +/** + * Why every Codex launch waits and not only the prompt-carrying ones: the remote + * shell is the host's to know, and it arms the ready marker for plain Codex too + * (#18767). On such a host the wait ends at the prompt and costs nothing. On one + * that never publishes a marker -- fish, sh, Windows, or a host predating #18767 -- + * the fallback below releases instead, at the same price prompt-carrying Codex + * already paid there. + */ +export function sshBackgroundLaunchWaitsForShellReady(startupPlan: { + launchCommand: string | null | undefined + startupCommandDelivery?: StartupCommandDelivery +}): boolean { + return ( + isCodexStartupCommand(startupPlan.launchCommand) || + shouldUseShellReadyStartupDelivery({ + command: startupPlan.launchCommand, + startupCommandDelivery: startupPlan.startupCommandDelivery + }) + ) +} + const SSH_SHELL_READY_STARTUP_FALLBACK_MS = 1500 // Why: a remote shell that has not emitted a single byte is still booting — // /etc/profile plus nvm/conda/pyenv over a cold link routinely needs more than @@ -22,6 +48,12 @@ export type SshBackgroundStartupDelivery = { handleData(data: string): string armFallback(ptyId: string): void schedule(ptyId: string): void + /** + * The host's verdict from the spawn reply, which lands after this delivery was built. + * `false` releases the wait: no marker will ever come. `true` keeps it. `undefined` is + * a host that predates the field, so the constructor-time prediction stands. + */ + applyHostShellReadyArmed(armed: boolean | undefined): void clear(): void } @@ -30,8 +62,13 @@ export function createSshBackgroundStartupDelivery( ): SshBackgroundStartupDelivery { let pendingCommand = options.command let lastPtyId: string | null = null - let startupShellReady = !options.waitForShellReady - const markerScan = options.waitForShellReady ? createShellReadyMarkerScanState() : null + let waitForShellReady = options.waitForShellReady + let startupShellReady = !waitForShellReady + // Why tracked apart from `startupShellReady`: only an observed marker proves the + // host wrapped the shell and armed bracketed paste. A fallback release means the + // host shell never published one, so the raw submit is the only safe form. + let markerObserved = false + let markerScan = waitForShellReady ? createShellReadyMarkerScanState() : null let injectTimer: ReturnType | null = null let fallbackTimer: ReturnType | null = null let sawOutput = false @@ -53,6 +90,7 @@ export function createSshBackgroundStartupDelivery( return } startupShellReady = true + markerObserved = true clearFallbackTimer() if (pendingCommand && lastPtyId) { schedule(lastPtyId) @@ -66,7 +104,7 @@ export function createSshBackgroundStartupDelivery( } // The long budget only buys time for the shell-ready marker; the fast path // pastes nothing prompt-sensitive, so delaying it there is pure latency. - const waitingForSilentShell = options.waitForShellReady && !sawOutput + const waitingForSilentShell = waitForShellReady && !sawOutput fallbackTimer = setTimeout( () => { fallbackTimer = null @@ -100,14 +138,14 @@ export function createSshBackgroundStartupDelivery( // Why: the SSH relay treats spawn.command as metadata for interactive // PTYs; hidden automation tabs still submit the command themselves. // Why bracketed paste: multiline prompts are pasted literally only when we - // synchronized on the Orca shell-ready marker (waitForShellReady) — that - // is the bash/zsh overlay with bracketed-paste mode armed. Submit with CR - // since the relay drives a remote shell. + // synchronized on the Orca shell-ready marker — that is the bash/zsh overlay + // with bracketed-paste mode armed. Submit with CR since the relay drives a + // remote shell. options.write( ptyId, buildStartupCommandSubmission(command, { submit: '\r', - bracketedPasteSafe: options.waitForShellReady + bracketedPasteSafe: markerObserved }) ) }, 50) @@ -135,6 +173,19 @@ export function createSshBackgroundStartupDelivery( }, armFallback, schedule, + applyHostShellReadyArmed(armed) { + if (armed !== false || !waitForShellReady) { + return + } + // Not a marker sighting: bracketed paste stays unproven, so the submit stays raw. + waitForShellReady = false + startupShellReady = true + markerScan = null + clearFallbackTimer() + if (lastPtyId) { + schedule(lastPtyId) + } + }, clear() { clearInjectTimer() clearFallbackTimer() diff --git a/src/renderer/src/lib/structured-agent-session-launch-callers.ts b/src/renderer/src/lib/structured-agent-session-launch-callers.ts index 7097b085cba..db9715c0c1a 100644 --- a/src/renderer/src/lib/structured-agent-session-launch-callers.ts +++ b/src/renderer/src/lib/structured-agent-session-launch-callers.ts @@ -1,6 +1,6 @@ -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' import { - settleStructuredCodexLaunchPrompt, + settleStructuredAgentLaunchPrompt, type StructuredPromptDeliveryResult } from '@/lib/structured-agent-session-launch-prompt' import type { StructuredAgentSessionOutboxEntry } from '../../../shared/structured-agent-session-outbox' @@ -10,7 +10,7 @@ export type StructuredRefusalFallback = () => | StructuredPromptDeliveryResult | Promise -export type StructuredCodexLaunchOptions = { +export type StructuredAgentLaunchOptions = { prompt?: string promptDelivery?: 'auto-submit' | 'submit-after-ready' onPromptDelivered?: () => void @@ -137,7 +137,7 @@ function trackPromptDelivery( export function addStructuredLaunchCaller(args: { group: StructuredLaunchCallerGroup launchResult: Promise<{ sessionId: string; fence: number }> - options: StructuredCodexLaunchOptions + options: StructuredAgentLaunchOptions stagedEntry: StructuredAgentSessionOutboxEntry | null }): StructuredLaunchCaller { const fallback = Promise.withResolvers() @@ -156,7 +156,7 @@ export function addStructuredLaunchCaller(args: { } } args.group.entries.add(caller) - const promptDeliveryResult = settleStructuredCodexLaunchPrompt({ + const promptDeliveryResult = settleStructuredAgentLaunchPrompt({ launchResult: args.launchResult, options: args.options, stagedEntry: args.stagedEntry diff --git a/src/renderer/src/lib/structured-agent-session-launch-prompt.ts b/src/renderer/src/lib/structured-agent-session-launch-prompt.ts index 513d3055c73..d52105057cd 100644 --- a/src/renderer/src/lib/structured-agent-session-launch-prompt.ts +++ b/src/renderer/src/lib/structured-agent-session-launch-prompt.ts @@ -78,7 +78,7 @@ async function dispatchStructuredLaunchPrompt( } } -export function settleStructuredCodexLaunchPrompt(args: { +export function settleStructuredAgentLaunchPrompt(args: { launchResult: Promise options: StructuredLaunchPromptOptions stagedEntry: StructuredAgentSessionOutboxEntry | null diff --git a/src/renderer/src/lib/structured-agent-session-launch-recovery.ts b/src/renderer/src/lib/structured-agent-session-launch-recovery.ts index 1c4347281d1..1af33421652 100644 --- a/src/renderer/src/lib/structured-agent-session-launch-recovery.ts +++ b/src/renderer/src/lib/structured-agent-session-launch-recovery.ts @@ -1,18 +1,18 @@ import type { AgentSessionHistoryResult } from '../../../shared/agent-session-wire' import { - launchStructuredCodexSession, + launchStructuredAgentSession, StructuredAgentSessionCreateRefusalError, type StructuredAgentSessionLaunchIntent -} from '@/lib/launch-structured-codex-session' +} from '@/lib/launch-structured-agent-session' import { callStructuredAgentSession } from '@/runtime/structured-agent-session-client' import { refreshLocalStructuredSessionTabs } from '@/runtime/local-structured-session-tabs-sync' import { useAppStore } from '@/store' -export type StructuredCodexLaunchReceipt = { sessionId: string; fence: number } +export type StructuredAgentLaunchReceipt = { sessionId: string; fence: number } export type StructuredLaunchRecoveryState = { intent: StructuredAgentSessionLaunchIntent - promise: Promise + promise: Promise visibilityUnknown: boolean cancelled: boolean onVisibilityChanged?: () => void @@ -64,7 +64,7 @@ function hasAdoptedStructuredSession(intent: StructuredAgentSessionLaunchIntent) async function recoverPublishedSessionReceipt( state: StructuredLaunchRecoveryState -): Promise { +): Promise { await verifyPublishedSession(state) const history = await callStructuredAgentSession( { kind: 'local' }, @@ -82,10 +82,10 @@ async function recoverPublishedSessionReceipt( async function retrySameIntent( state: StructuredLaunchRecoveryState, priorError: unknown -): Promise { +): Promise { throwIfLaunchCancelled(state) try { - const receipt = await launchStructuredCodexSession(state.intent) + const receipt = await launchStructuredAgentSession(state.intent) throwIfLaunchCancelled(state) await verifyPublishedSession(state) return receipt @@ -111,11 +111,11 @@ async function retrySameIntent( export async function launchAndReconcile( state: StructuredLaunchRecoveryState -): Promise { +): Promise { throwIfLaunchCancelled(state) - let receipt: StructuredCodexLaunchReceipt + let receipt: StructuredAgentLaunchReceipt try { - receipt = await launchStructuredCodexSession(state.intent) + receipt = await launchStructuredAgentSession(state.intent) } catch (error) { if (state.cancelled) { throw new StructuredAgentSessionLaunchCancelledError() @@ -143,7 +143,7 @@ export async function launchAndReconcile( export async function reconcileUnknownLaunch( state: StructuredLaunchRecoveryState -): Promise { +): Promise { throwIfLaunchCancelled(state) state.visibilityUnknown = false state.onVisibilityChanged?.() diff --git a/src/renderer/src/lib/structured-agent-session-launch.test.ts b/src/renderer/src/lib/structured-agent-session-launch.test.ts index 812248e7a49..61d24b248d2 100644 --- a/src/renderer/src/lib/structured-agent-session-launch.test.ts +++ b/src/renderer/src/lib/structured-agent-session-launch.test.ts @@ -20,12 +20,12 @@ vi.mock('sonner', () => ({ } })) -vi.mock('@/lib/launch-structured-codex-session', () => { +vi.mock('@/lib/launch-structured-agent-session', () => { class StructuredAgentSessionCreateRefusalError extends Error {} return { - createStructuredCodexSessionLaunchIntent: mocks.createIntent, + createStructuredAgentSessionLaunchIntent: mocks.createIntent, abandonStructuredAgentSessionLaunchIntent: mocks.abandonIntent, - launchStructuredCodexSession: mocks.launch, + launchStructuredAgentSession: mocks.launch, StructuredAgentSessionCreateRefusalError } }) @@ -51,17 +51,25 @@ vi.mock('@/store', () => ({ })) vi.mock('@/i18n/i18n', () => ({ - translate: (_key: string, fallback: string) => fallback + translate: (_key: string, fallback: string, options?: { value0?: string }) => + fallback.replace('{{value0}}', options?.value0 ?? '') +})) + +vi.mock('@/lib/agent-catalog', () => ({ + getAgentCatalog: () => [ + { id: 'claude', label: 'Claude' }, + { id: 'codex', label: 'Codex' } + ] })) import { StructuredAgentSessionCreateRefusalError, type StructuredAgentSessionLaunchIntent -} from '@/lib/launch-structured-codex-session' +} from '@/lib/launch-structured-agent-session' import { refreshLocalStructuredSessionTabs } from '@/runtime/local-structured-session-tabs-sync' import { - cancelStructuredCodexLaunch, - startStructuredCodexLaunch + cancelStructuredAgentLaunch, + startStructuredAgentLaunch } from './structured-agent-session-launch' import { readOutbox } from '@/components/native-chat/structured-agent-session-outbox-storage' @@ -72,6 +80,7 @@ function launchIntent( return { worktreeId, sessionId, + agent: 'codex', params: { envelope: { sessionId, @@ -112,13 +121,16 @@ async function flushLaunchSettlement(): Promise { } } -describe('startStructuredCodexLaunch', () => { +describe('startStructuredAgentLaunch', () => { beforeEach(() => { vi.clearAllMocks() localStorage.clear() mocks.rendererTabs = {} mocks.listeners.clear() - mocks.createIntent.mockImplementation((worktreeId: string) => launchIntent(worktreeId)) + mocks.createIntent.mockImplementation((worktreeId: string, agent: 'claude' | 'codex') => { + const intent = launchIntent(worktreeId, `${agent}-session-${worktreeId}`) + return { ...intent, agent, params: { ...intent.params, agent } } + }) mocks.callStructuredAgentSession.mockResolvedValue({ ok: true, page: { fence: 1 } @@ -138,7 +150,7 @@ describe('startStructuredCodexLaunch', () => { value: { submission: { dispatchState: 'accepted' } } }) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(mocks.launch).toHaveBeenCalledOnce() @@ -147,6 +159,42 @@ describe('startStructuredCodexLaunch', () => { expect(toast.error).not.toHaveBeenCalled() }) + it('keeps a Claude and a Codex launch in the same worktree apart', async () => { + const worktreeId = 'wt-two-agents' + mocks.launch.mockImplementation(async (intent: StructuredAgentSessionLaunchIntent) => { + mocks.rendererTabs[worktreeId] = [ + ...(mocks.rendererTabs[worktreeId] ?? []), + { contentType: 'agent-session', entityId: intent.sessionId, worktreeId } + ] + return { sessionId: intent.sessionId, fence: 1 } + }) + + const claude = startStructuredAgentLaunch(worktreeId, 'claude') + const codex = startStructuredAgentLaunch(worktreeId, 'codex') + await flushLaunchSettlement() + + expect(mocks.createIntent).toHaveBeenNthCalledWith(1, worktreeId, 'claude') + expect(mocks.createIntent).toHaveBeenNthCalledWith(2, worktreeId, 'codex') + expect(mocks.launch).toHaveBeenCalledTimes(2) + expect(vi.mocked(mocks.launch).mock.calls.map(([intent]) => intent.params.agent)).toEqual([ + 'claude', + 'codex' + ]) + expect(claude.sessionId).not.toBe(codex.sessionId) + expect(toast.error).not.toHaveBeenCalled() + }) + + it('names the refused agent in the launch failure toast', async () => { + const worktreeId = 'wt-claude-toast' + mocks.launch.mockRejectedValue(new Error('boom')) + vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([]) + + startStructuredAgentLaunch(worktreeId, 'claude') + await flushLaunchSettlement() + + expect(toast.error).toHaveBeenCalledWith('Could not open Claude chat', expect.anything()) + }) + it('completes from the host-emitted projection without listing inventory', async () => { const worktreeId = 'wt-host-frame' const intent = launchIntent(worktreeId, 'session-host-frame') @@ -161,7 +209,7 @@ describe('startStructuredCodexLaunch', () => { return { sessionId: intent.sessionId, fence: 1 } }) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(mocks.launch).toHaveBeenCalledOnce() @@ -186,8 +234,8 @@ describe('startStructuredCodexLaunch', () => { value: { submission: { dispatchState: 'accepted' } } }) - startStructuredCodexLaunch(worktreeId) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') + startStructuredAgentLaunch(worktreeId, 'codex') expect(mocks.createIntent).toHaveBeenCalledOnce() expect(mocks.launch).toHaveBeenCalledOnce() @@ -213,8 +261,8 @@ describe('startStructuredCodexLaunch', () => { ok: true, value: { submission: { dispatchState: 'accepted' } } }) - startStructuredCodexLaunch(worktreeId) - const second = startStructuredCodexLaunch(worktreeId, { prompt: 'second prompt' }) + startStructuredAgentLaunch(worktreeId, 'codex') + const second = startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'second prompt' }) resolveLaunch({ sessionId: intent.sessionId, fence: 1 }) await expect(second.promptDeliveryResult).resolves.toEqual({ @@ -251,12 +299,12 @@ describe('startStructuredCodexLaunch', () => { () => new Promise((resolve) => (resolveDelivery = resolve)) ) - startStructuredCodexLaunch(worktreeId) - const coalesced = startStructuredCodexLaunch(worktreeId, { prompt: 'second prompt' }) + startStructuredAgentLaunch(worktreeId, 'codex') + const coalesced = startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'second prompt' }) resolveLaunch({ sessionId: intent.sessionId, fence: 1 }) await vi.waitFor(() => expect(mocks.callStructuredAgentSession).toHaveBeenCalledOnce()) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') expect(mocks.createIntent).toHaveBeenCalledOnce() expect(mocks.launch).toHaveBeenCalledOnce() @@ -274,9 +322,9 @@ describe('startStructuredCodexLaunch', () => { mocks.launch.mockRejectedValue(new Error('offline')) vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([]) - startStructuredCodexLaunch(worktreeId, { prompt: 'first prompt' }) + startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'first prompt' }) await flushLaunchSettlement() - startStructuredCodexLaunch(worktreeId, { prompt: 'second prompt' }) + startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'second prompt' }) await flushLaunchSettlement() expect(mocks.createIntent).toHaveBeenCalledOnce() @@ -291,7 +339,7 @@ describe('startStructuredCodexLaunch', () => { publishedSnapshot(worktreeId, intent.sessionId) ]) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(mocks.createIntent).toHaveBeenCalledOnce() @@ -299,6 +347,47 @@ describe('startStructuredCodexLaunch', () => { expect(toast.error).not.toHaveBeenCalled() }) + it('does not claim a terminal opened when a definitive refusal fallback only settled', async () => { + const worktreeId = 'wt-refused-fallback-toast' + const intent = launchIntent(worktreeId) + const fallback = vi.fn().mockResolvedValue({ delivered: false, failureNotified: true }) + mocks.createIntent.mockReturnValueOnce(intent) + mocks.launch.mockRejectedValue(new StructuredAgentSessionCreateRefusalError('refused')) + + const launch = startStructuredAgentLaunch(worktreeId, 'codex') + void launch.claimDefinitiveRefusalFallback(fallback) + await expect(launch.launchResult).rejects.toBeInstanceOf( + StructuredAgentSessionCreateRefusalError + ) + await flushLaunchSettlement() + + expect(toast.error).not.toHaveBeenCalled() + expect(toast.message).toHaveBeenCalledWith( + "Structured chat isn't available", + expect.objectContaining({ + description: 'Orca tried to open a Codex terminal instead.' + }) + ) + }) + + it('keeps the raw error out of the failure toast', async () => { + const worktreeId = 'wt-no-raw-error-in-toast' + const intent = launchIntent(worktreeId) + mocks.createIntent.mockReturnValueOnce(intent) + mocks.launch.mockRejectedValue( + new Error("EEXIST: file already exists, mkdir '/tmp/o97b/agent-sessions'") + ) + vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([]) + + startStructuredAgentLaunch(worktreeId, 'codex') + await flushLaunchSettlement() + + expect(toast.error).toHaveBeenCalledOnce() + const description = String(vi.mocked(toast.error).mock.calls[0]?.[1]?.description ?? '') + expect(description).not.toContain('EEXIST') + expect(description).not.toContain('/tmp/') + }) + it('retries an absent unknown outcome with the exact same intent', async () => { const worktreeId = 'wt-same-envelope-retry' const intent = launchIntent(worktreeId) @@ -310,7 +399,7 @@ describe('startStructuredCodexLaunch', () => { .mockResolvedValueOnce([]) .mockResolvedValueOnce([publishedSnapshot(worktreeId, intent.sessionId)]) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(mocks.launch).toHaveBeenCalledTimes(2) @@ -327,14 +416,14 @@ describe('startStructuredCodexLaunch', () => { mocks.launch.mockRejectedValue(new Error('offline')) vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([]) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(toast.error).toHaveBeenCalledOnce() vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([ publishedSnapshot(worktreeId, intent.sessionId) ]) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(mocks.createIntent).toHaveBeenCalledOnce() @@ -357,7 +446,7 @@ describe('startStructuredCodexLaunch', () => { }) vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValueOnce([]).mockResolvedValueOnce([]) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(mocks.createIntent).toHaveBeenCalledOnce() @@ -374,7 +463,7 @@ describe('startStructuredCodexLaunch', () => { mocks.launch.mockRejectedValue(new Error('offline')) vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([]) - const first = startStructuredCodexLaunch(worktreeId, { prompt: 'only once' }) + const first = startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'only once' }) const firstFallbackResult = first.claimDefinitiveRefusalFallback(firstFallback) await expect(first.launchResult).rejects.toThrow('offline') expect(first.releaseCallerAfterUnknownOutcome()).toBe(true) @@ -382,7 +471,7 @@ describe('startStructuredCodexLaunch', () => { vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([ publishedSnapshot(worktreeId, intent.sessionId) ]) - const retry = startStructuredCodexLaunch(worktreeId) + const retry = startStructuredAgentLaunch(worktreeId, 'codex') await expect(retry.launchResult).resolves.toEqual({ sessionId: intent.sessionId, fence: 1 }) await expect(firstFallbackResult).resolves.toBe(false) @@ -408,7 +497,7 @@ describe('startStructuredCodexLaunch', () => { mocks.launch.mockRejectedValue(new Error('offline')) vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([]) - const first = startStructuredCodexLaunch(worktreeId) + const first = startStructuredAgentLaunch(worktreeId, 'codex') const firstFallbackResult = first.claimDefinitiveRefusalFallback(firstFallback) await expect(first.launchResult).rejects.toThrow('offline') expect(first.releaseCallerAfterUnknownOutcome()).toBe(true) @@ -416,7 +505,7 @@ describe('startStructuredCodexLaunch', () => { mocks.launch.mockRejectedValueOnce( new StructuredAgentSessionCreateRefusalError('structured launch disabled') ) - const retry = startStructuredCodexLaunch(worktreeId) + const retry = startStructuredAgentLaunch(worktreeId, 'codex') const retryFallbackResult = retry.claimDefinitiveRefusalFallback(retryFallback) await expect(retry.launchResult).rejects.toBeInstanceOf( @@ -428,6 +517,32 @@ describe('startStructuredCodexLaunch', () => { expect(retryFallback).toHaveBeenCalledOnce() }) + it('never starts a sibling fallback for a post-attach unknown refusal', async () => { + const worktreeId = 'wt-post-attach-unknown' + const intent = launchIntent(worktreeId) + const fallback = vi.fn() + mocks.createIntent.mockReturnValueOnce(intent) + mocks.launch.mockRejectedValue( + Object.assign(new Error('The chat may already exist.'), { + code: 'agent_session_operation_unknown' + }) + ) + vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([]) + + const launch = startStructuredAgentLaunch(worktreeId, 'codex') + const fallbackResult = launch.claimDefinitiveRefusalFallback(fallback) + + await expect(launch.launchResult).rejects.toMatchObject({ + code: 'agent_session_operation_unknown' + }) + expect(launch.isVisibilityUnknown()).toBe(true) + expect(launch.releaseCallerAfterUnknownOutcome()).toBe(true) + await expect(fallbackResult).resolves.toBe(false) + expect(fallback).not.toHaveBeenCalled() + expect(mocks.createIntent).toHaveBeenCalledOnce() + expect(mocks.launch).toHaveBeenCalledTimes(2) + }) + it('releases a definitively refused intent so a new click can create a new identity', async () => { const worktreeId = 'wt-refused' const first = launchIntent(worktreeId, 'session-first') @@ -440,9 +555,9 @@ describe('startStructuredCodexLaunch', () => { publishedSnapshot(worktreeId, second.sessionId) ]) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(mocks.createIntent).toHaveBeenCalledTimes(2) @@ -460,7 +575,7 @@ describe('startStructuredCodexLaunch', () => { throw new Error('storage unavailable') }) - const result = startStructuredCodexLaunch(worktreeId, { prompt: 'start this task' }) + const result = startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'start this task' }) const fallbackResult = result.claimDefinitiveRefusalFallback(fallback) await expect(result.launchResult).rejects.toBeInstanceOf( @@ -481,8 +596,8 @@ describe('startStructuredCodexLaunch', () => { () => new Promise((_resolve, reject) => (rejectLaunch = reject)) ) - const first = startStructuredCodexLaunch(worktreeId, { prompt: 'first prompt' }) - const second = startStructuredCodexLaunch(worktreeId, { prompt: 'second prompt' }) + const first = startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'first prompt' }) + const second = startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'second prompt' }) const firstFallback = vi.fn().mockResolvedValue({ delivered: true, failureNotified: false @@ -525,9 +640,9 @@ describe('startStructuredCodexLaunch', () => { () => new Promise((resolve) => (resolveRefresh = resolve)) ) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await vi.waitFor(() => expect(refreshLocalStructuredSessionTabs).toHaveBeenCalledOnce()) - expect(cancelStructuredCodexLaunch(worktreeId, intent.sessionId)).toBe(true) + expect(cancelStructuredAgentLaunch(worktreeId, intent.sessionId)).toBe(true) resolveRefresh([]) await flushLaunchSettlement() @@ -546,12 +661,12 @@ describe('startStructuredCodexLaunch', () => { () => new Promise((resolve) => (resolveRefresh = resolve)) ) - startStructuredCodexLaunch(worktreeId, { prompt: 'first prompt' }) - startStructuredCodexLaunch(worktreeId, { prompt: 'second prompt' }) + startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'first prompt' }) + startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'second prompt' }) await vi.waitFor(() => expect(refreshLocalStructuredSessionTabs).toHaveBeenCalledOnce()) expect(readOutbox(intent.sessionId)).toHaveLength(2) - expect(cancelStructuredCodexLaunch(worktreeId, intent.sessionId)).toBe(true) + expect(cancelStructuredAgentLaunch(worktreeId, intent.sessionId)).toBe(true) expect(readOutbox(intent.sessionId)).toEqual([]) resolveRefresh([]) await flushLaunchSettlement() @@ -569,9 +684,9 @@ describe('startStructuredCodexLaunch', () => { .mockResolvedValueOnce([]) .mockImplementationOnce(() => new Promise((resolve) => (resolveRetryRefresh = resolve))) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await vi.waitFor(() => expect(refreshLocalStructuredSessionTabs).toHaveBeenCalledTimes(2)) - expect(cancelStructuredCodexLaunch(worktreeId, intent.sessionId)).toBe(true) + expect(cancelStructuredAgentLaunch(worktreeId, intent.sessionId)).toBe(true) resolveRetryRefresh([]) await flushLaunchSettlement() diff --git a/src/renderer/src/lib/structured-agent-session-launch.ts b/src/renderer/src/lib/structured-agent-session-launch.ts index e5e3083e8fd..2a97f6acb36 100644 --- a/src/renderer/src/lib/structured-agent-session-launch.ts +++ b/src/renderer/src/lib/structured-agent-session-launch.ts @@ -1,11 +1,13 @@ import { useSyncExternalStore } from 'react' import { toast } from 'sonner' +import type { AgentSessionHandleProvider } from '../../../shared/agent-session-provider-handle' +import { getAgentCatalog } from '@/lib/agent-catalog' import { translate } from '@/i18n/i18n' import { abandonStructuredAgentSessionLaunchIntent, - createStructuredCodexSessionLaunchIntent, + createStructuredAgentSessionLaunchIntent, StructuredAgentSessionCreateRefusalError -} from '@/lib/launch-structured-codex-session' +} from '@/lib/launch-structured-agent-session' import { discardStructuredAgentSessionLaunchOutbox, enqueueStructuredAgentSessionLaunchPrompt @@ -14,7 +16,7 @@ import { launchAndReconcile, reconcileUnknownLaunch, StructuredAgentSessionLaunchCancelledError, - type StructuredCodexLaunchReceipt, + type StructuredAgentLaunchReceipt, type StructuredLaunchRecoveryState } from '@/lib/structured-agent-session-launch-recovery' import type { StructuredPromptDeliveryResult } from '@/lib/structured-agent-session-launch-prompt' @@ -26,13 +28,13 @@ import { settleStructuredLaunchCallersWithFallback, settleStructuredLaunchCallersWithoutFallback, structuredLaunchCallersHavePendingWork, - type StructuredCodexLaunchOptions, + type StructuredAgentLaunchOptions, type StructuredLaunchCaller, type StructuredLaunchCallerGroup, type StructuredRefusalFallback } from '@/lib/structured-agent-session-launch-callers' -export type { StructuredCodexLaunchOptions, StructuredCodexLaunchReceipt } +export type { StructuredAgentLaunchOptions, StructuredAgentLaunchReceipt } type StructuredLaunchState = StructuredLaunchRecoveryState & { identity: string @@ -44,16 +46,20 @@ type StructuredLaunchStateResult = { caller: StructuredLaunchCaller } -export type StructuredCodexLaunchResult = { +export type StructuredAgentLaunchResult = { sessionId: string - launchResult: Promise + launchResult: Promise promptDeliveryResult?: Promise isVisibilityUnknown: () => boolean releaseCallerAfterUnknownOutcome: () => boolean claimDefinitiveRefusalFallback: (fallback: StructuredRefusalFallback) => Promise } -export type StructuredCodexLaunchStatus = 'idle' | 'pending' | 'unknown' +export type StructuredAgentLaunchStatus = 'idle' | 'pending' | 'unknown' + +function structuredAgentLabel(agent: AgentSessionHandleProvider): string { + return getAgentCatalog().find((entry) => entry.id === agent)?.label ?? agent +} const pendingStructuredLaunchesByIdentity = new Map() const structuredLaunchListeners = new Set<() => void>() @@ -64,29 +70,37 @@ function notifyStructuredLaunchListeners(): void { } } -export function subscribeStructuredCodexLaunchStatus(listener: () => void): () => void { +export function subscribeStructuredAgentLaunchStatus(listener: () => void): () => void { structuredLaunchListeners.add(listener) return () => structuredLaunchListeners.delete(listener) } -export function getStructuredCodexLaunchStatus(worktreeId: string): StructuredCodexLaunchStatus { - const state = pendingStructuredLaunchesByIdentity.get(worktreeId) +export function getStructuredAgentLaunchStatus( + worktreeId: string, + agent: AgentSessionHandleProvider +): StructuredAgentLaunchStatus { + const state = pendingStructuredLaunchesByIdentity.get(launchIdentity(worktreeId, agent)) if (!state) { return 'idle' } return state.visibilityUnknown ? 'unknown' : 'pending' } -export function useStructuredCodexLaunchStatus(worktreeId: string): StructuredCodexLaunchStatus { +export function useStructuredAgentLaunchStatus( + worktreeId: string, + agent: AgentSessionHandleProvider +): StructuredAgentLaunchStatus { return useSyncExternalStore( - subscribeStructuredCodexLaunchStatus, - () => getStructuredCodexLaunchStatus(worktreeId), + subscribeStructuredAgentLaunchStatus, + () => getStructuredAgentLaunchStatus(worktreeId, agent), () => 'idle' ) } -function launchIdentity(worktreeId: string): string { - return worktreeId +// Why keyed by agent too: one worktree can hold a Claude and a Codex launch at once, and a shared +// key would hand the second caller the first agent's intent. +function launchIdentity(worktreeId: string, agent: AgentSessionHandleProvider): string { + return `${agent}:${worktreeId}` } function cleanupLaunchState(state: StructuredLaunchState): void { @@ -114,7 +128,7 @@ function settleDefinitiveRefusalFallback(state: StructuredLaunchState): void { function trackLaunchSettlement( state: StructuredLaunchState, - promise: Promise + promise: Promise ): void { void promise.then( () => { @@ -146,27 +160,54 @@ function trackLaunchFailureToast(state: StructuredLaunchState): void { if (error instanceof StructuredAgentSessionLaunchCancelledError) { return } + const agentLabel = structuredAgentLabel(state.intent.agent) if ( error instanceof StructuredAgentSessionCreateRefusalError && (await state.callers.refusalSettlement.promise.catch(() => false)) ) { + // Why: the callback proves the fallback was attempted, not that its terminal became visible. + toast.message( + translate( + 'components.native-chat.structuredSessionFellBackToTerminal', + "Structured chat isn't available" + ), + { + description: translate( + 'components.native-chat.structuredSessionFellBackToTerminalDescription', + 'Orca tried to open a {{value0}} terminal instead.', + { value0: agentLabel } + ) + } + ) return } + // Why: the raw error carries errnos and absolute paths; it belongs in the log, not the toast. + console.warn('[native-chat] structured launch failed', error) toast.error( translate( 'components.native-chat.structuredSessionLaunchFailed', - 'Could not open Codex chat' + 'Could not open {{value0}} chat', + { + value0: agentLabel + } ), - { description: error instanceof Error ? error.message : String(error) } + { + description: translate( + 'components.native-chat.structuredSessionLaunchFailedDescription', + 'Orca could not open a structured {{value0}} chat. See the logs for details.', + { value0: agentLabel } + ) + } ) }) } -function structuredCodexLaunchState( +function structuredAgentLaunchState( worktreeId: string, - options: StructuredCodexLaunchOptions + agent: AgentSessionHandleProvider, + options: StructuredAgentLaunchOptions ): StructuredLaunchStateResult { - const identity = launchIdentity(worktreeId) + const identity = launchIdentity(worktreeId, agent) const existing = pendingStructuredLaunchesByIdentity.get(identity) if (existing) { if (existing.visibilityUnknown) { @@ -192,7 +233,7 @@ function structuredCodexLaunchState( } } - const intent = createStructuredCodexSessionLaunchIntent(worktreeId) + const intent = createStructuredAgentSessionLaunchIntent(worktreeId, agent) const text = options.prompt?.trim() ?? '' const stagedPrompt = text ? enqueueStructuredAgentSessionLaunchPrompt(intent.sessionId, text) @@ -212,7 +253,7 @@ function structuredCodexLaunchState( text && !stagedPrompt ? Promise.reject( new StructuredAgentSessionCreateRefusalError( - 'Could not durably stage the Codex launch prompt.' + `Could not durably stage the ${structuredAgentLabel(agent)} launch prompt.` ) ) : launchAndReconcile(state) @@ -232,7 +273,7 @@ function structuredCodexLaunchState( } } -export function cancelStructuredCodexLaunch(worktreeId: string, sessionId: string): boolean { +export function cancelStructuredAgentLaunch(worktreeId: string, sessionId: string): boolean { const state = [...pendingStructuredLaunchesByIdentity.values()].find( (candidate) => candidate.intent.worktreeId === worktreeId && candidate.intent.sessionId === sessionId @@ -249,11 +290,12 @@ export function cancelStructuredCodexLaunch(worktreeId: string, sessionId: strin return true } -export function startStructuredCodexLaunch( +export function startStructuredAgentLaunch( worktreeId: string, - options: StructuredCodexLaunchOptions = {} -): StructuredCodexLaunchResult { - const { state, caller } = structuredCodexLaunchState(worktreeId, options) + agent: AgentSessionHandleProvider, + options: StructuredAgentLaunchOptions = {} +): StructuredAgentLaunchResult { + const { state, caller } = structuredAgentLaunchState(worktreeId, agent, options) return { sessionId: state.intent.sessionId, launchResult: state.promise, diff --git a/src/renderer/src/lib/terminal-links.test.ts b/src/renderer/src/lib/terminal-links.test.ts index 66ff8d870a8..d16e3d3b0cc 100644 --- a/src/renderer/src/lib/terminal-links.test.ts +++ b/src/renderer/src/lib/terminal-links.test.ts @@ -115,6 +115,15 @@ describe('terminal path helpers', () => { }) describe('extractTerminalFileLinks local path tokens', () => { + it('keeps Unicode path segments in the detected range', () => { + expect(extractTerminalFileLinks('/tmp/报告.html').map((link) => link.displayText)).toEqual([ + '/tmp/报告.html' + ]) + expect( + extractTerminalFileLinks('docs/café/report.pdf').map((link) => link.displayText) + ).toEqual(['docs/café/report.pdf']) + }) + it('detects tilde-prefixed POSIX paths', () => { const links = extractTerminalFileLinks('~/Documents/Path/file_name') expect(links).toHaveLength(1) @@ -215,6 +224,15 @@ describe('terminal path helpers', () => { expect(links).toHaveLength(20_000) expect(links[0].pathText).toBe('/tmp/Foo Bar/file') }, 5_000) + + it('keeps extension-heavy assistant prose on a bounded scan path', () => { + const filenames = Array.from({ length: 20_000 }, (_, index) => `report-${index}.txt`) + const links = extractTerminalFileLinks(`/tmp/root/${filenames.join(' ')}`) + + expect(links).toHaveLength(2) + expect(links[0].pathText).toBe(`/tmp/root/${filenames.slice(0, -1).join(' ')}`) + expect(links.at(-1)?.pathText).toBe('report-19999.txt') + }) }) it('supports Windows cwd resolution for terminal file links', () => { diff --git a/src/renderer/src/lib/terminal-links.ts b/src/renderer/src/lib/terminal-links.ts index 729f2d839e7..970c6d9208c 100644 --- a/src/renderer/src/lib/terminal-links.ts +++ b/src/renderer/src/lib/terminal-links.ts @@ -34,7 +34,7 @@ export type ResolvedTerminalFileLink = Pick 1) { + while (nextPathStart && nextPathStart.index + nextPathStart[0].length <= end) { + pathStartCount += 1 + nextPathStart = pathStartPattern.exec(range.text) + } + if (pathStartCount > 1) { continue } if ( @@ -150,15 +157,6 @@ function trimSpacedPathTrailingProse( } } -function countPathStarts(text: string): number { - let count = 0 - for (const match of text.matchAll(/(?:^|\s)(?:~[\\/]|[\\/]|\.{1,2}[\\/]|[A-Za-z]:[\\/])/g)) { - void match - count += 1 - } - return count -} - function trimTrailingWhitespace( range: DetectedTerminalFileLinkRange ): DetectedTerminalFileLinkRange { diff --git a/src/renderer/src/lib/worktree-creation-flow-execute.ts b/src/renderer/src/lib/worktree-creation-flow-execute.ts index b6c48da719c..9b6ba569b1e 100644 --- a/src/renderer/src/lib/worktree-creation-flow-execute.ts +++ b/src/renderer/src/lib/worktree-creation-flow-execute.ts @@ -13,6 +13,7 @@ import { formatWorkspaceCreateError, getWorkspaceCreateErrorToastMessage } from '@/lib/workspace-create-error-format' +import { isAgentSessionHandleProvider } from '../../../shared/agent-session-provider-handle' import type { CreateWorktreeResult } from '../../../shared/worktree/create-types' import type { WorktreeCreationRequest } from '@/lib/pending-worktree-creation' import { createBrowserUuid } from '@/lib/browser-uuid' @@ -208,7 +209,7 @@ export async function executeWorktreeCreation( } let structuredLaunchAccepted = structuredLaunch - if (structuredLaunch && preparedRequest.agent === 'codex') { + if (structuredLaunch && isAgentSessionHandleProvider(preparedRequest.agent)) { const structuredSession = await launchStructuredWorktreeSession({ creationId, request: preparedRequest, diff --git a/src/renderer/src/lib/worktree-creation-structured-session.test.ts b/src/renderer/src/lib/worktree-creation-structured-session.test.ts index 0fc91c18393..c58fcd40dce 100644 --- a/src/renderer/src/lib/worktree-creation-structured-session.test.ts +++ b/src/renderer/src/lib/worktree-creation-structured-session.test.ts @@ -6,8 +6,8 @@ const mocks = vi.hoisted(() => ({ }, listener: null as ((state: { pendingWorktreeCreations: Record }) => void) | null, unsubscribe: vi.fn(), - startStructuredCodexLaunch: vi.fn(), - cancelStructuredCodexLaunch: vi.fn(), + startStructuredAgentLaunch: vi.fn(), + cancelStructuredAgentLaunch: vi.fn(), closeStructuredAgentSession: vi.fn(), callRuntimeRpc: vi.fn(), activateStructuredAgentSessionById: vi.fn() @@ -26,8 +26,8 @@ vi.mock('@/store', () => ({ })) vi.mock('@/lib/structured-agent-session-launch', () => ({ - startStructuredCodexLaunch: mocks.startStructuredCodexLaunch, - cancelStructuredCodexLaunch: mocks.cancelStructuredCodexLaunch + startStructuredAgentLaunch: mocks.startStructuredAgentLaunch, + cancelStructuredAgentLaunch: mocks.cancelStructuredAgentLaunch })) vi.mock('@/runtime/structured-agent-session-close', () => ({ @@ -58,7 +58,7 @@ vi.mock('@/lib/agent-trust-preflight', () => ({ preflightAgentTrust: vi.fn() })) -vi.mock('@/lib/launch-structured-codex-session', () => ({ +vi.mock('@/lib/launch-structured-agent-session', () => ({ StructuredAgentSessionCreateRefusalError: class extends Error {} })) @@ -78,7 +78,7 @@ describe('launchStructuredWorktreeSession', () => { const launchResult = new Promise<{ sessionId: string; fence: number }>((resolve) => { resolveLaunch = resolve }) - mocks.startStructuredCodexLaunch.mockReturnValue({ + mocks.startStructuredAgentLaunch.mockReturnValue({ sessionId: 'session-1', launchResult, isVisibilityUnknown: () => false, @@ -117,7 +117,7 @@ describe('launchStructuredWorktreeSession', () => { activation: false, primaryTabId: null }) - expect(mocks.cancelStructuredCodexLaunch).toHaveBeenCalledWith('worktree-1', 'session-1') + expect(mocks.cancelStructuredAgentLaunch).toHaveBeenCalledWith('worktree-1', 'session-1') expect(mocks.closeStructuredAgentSession).toHaveBeenCalledWith({ kind: 'local' }, 'session-1') expect(mocks.callRuntimeRpc).toHaveBeenCalledWith({ kind: 'local' }, 'session.tabs.close', { worktree: { id: 'worktree-1' }, @@ -130,7 +130,7 @@ describe('launchStructuredWorktreeSession', () => { it('reports an unknown launch without claiming a visible surface', async () => { const releaseCallerAfterUnknownOutcome = vi.fn() - mocks.startStructuredCodexLaunch.mockReturnValue({ + mocks.startStructuredAgentLaunch.mockReturnValue({ sessionId: 'session-unknown', launchResult: Promise.reject(new Error('connection lost')), isVisibilityUnknown: () => true, diff --git a/src/renderer/src/lib/worktree-creation-structured-session.ts b/src/renderer/src/lib/worktree-creation-structured-session.ts index d163852e3ac..2e2cd07698a 100644 --- a/src/renderer/src/lib/worktree-creation-structured-session.ts +++ b/src/renderer/src/lib/worktree-creation-structured-session.ts @@ -2,10 +2,11 @@ import { useAppStore } from '@/store' import { ensureWorktreeHasInitialTerminal } from '@/lib/worktree-initial-terminal-seeding' import { activateAndRevealWorktree, type ActivateAndRevealResult } from '@/lib/worktree-activation' import { - cancelStructuredCodexLaunch, - startStructuredCodexLaunch + cancelStructuredAgentLaunch, + startStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' +import { isAgentSessionHandleProvider } from '../../../shared/agent-session-provider-handle' import { activateStructuredAgentSessionById } from '@/lib/structured-agent-session-tab-activation' import { preflightAgentTrust } from '@/lib/agent-trust-preflight' import type { WorktreeCreationRequest } from '@/lib/pending-worktree-creation' @@ -48,15 +49,17 @@ export async function launchStructuredWorktreeSession(args: { let { activation, primaryTabId } = args let accepted = true let visibilityUnknown = false - if (args.request.agent !== 'codex') { + const agent = args.request.agent + if (!isAgentSessionHandleProvider(agent)) { return { accepted, cancelled: false, visibilityUnknown, activation, primaryTabId } } if (!useAppStore.getState().pendingWorktreeCreations[args.creationId]) { return { accepted, cancelled: true, visibilityUnknown, activation, primaryTabId } } - const launch = startStructuredCodexLaunch( + const launch = startStructuredAgentLaunch( args.worktreeId, + agent, args.recoverUnknownLaunch ? {} : { prompt: args.request.launchDraftPrompt ?? args.request.quickPrompt } @@ -67,7 +70,7 @@ export async function launchStructuredWorktreeSession(args: { return } cancelled = true - cancelStructuredCodexLaunch(args.worktreeId, launch.sessionId) + cancelStructuredAgentLaunch(args.worktreeId, launch.sessionId) } const unsubscribe = useAppStore.subscribe((state) => { if (!state.pendingWorktreeCreations[args.creationId]) { diff --git a/src/renderer/src/renderer-node-builtin-boundary.test.ts b/src/renderer/src/renderer-node-builtin-boundary.test.ts new file mode 100644 index 00000000000..1cc28bd73af --- /dev/null +++ b/src/renderer/src/renderer-node-builtin-boundary.test.ts @@ -0,0 +1,111 @@ +import { existsSync, readFileSync } from 'node:fs' +import path from 'node:path' +import { describe, expect, it } from 'vitest' + +/** + * The renderer runs sandboxed with contextIsolation: `node:*` builtins do not resolve and even a + * bare `process` read throws. A module that reaches one is not a degraded feature — the chunk + * fails at evaluation, React never mounts, and the window stays blank with `workspaceSessionReady` + * stuck false (#18742 did exactly this by importing one constant out of a `node:child_process` + * module). Bundling hides it: the offending module can sit in a shared chunk far from the import + * that pulled it in. + * + * So walk the import graph from every renderer entry — lazy routes included, since a `node:` + * builtin behind one is just a blank route instead of a blank app — and refuse any builtin. + */ +const RENDERER_SRC = import.meta.dirname +const REPO_SRC = path.resolve(RENDERER_SRC, '../..') +const ENTRIES = ['main.tsx', 'popout.tsx', 'web/main.tsx'] +const EXTENSIONS = ['.ts', '.tsx', '.js', '.jsx'] + +/** `import`/`export ... from` and `import(...)` specifiers, minus type-only ones, which erase. */ +function collectValueImportSpecifiers(source: string): string[] { + const specifiers: string[] = [] + const pattern = + /(?:^|[\s;}])(?:import|export)(\s+type\s|\s*\{[^}]*\}|[^'"]*?)?\s*from\s*['"]([^'"]+)['"]|(?:^|[\s;}])import\s*['"]([^'"]+)['"]|import\s*\(\s*['"]([^'"]+)['"]\s*\)/g + for (const match of source.matchAll(pattern)) { + const clause = match[1] ?? '' + const specifier = match[2] ?? match[3] ?? match[4] + if (!specifier || /^\s*type\s/.test(clause)) { + continue + } + // A brace clause whose every binding is `type`-prefixed also erases entirely. + const bindings = clause.trim().startsWith('{') ? clause.trim().slice(1, -1).split(',') : null + if (bindings && bindings.some((b) => b.trim()) && bindings.every((b) => /^\s*type\s/.test(b))) { + continue + } + specifiers.push(specifier) + } + return specifiers +} + +function resolveModule(specifier: string, fromFile: string): string | null { + let base: string + if (specifier.startsWith('@renderer/')) { + base = path.join(RENDERER_SRC, specifier.slice('@renderer/'.length)) + } else if (specifier.startsWith('@/')) { + base = path.join(RENDERER_SRC, specifier.slice(2)) + } else if (specifier.startsWith('.')) { + base = path.resolve(path.dirname(fromFile), specifier) + } else { + // Bare package specifiers are npm dependencies, not first-party source. + return null + } + for (const candidate of [ + ...EXTENSIONS.map((ext) => `${base}${ext}`), + ...EXTENSIONS.map((ext) => path.join(base, `index${ext}`)) + ]) { + if (existsSync(candidate)) { + return candidate + } + } + return null +} + +function walkRendererImportGraph(): Map { + /** file -> the chain of first-party importers that reached it, entry first. */ + const pathToFile = new Map() + const queue: string[] = [] + for (const entry of ENTRIES) { + const file = path.join(RENDERER_SRC, entry) + pathToFile.set(file, [file]) + queue.push(file) + } + while (queue.length > 0) { + const file = queue.shift() as string + const chain = pathToFile.get(file) as string[] + for (const specifier of collectValueImportSpecifiers(readFileSync(file, 'utf8'))) { + const resolved = resolveModule(specifier, file) + if (!resolved || pathToFile.has(resolved)) { + continue + } + pathToFile.set(resolved, [...chain, resolved]) + queue.push(resolved) + } + } + return pathToFile +} + +describe('renderer node-builtin boundary', () => { + it('reaches no module that imports a node: builtin', () => { + const graph = walkRendererImportGraph() + const offenders: string[] = [] + for (const [file, chain] of graph) { + const builtins = collectValueImportSpecifiers(readFileSync(file, 'utf8')).filter( + (specifier) => specifier.startsWith('node:') + ) + if (builtins.length === 0) { + continue + } + const relativeChain = chain.map((step) => path.relative(REPO_SRC, step)).join('\n -> ') + offenders.push(`${builtins.join(', ')} via\n ${relativeChain}`) + } + expect(offenders.join('\n\n')).toBe('') + }) + + it('walks a real graph, so an empty offender list means something', () => { + const graph = walkRendererImportGraph() + expect(graph.size).toBeGreaterThan(3_000) + expect(graph.has(path.join(REPO_SRC, 'shared/process-table-snapshot.ts'))).toBe(true) + }) +}) diff --git a/src/renderer/src/runtime/runtime-terminal-inspection.ts b/src/renderer/src/runtime/runtime-terminal-inspection.ts index 8c354cdc415..b34d75f4555 100644 --- a/src/renderer/src/runtime/runtime-terminal-inspection.ts +++ b/src/renderer/src/runtime/runtime-terminal-inspection.ts @@ -138,7 +138,7 @@ export function recordRuntimeTerminalInputForPtyId(ptyId: string, timestamp = Da export async function inspectRuntimeTerminalProcess( settings: Pick | null | undefined, ptyId: string, - options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } + options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean; steadyState?: boolean } ): Promise { const ownerEnvironmentId = getRemoteRuntimePtyEnvironmentId(ptyId) const target = ownerEnvironmentId diff --git a/src/renderer/src/runtime/structured-agent-session-handoff-store.ts b/src/renderer/src/runtime/structured-agent-session-handoff-store.ts new file mode 100644 index 00000000000..8800f349802 --- /dev/null +++ b/src/renderer/src/runtime/structured-agent-session-handoff-store.ts @@ -0,0 +1,50 @@ +import { useSyncExternalStore } from 'react' +import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' + +export type TerminalStructuredHandoff = { + sessionId: string + fence: number + status: AgentSessionHandoffStatus +} + +const byTerminalTabId = new Map() +const listeners = new Set<() => void>() + +export function publishStructuredHandoff(input: TerminalStructuredHandoff): void { + for (const [tabId, current] of byTerminalTabId) { + if (current.sessionId === input.sessionId && tabId !== input.status.terminal?.tabId) { + byTerminalTabId.delete(tabId) + } + } + if (input.status.terminal) { + byTerminalTabId.set(input.status.terminal.tabId, input) + } + for (const listener of listeners) { + listener() + } +} + +export function clearStructuredHandoff(sessionId: string): void { + let changed = false + for (const [tabId, current] of byTerminalTabId) { + if (current.sessionId === sessionId) { + byTerminalTabId.delete(tabId) + changed = true + } + } + if (changed) { + for (const listener of listeners) { + listener() + } + } +} + +export function useTerminalStructuredHandoff(tabId: string): TerminalStructuredHandoff | null { + return useSyncExternalStore( + (listener) => { + listeners.add(listener) + return () => listeners.delete(listener) + }, + () => byTerminalTabId.get(tabId) ?? null + ) +} diff --git a/src/renderer/src/store/repos/repo-add-actions.ts b/src/renderer/src/store/repos/repo-add-actions.ts index d9aa5d6d968..02d8440dc03 100644 --- a/src/renderer/src/store/repos/repo-add-actions.ts +++ b/src/renderer/src/store/repos/repo-add-actions.ts @@ -3,6 +3,7 @@ import { toast } from 'sonner' import type { AppState } from '../types' import type { Repo } from '../../../../shared/repo-types' import { isGitRepoKind } from '../../../../shared/repo-kind' +import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' import { getRepoHostIdentity } from '../slices/repo-host-identity' import { callRuntimeRpc, getActiveRuntimeTarget } from '../../runtime/runtime-rpc-client' import { resolveDismissedOnboardingFolderAgentLaunch } from '@/lib/onboarding-folder-agent-startup' @@ -202,13 +203,16 @@ export function createRepoAddActions( ...(launch.startup ? { startup: launch.startup } : {}), ...(launch.route === 'structured-native-chat' ? { providesInitialSurface: true } : {}) }) - if (launch.route === 'structured-native-chat' && launch.agent === 'codex') { - const [{ startStructuredCodexLaunch }, { StructuredAgentSessionCreateRefusalError }] = + if ( + launch.route === 'structured-native-chat' && + isAgentSessionHandleProvider(launch.agent) + ) { + const [{ startStructuredAgentLaunch }, { StructuredAgentSessionCreateRefusalError }] = await Promise.all([ import('@/lib/structured-agent-session-launch'), - import('@/lib/launch-structured-codex-session') + import('@/lib/launch-structured-agent-session') ]) - const structured = startStructuredCodexLaunch(folderWorktree.id) + const structured = startStructuredAgentLaunch(folderWorktree.id, launch.agent) const fallback = structured.claimDefinitiveRefusalFallback(() => { activateAndRevealWorktree(folderWorktree.id, { sidebarRevealBehavior: 'auto', diff --git a/src/renderer/src/store/slices/agent-status-provider-session.test.ts b/src/renderer/src/store/slices/agent-status-provider-session.test.ts index f854517d94f..0d60d023956 100644 --- a/src/renderer/src/store/slices/agent-status-provider-session.test.ts +++ b/src/renderer/src/store/slices/agent-status-provider-session.test.ts @@ -16,6 +16,27 @@ function makePiCompatibleProviderSession(agent: 'pi' | 'omp' | 'prime-agent') { } describe('recordAgentProviderSession', () => { + it('does not capture a structured native owner for terminal resume on restart', () => { + const store = createTestStore() + const paneKey = 'structured-tab:leaf-1' + const providerSession = { key: 'session_id' as const, id: 'provider-session-uuid' } + + store + .getState() + .setAgentStatus( + paneKey, + { state: 'working', prompt: 'keep going', agentType: 'claude' }, + 'Claude Chat', + undefined, + { tabId: 'structured-tab', worktreeId: 'wt-1' }, + { providerSession, terminalResumeEligible: false } + ) + store.getState().captureAllSleepingAgentSessions('quit') + + expect(store.getState().agentStatusByPaneKey[paneKey]?.providerSession).toEqual(providerSession) + expect(store.getState().sleepingAgentSessionsByPaneKey[paneKey]).toBeUndefined() + }) + it('preserves the root session while a child permission hook moves Codex to waiting', () => { const store = createTestStore() const providerSession = { key: 'session_id' as const, id: 'root-session' } diff --git a/src/renderer/src/store/slices/worktrees/listing/detected-worktree-host-merge.ts b/src/renderer/src/store/slices/worktrees/listing/detected-worktree-host-merge.ts index 073bd712f32..f8078a9925c 100644 --- a/src/renderer/src/store/slices/worktrees/listing/detected-worktree-host-merge.ts +++ b/src/renderer/src/store/slices/worktrees/listing/detected-worktree-host-merge.ts @@ -22,6 +22,7 @@ export function mergeDetectedWorktreesForHost( current.repoId === refreshed.repoId && current.authoritative === refreshed.authoritative && current.source === refreshed.source && + current.unavailableReason === refreshed.unavailableReason && current.worktrees === worktrees ) { return current diff --git a/src/renderer/src/store/slices/worktrees/listing/detected-worktree-unavailable-reason.test.ts b/src/renderer/src/store/slices/worktrees/listing/detected-worktree-unavailable-reason.test.ts new file mode 100644 index 00000000000..da44f24c0d3 --- /dev/null +++ b/src/renderer/src/store/slices/worktrees/listing/detected-worktree-unavailable-reason.test.ts @@ -0,0 +1,31 @@ +import { describe, expect, it } from 'vitest' +import { makeDetectedResult } from '../../worktrees-detected-listing-fixtures' +import { mergeDetectedWorktreesForHost } from './detected-worktree-host-merge' +import { areDetectedWorktreeResultsEqual } from './worktree-catalog-visibility' + +const failed = (unavailableReason?: string) => + makeDetectedResult('repo-1', [], { + authoritative: false, + source: 'metadata-fallback', + ...(unavailableReason ? { unavailableReason } : {}) + }) + +// Why: two failed scans differ only by cause; dropping that from equality would freeze the first +// reason on the header until the listing's rows or authority changed. +describe('detected listing unavailable reason', () => { + it('is part of listing equality', () => { + expect(areDetectedWorktreeResultsEqual(failed('distro gone'), failed('distro gone'))).toBe(true) + expect(areDetectedWorktreeResultsEqual(failed('distro gone'), failed('mount hung'))).toBe(false) + expect(areDetectedWorktreeResultsEqual(failed('distro gone'), failed())).toBe(false) + }) + + it('survives the host merge when only the reason changed', () => { + const merged = mergeDetectedWorktreesForHost( + failed('distro gone'), + failed('mount hung'), + 'local' + ) + + expect(merged.unavailableReason).toBe('mount hung') + }) +}) diff --git a/src/renderer/src/store/slices/worktrees/listing/worktree-catalog-visibility.ts b/src/renderer/src/store/slices/worktrees/listing/worktree-catalog-visibility.ts index ce7a13c4b25..c6d3a9636f7 100644 --- a/src/renderer/src/store/slices/worktrees/listing/worktree-catalog-visibility.ts +++ b/src/renderer/src/store/slices/worktrees/listing/worktree-catalog-visibility.ts @@ -14,6 +14,7 @@ export function areDetectedWorktreeResultsEqual( current.repoId === next.repoId && current.authoritative === next.authoritative && current.source === next.source && + current.unavailableReason === next.unavailableReason && catalogRowsEqual(current.worktrees, next.worktrees) ) } diff --git a/src/shared/agent-session-definitive-refusal.test.ts b/src/shared/agent-session-definitive-refusal.test.ts new file mode 100644 index 00000000000..74dd3529cf6 --- /dev/null +++ b/src/shared/agent-session-definitive-refusal.test.ts @@ -0,0 +1,48 @@ +import { describe, expect, it } from 'vitest' +import { AGENT_SESSION_WIRE_REFUSAL_CODES } from './agent-session-wire' +import { agentSessionRefusalOperationState } from './agent-session-refusal-retry' +import { isDefinitiveAgentSessionCreateRefusal } from './agent-session-definitive-refusal' + +describe('definitive agent-session create refusals', () => { + it('treats an unsupported structured session as definitive', () => { + expect(isDefinitiveAgentSessionCreateRefusal('structured_agent_session_unsupported')).toBe(true) + }) + + it('never treats an unproven outcome as definitive', () => { + expect(isDefinitiveAgentSessionCreateRefusal('agent_session_operation_unknown')).toBe(false) + expect(isDefinitiveAgentSessionCreateRefusal('agent_session_ownership_unknown')).toBe(false) + }) + + it('leaves transport failures, timeouts and a missing code unknown', () => { + expect(isDefinitiveAgentSessionCreateRefusal('runtime_error')).toBe(false) + expect(isDefinitiveAgentSessionCreateRefusal('remote_runtime_unavailable')).toBe(false) + expect(isDefinitiveAgentSessionCreateRefusal('timeout')).toBe(false) + expect(isDefinitiveAgentSessionCreateRefusal('runtime_timeout')).toBe(false) + expect(isDefinitiveAgentSessionCreateRefusal(undefined)).toBe(false) + expect(isDefinitiveAgentSessionCreateRefusal(null)).toBe(false) + expect(isDefinitiveAgentSessionCreateRefusal('')).toBe(false) + }) + + it('counts a method an old host never registered as definitive', () => { + expect(isDefinitiveAgentSessionCreateRefusal('method_not_found')).toBe(true) + }) + + it('is an allowlist: every other wire refusal code is unknown', () => { + const definitive = AGENT_SESSION_WIRE_REFUSAL_CODES.filter((code) => + isDefinitiveAgentSessionCreateRefusal(code) + ) + expect(definitive).toEqual(['structured_agent_session_unsupported']) + }) + + it('does not answer the durable-settlement question, which disagrees on the one code that matters', () => { + // Guards the reuse this allowlist exists to avoid: settlement state calls the definitive + // refusal pending-admission, which would rule out the fallback it is meant to allow. + expect( + agentSessionRefusalOperationState( + 'agentSession.create', + 'structured_agent_session_unsupported' + ) + ).toBe('pending-admission') + expect(isDefinitiveAgentSessionCreateRefusal('structured_agent_session_unsupported')).toBe(true) + }) +}) diff --git a/src/shared/agent-session-definitive-refusal.ts b/src/shared/agent-session-definitive-refusal.ts new file mode 100644 index 00000000000..80721592ed9 --- /dev/null +++ b/src/shared/agent-session-definitive-refusal.ts @@ -0,0 +1,35 @@ +/** + * "May a caller create something else instead?" — the fallback question. + * + * Deliberately NOT `agentSessionRefusalOperationState`: that answers "did this operation durably + * settle?", and for that question `structured_agent_session_unsupported` is correctly + * pending-admission. Reused here it would rule out a fallback on the one refusal that most needs + * one. The two questions only look alike. + * + * An allowlist, never a negation: falling back on an outcome the host could not describe is how a + * user ends up with two sessions for one intent. Everything absent — transport failures, timeouts, + * `agent_session_operation_unknown`, `agent_session_ownership_unknown` — is unknown, and unknown + * never falls back. + */ + +import type { AgentSessionWireRefusalCode } from './agent-session-wire' + +/** Proves the host neither created a session nor will on a retry. */ +const DEFINITIVE_REFUSAL_CODES: ReadonlySet = new Set([ + 'structured_agent_session_unsupported' +]) + +/** A dispatcher that never registered the method ran no handler at all, which is as definitive as + * a refusal — and the only transport-level answer that is. */ +const DEFINITIVE_RPC_ERROR_CODES: ReadonlySet = new Set(['method_not_found']) + +/** + * True only when the code proves nothing was created. Accepts a wire refusal code or an RPC error + * code; the two namespaces are disjoint. + */ +export function isDefinitiveAgentSessionCreateRefusal(code: string | null | undefined): boolean { + if (typeof code !== 'string') { + return false + } + return DEFINITIVE_REFUSAL_CODES.has(code) || DEFINITIVE_RPC_ERROR_CODES.has(code) +} diff --git a/src/shared/agent-session-journal-schemas.ts b/src/shared/agent-session-journal-schemas.ts index 505aa22c82f..ee200cdcd94 100644 --- a/src/shared/agent-session-journal-schemas.ts +++ b/src/shared/agent-session-journal-schemas.ts @@ -64,7 +64,24 @@ const Block = z.union([ z.object({ type: z.string() }).refine((block) => !KNOWN_BLOCK_TYPES.has(block.type)) ]) -const PromptOption = z.object({ id: z.string(), label: z.string() }) +const PromptOption = z + .object({ + id: z.string(), + label: z.string(), + description: z.string().optional() + }) + .strict() + +const Question = z + .object({ + id: z.string(), + question: z.string(), + header: z.string().optional(), + multiSelect: z.boolean(), + options: z.array(PromptOption), + freeTextQuestionId: z.string().optional() + }) + .strict() const Resolution = z.object({ state: z.string().min(1), @@ -101,6 +118,7 @@ export const AgentJournalItemBodySchema = z.discriminatedUnion('kind', [ kind: z.literal('question'), question: z.string(), options: z.array(PromptOption), + questions: z.array(Question).optional(), freeTextQuestionId: z.string().optional(), resolution: Resolution }), diff --git a/src/shared/agent-session-journal-types.ts b/src/shared/agent-session-journal-types.ts index f37547b7e79..cf6d89070bf 100644 --- a/src/shared/agent-session-journal-types.ts +++ b/src/shared/agent-session-journal-types.ts @@ -113,6 +113,17 @@ export type AgentJournalResolution = { export type AgentJournalPromptOption = { id: string label: string + description?: string +} + +export type AgentJournalQuestion = { + id: string + question: string + header?: string + multiSelect: boolean + options: AgentJournalPromptOption[] + /** Present when the provider accepts an answer outside the offered options. */ + freeTextQuestionId?: string } export type AgentJournalApprovalItem = { @@ -127,6 +138,7 @@ export type AgentJournalQuestionItem = { kind: 'question' question: string options: AgentJournalPromptOption[] + questions?: AgentJournalQuestion[] /** Present when the provider accepts an answer outside the offered options. */ freeTextQuestionId?: string resolution: AgentJournalResolution diff --git a/src/shared/agent-session-question-answer.test.ts b/src/shared/agent-session-question-answer.test.ts new file mode 100644 index 00000000000..529bfdfd72d --- /dev/null +++ b/src/shared/agent-session-question-answer.test.ts @@ -0,0 +1,49 @@ +import { describe, expect, it } from 'vitest' +import { + decodeAgentSessionQuestionAnswers, + encodeAgentSessionQuestionAnswers, + isValidAgentSessionQuestionAnswers, + type AgentSessionQuestionAnswer +} from './agent-session-question-answer' + +describe('agent-session grouped question answers', () => { + const answers: AgentSessionQuestionAnswer[] = [ + { questionId: 'q1', optionIds: ['target-web', 'target-mobile'] }, + { questionId: 'q2', optionIds: [], other: 'SSH host' } + ] + + it('round-trips grouped multi-select and free-text answers', () => { + expect(decodeAgentSessionQuestionAnswers(encodeAgentSessionQuestionAnswers(answers))).toEqual( + answers + ) + }) + + it('validates each grouped answer against its question shape', () => { + const questions = [ + { + id: 'q1', + question: 'Targets', + multiSelect: true, + options: [ + { id: 'target-web', label: 'Web' }, + { id: 'target-mobile', label: 'Mobile' } + ] + }, + { + id: 'q2', + question: 'Host', + multiSelect: false, + options: [], + freeTextQuestionId: 'q2' + } + ] + + expect(isValidAgentSessionQuestionAnswers(questions, answers)).toBe(true) + expect( + isValidAgentSessionQuestionAnswers(questions, [ + { questionId: 'q1', optionIds: ['unknown'] }, + answers[1]! + ]) + ).toBe(false) + }) +}) diff --git a/src/shared/agent-session-question-answer.ts b/src/shared/agent-session-question-answer.ts new file mode 100644 index 00000000000..f90df30ddff --- /dev/null +++ b/src/shared/agent-session-question-answer.ts @@ -0,0 +1,84 @@ +import type { AgentJournalQuestion } from './agent-session-journal-types' + +const GROUP_ANSWER_PREFIX = 'question-group:' + +export type AgentSessionQuestionAnswer = { + questionId: string + optionIds: string[] + other?: string +} + +export function encodeAgentSessionQuestionAnswers( + answers: readonly AgentSessionQuestionAnswer[] +): string { + return `${GROUP_ANSWER_PREFIX}${encodeURIComponent(JSON.stringify(answers))}` +} + +export function decodeAgentSessionQuestionAnswers( + encoded: string +): AgentSessionQuestionAnswer[] | null { + if (!encoded.startsWith(GROUP_ANSWER_PREFIX)) { + return null + } + try { + const parsed: unknown = JSON.parse( + decodeURIComponent(encoded.slice(GROUP_ANSWER_PREFIX.length)) + ) + if (!Array.isArray(parsed)) { + return null + } + const answers = parsed.flatMap((value): AgentSessionQuestionAnswer[] => { + if (!value || typeof value !== 'object' || Array.isArray(value)) { + return [] + } + const record = value as Record + if ( + typeof record.questionId !== 'string' || + !Array.isArray(record.optionIds) || + !record.optionIds.every((optionId) => typeof optionId === 'string') || + (record.other !== undefined && typeof record.other !== 'string') + ) { + return [] + } + return [ + { + questionId: record.questionId, + optionIds: record.optionIds, + ...(typeof record.other === 'string' ? { other: record.other } : {}) + } + ] + }) + return answers.length === parsed.length ? answers : null + } catch { + return null + } +} + +export function isValidAgentSessionQuestionAnswers( + questions: readonly AgentJournalQuestion[], + answers: readonly AgentSessionQuestionAnswer[] +): boolean { + if (answers.length !== questions.length) { + return false + } + const byId = new Map(answers.map((answer) => [answer.questionId, answer])) + if (byId.size !== answers.length) { + return false + } + return questions.every((question) => { + const answer = byId.get(question.id) + if (!answer) { + return false + } + const offered = new Set(question.options.map((option) => option.id)) + if (answer.optionIds.some((optionId) => !offered.has(optionId))) { + return false + } + const other = answer.other?.trim() ?? '' + if (other && !question.freeTextQuestionId) { + return false + } + const answerCount = answer.optionIds.length + (other ? 1 : 0) + return answerCount > 0 && (question.multiSelect || answerCount === 1) + }) +} diff --git a/src/shared/agent-session-wire.ts b/src/shared/agent-session-wire.ts index 893534f7f10..40a408df899 100644 --- a/src/shared/agent-session-wire.ts +++ b/src/shared/agent-session-wire.ts @@ -49,6 +49,18 @@ export type AgentSessionHandoffRequest = { export type AgentSessionHandoffResult = { status: AgentSessionHandoffStatus } +export type AgentSessionBackgroundTask = { + id: string + kind: 'agent' | 'workflow' | 'command' | 'monitor' | 'unknown' + description?: string +} + +export type AgentSessionBackgroundTaskState = { + state: 'monitoring' + /** Optional so mixed-version clients can consume state-only hosts. */ + tasks?: AgentSessionBackgroundTask[] +} + /** Backward paging is the client's normal read; 40 matches the page size the * mobile list renders without a visible fill-in. */ export const AGENT_SESSION_HISTORY_DEFAULT_LIMIT = 40 @@ -92,6 +104,8 @@ export type AgentSessionHistoryPage = { liveCursor?: AgentJournalCursor hasOlder: boolean hasNewer: boolean + /** Present on hosts that expose provider-owned background task lifecycle. */ + backgroundTasks?: AgentSessionBackgroundTaskState | null } export type AgentSessionHistoryResult = @@ -123,6 +137,7 @@ export type AgentSessionSubscribeEvent = page: AgentSessionHistoryPage fence: number handoff?: AgentSessionHandoffStatus + backgroundTasks?: AgentSessionBackgroundTaskState | null } | { type: 'batch' @@ -131,6 +146,7 @@ export type AgentSessionSubscribeEvent = /** Added with handoff state so mixed-version cursors retain the ownership fence. */ fence?: number handoff?: AgentSessionHandoffStatus + backgroundTasks?: AgentSessionBackgroundTaskState | null } | { type: 'reset' @@ -139,6 +155,7 @@ export type AgentSessionSubscribeEvent = page: AgentSessionHistoryPage fence: number handoff?: AgentSessionHandoffStatus + backgroundTasks?: AgentSessionBackgroundTaskState | null } | { type: 'end' } @@ -254,5 +271,11 @@ export type AgentSessionOptionsResult = { current: { model: string effort?: string + /** + * Option ids whose value the provider reported back, not merely accepted. + * Optional: a host that predates it sends nothing and the client keeps + * treating the value as unconfirmed, which is what it was before. + */ + confirmed?: readonly string[] } } diff --git a/src/shared/cheap-process-table-snapshot-reader.ts b/src/shared/cheap-process-table-snapshot-reader.ts new file mode 100644 index 00000000000..1d2401f2045 --- /dev/null +++ b/src/shared/cheap-process-table-snapshot-reader.ts @@ -0,0 +1,51 @@ +import { runProcess } from './child-process/run-process' +import { + CHEAP_PS_ARGS, + PS_MAX_BUFFER_BYTES, + ProcessTableCaptureError, + parseCheapProcessTableRows, + type CheapProcessTableRow +} from './process-table-snapshot' +import { + PS_TIMEOUT_MS, + createProcessTableSnapshotReader, + withEvidenceBudget +} from './process-table-snapshot-reader' + +/** + * The cheap-tier sibling of the strict evidence reader: same coalescing and TTL, a + * column set without `tty=`/`command=`. Separate instance because the two column sets + * parse differently and a cheap capture must never be served to an evidence consumer. + */ +const cheapProcessTableReader = createProcessTableSnapshotReader({ + runPs: async () => { + const result = await runProcess({ + program: 'ps', + args: CHEAP_PS_ARGS, + timeoutMs: PS_TIMEOUT_MS, + maxOutputBytes: PS_MAX_BUFFER_BYTES + }) + // A ceiling hit is truncation, not absence: name it in the domain vocabulary. + if (result.outputTruncated) { + throw new ProcessTableCaptureError('capture_truncated') + } + if (result.timedOut) { + throw new ProcessTableCaptureError('capture_timeout') + } + if (result.code !== 0) { + throw new ProcessTableCaptureError(`ps_exit_${result.code ?? result.signal ?? 'unknown'}`) + } + return parseCheapProcessTableRows(result.stdout) + }, + now: () => Date.now() +}) + +/** Same wait bound as the evidence read: a stalled cheap capture must fall through to the full + * path's own handling rather than pin a polled tick. */ +export async function getCheapProcessTableSnapshot(): Promise { + return withEvidenceBudget(cheapProcessTableReader.getSnapshot()) +} + +export function resetCheapProcessTableSnapshotForTests(): void { + cheapProcessTableReader.reset() +} diff --git a/src/shared/cheap-process-table-snapshot.test.ts b/src/shared/cheap-process-table-snapshot.test.ts new file mode 100644 index 00000000000..74bbba2b69b --- /dev/null +++ b/src/shared/cheap-process-table-snapshot.test.ts @@ -0,0 +1,147 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { runProcessMock } = vi.hoisted(() => ({ runProcessMock: vi.fn() })) + +// The cheap reader goes through Orca's single child-process entry point (windowsHide, argv +// encoding, tree termination); mock at that seam rather than node:child_process. +vi.mock('./child-process/run-process', () => ({ runProcess: runProcessMock })) + +import { + getCheapProcessTableSnapshot, + resetCheapProcessTableSnapshotForTests +} from './cheap-process-table-snapshot-reader' +import { PS_TIMEOUT_MS } from './process-table-snapshot-reader' +import { + CHEAP_PS_ARGS, + PS_ARGS, + PS_MAX_BUFFER_BYTES, + parseCheapProcessTableRows, + ProcessTableCaptureError +} from './process-table-snapshot' + +function installPs(stdout: string, outputTruncated = false): string[][] { + const calls: string[][] = [] + runProcessMock.mockImplementation(async (spec: { program: string; args: readonly string[] }) => { + calls.push([spec.program, ...spec.args]) + return { code: 0, signal: null, stdout, stderr: '', timedOut: false, outputTruncated } + }) + return calls +} + +describe('parseCheapProcessTableRows', () => { + it('parses the macOS column set with a padded lstart marker', () => { + const rows = parseCheapProcessTableRows( + [ + ' 1 0 1 0 Ss Tue Sep 1 01:49:39 2026', + ' 4242 4200 4242 4243 S Thu Sep 3 16:02:01 2026', + ' 4243 4242 4243 4243 S+ Thu Sep 3 16:02:05 2026', + '' + ].join('\n') + ) + expect(rows).toEqual([ + { pid: 1, ppid: 0, pgid: 1, tpgid: 0, stat: 'Ss', startTime: 'Tue Sep 1 01:49:39 2026' }, + { + pid: 4242, + ppid: 4200, + pgid: 4242, + tpgid: 4243, + stat: 'S', + startTime: 'Thu Sep 3 16:02:01 2026' + }, + { + pid: 4243, + ppid: 4242, + pgid: 4243, + tpgid: 4243, + stat: 'S+', + startTime: 'Thu Sep 3 16:02:05 2026' + } + ]) + }) + + it('parses the Linux column set, which carries no start marker', () => { + const rows = parseCheapProcessTableRows( + ' 2 0 0 -1 S\r\n 900 1 900 900 Ss+\r\n' + ) + expect(rows).toEqual([ + { pid: 2, ppid: 0, pgid: 0, tpgid: -1, stat: 'S' }, + { pid: 900, ppid: 1, pgid: 900, tpgid: 900, stat: 'Ss+' } + ]) + }) + + it('skips malformed rows rather than failing the capture', () => { + expect(parseCheapProcessTableRows('garbage\n 7 1 7 7 S\n')).toEqual([ + { pid: 7, ppid: 1, pgid: 7, tpgid: 7, stat: 'S' } + ]) + }) + + it('treats an empty capture as unreadable, never as "no processes"', () => { + expect(() => parseCheapProcessTableRows('\n\n')).toThrow(ProcessTableCaptureError) + }) +}) + +describe('getCheapProcessTableSnapshot', () => { + beforeEach(() => { + runProcessMock.mockReset() + resetCheapProcessTableSnapshotForTests() + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(0) + }) + + afterEach(() => { + vi.useRealTimers() + }) + + it('forks ps with the cheap column set only, never tty or command', async () => { + const calls = installPs(' 7 1 7 7 S\n') + await getCheapProcessTableSnapshot() + expect(calls).toEqual([['ps', ...CHEAP_PS_ARGS]]) + expect(CHEAP_PS_ARGS.join(' ')).not.toMatch(/tty=|command=|etimes=/) + expect(CHEAP_PS_ARGS).not.toEqual(PS_ARGS) + }) + + it('coalesces concurrent readers onto one fork and honours the TTL', async () => { + const calls = installPs(' 7 1 7 7 S\n') + await Promise.all([getCheapProcessTableSnapshot(), getCheapProcessTableSnapshot()]) + await getCheapProcessTableSnapshot() + expect(calls).toHaveLength(1) + vi.setSystemTime(600) + await getCheapProcessTableSnapshot() + expect(calls).toHaveLength(2) + }) + + it('passes the full-tier buffer ceiling and timeout to the runner', async () => { + installPs(' 7 1 7 7 S\n') + await getCheapProcessTableSnapshot() + expect(runProcessMock).toHaveBeenCalledWith( + expect.objectContaining({ maxOutputBytes: PS_MAX_BUFFER_BYTES, timeoutMs: PS_TIMEOUT_MS }) + ) + }) + + it('names a clipped capture as truncated, a killed one as a timeout, and a non-zero exit by its code', async () => { + installPs(' 7 1 7 7 S\n', true) + await expect(getCheapProcessTableSnapshot()).rejects.toMatchObject({ + reason: 'capture_truncated' + }) + resetCheapProcessTableSnapshotForTests() + runProcessMock.mockResolvedValueOnce({ + code: null, + signal: 'SIGKILL', + stdout: '', + stderr: '', + timedOut: true + }) + await expect(getCheapProcessTableSnapshot()).rejects.toMatchObject({ + reason: 'capture_timeout' + }) + resetCheapProcessTableSnapshotForTests() + runProcessMock.mockResolvedValueOnce({ + code: 1, + signal: null, + stdout: '', + stderr: 'ps: bad column', + timedOut: false + }) + await expect(getCheapProcessTableSnapshot()).rejects.toMatchObject({ reason: 'ps_exit_1' }) + }) +}) diff --git a/src/shared/child-process/process-tree-kill-gate.ts b/src/shared/child-process/process-tree-kill-gate.ts new file mode 100644 index 00000000000..420fbfeda9b --- /dev/null +++ b/src/shared/child-process/process-tree-kill-gate.ts @@ -0,0 +1,43 @@ +/** + * Seam that lets the main process decide, and record, the tree-kills issued + * from code it does not own. + * + * Why a seam and not a direct call: `signalProcessTree` is the choke point every + * `runProcess` termination funnels through, and the codex app-server and + * ephemeral-VM kills are shared with the CLI — all of them live outside + * `src/main` and cannot import the own-Chromium guard or the crash breadcrumb + * store. Main registers the guard at startup; everywhere else this admits every + * kill and records nothing. + */ + +/** Blast radius, not mechanism: `win-taskkill-tree` is addressed by pid and walks + * whatever tree that pid has *now*, so it can land on a recycled pid that is + * since one of Orca's own Chromium processes. A process group can only contain + * processes Orca itself put there. */ +export type ProcessTreeKillScope = 'win-taskkill-tree' | 'posix-process-group' + +export type ProcessTreeKill = { + pid: number + site: string + scope: ProcessTreeKillScope +} + +/** False means the caller must not walk that pid's tree — main is accounting for + * it. Killing the root through its own child handle stays correct and required: + * a handle cannot land on the recycled pid the refusal is about. */ +type ProcessTreeKillGate = (kill: ProcessTreeKill) => boolean + +let gate: ProcessTreeKillGate | null = null + +export function setProcessTreeKillGate(next: ProcessTreeKillGate | null): void { + gate = next +} + +export function admitProcessTreeKill(kill: ProcessTreeKill): boolean { + try { + return gate?.(kill) ?? true + } catch { + // Diagnostics must never turn a successful termination into a failed one. + return true + } +} diff --git a/src/shared/child-process/process-tree-kill-observer.ts b/src/shared/child-process/process-tree-kill-observer.ts deleted file mode 100644 index 8abf971798a..00000000000 --- a/src/shared/child-process/process-tree-kill-observer.ts +++ /dev/null @@ -1,37 +0,0 @@ -/** - * Seam that lets the main process record the tree-kills issued from here. - * - * Why a seam and not a direct call: `signalProcessTree` is the choke point every - * `runProcess` termination funnels through, but it lives in `src/shared` and so - * runs in the CLI and relay too — it cannot import the main-process crash - * breadcrumb store. Main registers the recorder at startup; everywhere else this - * stays a no-op. - */ - -/** Blast radius, not mechanism: `win-taskkill-tree` is addressed by pid and walks - * whatever tree that pid has *now*, so it can land on a recycled pid that is - * since one of Orca's own Chromium processes. A process group can only contain - * processes Orca itself put there. */ -export type ProcessTreeKillScope = 'win-taskkill-tree' | 'posix-process-group' - -export type ProcessTreeKill = { - pid: number - site: string - scope: ProcessTreeKillScope -} - -type ProcessTreeKillObserver = (kill: ProcessTreeKill) => void - -let observer: ProcessTreeKillObserver | null = null - -export function setProcessTreeKillObserver(next: ProcessTreeKillObserver | null): void { - observer = next -} - -export function notifyProcessTreeKill(kill: ProcessTreeKill): void { - try { - observer?.(kill) - } catch { - // Diagnostics must never turn a successful termination into a failed one. - } -} diff --git a/src/shared/child-process/process-tree-termination.test.ts b/src/shared/child-process/process-tree-termination.test.ts index cc3394fb5a3..10a4c725089 100644 --- a/src/shared/child-process/process-tree-termination.test.ts +++ b/src/shared/child-process/process-tree-termination.test.ts @@ -7,7 +7,7 @@ const { spawnMock } = vi.hoisted(() => ({ spawnMock: vi.fn() })) vi.mock('node:child_process', () => ({ spawn: spawnMock })) import { forceTerminateProcessTree, signalProcessTree } from './process-tree-termination' -import { setProcessTreeKillObserver, type ProcessTreeKill } from './process-tree-kill-observer' +import { setProcessTreeKillGate, type ProcessTreeKill } from './process-tree-kill-gate' function mockProcess(pid: number): ChildProcess { const child = new EventEmitter() as EventEmitter & { @@ -103,11 +103,14 @@ describe('process-tree-kill breadcrumb seam', () => { beforeEach(() => { observed.length = 0 - setProcessTreeKillObserver((kill) => observed.push(kill)) + setProcessTreeKillGate((kill) => { + observed.push(kill) + return true + }) }) afterEach(() => { - setProcessTreeKillObserver(null) + setProcessTreeKillGate(null) spawnMock.mockReset() vi.restoreAllMocks() }) diff --git a/src/shared/child-process/process-tree-termination.ts b/src/shared/child-process/process-tree-termination.ts index 15264b2f6ce..ffcb93b2245 100644 --- a/src/shared/child-process/process-tree-termination.ts +++ b/src/shared/child-process/process-tree-termination.ts @@ -1,5 +1,5 @@ import { spawn as nodeSpawn, type ChildProcess } from 'node:child_process' -import { notifyProcessTreeKill } from './process-tree-kill-observer' +import { admitProcessTreeKill } from './process-tree-kill-gate' const PROBE_INTERVAL_MS = 25 const SUBPROCESS_TIMEOUT_MS = 2_000 @@ -13,8 +13,17 @@ const MAX_PS_OUTPUT_BYTES = 8 * 1024 * 1024 * leader would hand it to whatever group it inherited instead. * * Runs in every host — Electron main, the daemon, the relay, the CLI — so the - * main-process own-Chromium guard cannot reach here; the exit check below is - * what keeps the Windows branch off a pid that is no longer ours. + * own-Chromium guard arrives through `process-tree-kill-gate`, which main + * installs and every other host leaves admitting. The exit check below is what + * keeps the Windows branch off a pid that is no longer ours on those hosts. + * + * Both arms ask before walking the tree, so a refused pid never gets a + * pid-addressed kill; that also means the recorded crumb says "about to kill", + * not "killed". The root is still killed through its handle, which cannot reach + * the recycled pid the refusal was about, so a refusal is never a leak. Main's + * gate only ever refuses the `win-taskkill-tree` scope — a POSIX group holds + * only what Orca put in it — so the POSIX refusal arm is the seam's contract, + * not something any installed gate exercises today. */ export function signalProcessTree(child: ChildProcess, signal?: NodeJS.Signals): Promise { if (!child.pid) { @@ -39,13 +48,20 @@ export function signalProcessTree(child: ChildProcess, signal?: NodeJS.Signals): } return taskkillTree(child, child.pid, signal) } - try { - process.kill(-child.pid, signal) - notifyProcessTreeKill({ + if ( + !admitProcessTreeKill({ pid: child.pid, site: 'run-process-tree', scope: 'posix-process-group' }) + ) { + // Same shape as the reaped-pid skip above: refuse the group, still kill the + // root by handle, and report unverified. + killRoot(child, signal) + return Promise.resolve(false) + } + try { + process.kill(-child.pid, signal) return Promise.resolve(true) } catch { return Promise.resolve(!processGroupExists(child.pid)) @@ -73,6 +89,13 @@ function taskkillTree( rootPid: number, signal?: NodeJS.Signals ): Promise { + // Asked before the spawn, not after: a refusal has to prevent the taskkill. + if ( + !admitProcessTreeKill({ pid: rootPid, site: 'run-process-tree', scope: 'win-taskkill-tree' }) + ) { + killRoot(child, signal) + return Promise.resolve(false) + } return new Promise((resolve) => { let killer: ChildProcess try { @@ -86,7 +109,6 @@ function taskkillTree( resolve(false) return } - notifyProcessTreeKill({ pid: rootPid, site: 'run-process-tree', scope: 'win-taskkill-tree' }) let settled = false const finish = (fallback: boolean): void => { if (settled) { diff --git a/src/shared/codex-startup-delivery.test.ts b/src/shared/codex-startup-delivery.test.ts index e7788f778fe..87e65f7514c 100644 --- a/src/shared/codex-startup-delivery.test.ts +++ b/src/shared/codex-startup-delivery.test.ts @@ -1,5 +1,8 @@ import { describe, expect, it } from 'vitest' -import { hasCodexNativeDraftFlag } from './codex-startup-delivery' +import { + hasCodexNativeDraftFlag, + shouldUseShellReadyStartupDelivery +} from './codex-startup-delivery' describe('hasCodexNativeDraftFlag', () => { it('matches Codex --prefill option tokens', () => { @@ -26,3 +29,42 @@ describe('hasCodexNativeDraftFlag', () => { expect(hasCodexNativeDraftFlag('codex --model gpt-5')).toBe(false) }) }) + +describe('shouldUseShellReadyStartupDelivery', () => { + it('honours an explicit shell-ready hint whatever the command', () => { + expect( + shouldUseShellReadyStartupDelivery({ + command: 'claude', + startupCommandDelivery: 'shell-ready' + }) + ).toBe(true) + }) + + it('keeps plain Codex on the fast path when the shell is unknown', () => { + expect(shouldUseShellReadyStartupDelivery({ command: 'codex' })).toBe(false) + }) + + it('waits for plain Codex on shells that publish the marker from the line editor', () => { + expect(shouldUseShellReadyStartupDelivery({ command: 'codex', shellPath: '/bin/bash' })).toBe( + true + ) + expect( + shouldUseShellReadyStartupDelivery({ command: 'codex', shellPath: '/opt/homebrew/bin/zsh' }) + ).toBe(true) + }) + + it('leaves plain Codex unwaited on shells that emit the marker before the reader', () => { + expect( + shouldUseShellReadyStartupDelivery({ command: 'codex', shellPath: '/usr/bin/fish' }) + ).toBe(false) + }) + + it('does not change non-Codex commands, which their transports already wait for', () => { + expect(shouldUseShellReadyStartupDelivery({ command: 'claude', shellPath: '/bin/bash' })).toBe( + false + ) + expect(shouldUseShellReadyStartupDelivery({ command: undefined, shellPath: '/bin/bash' })).toBe( + false + ) + }) +}) diff --git a/src/shared/codex-startup-delivery.ts b/src/shared/codex-startup-delivery.ts index a45337dd0be..538b3befa19 100644 --- a/src/shared/codex-startup-delivery.ts +++ b/src/shared/codex-startup-delivery.ts @@ -1,4 +1,5 @@ import { recognizeAgentProcessFromCommandLine } from './agent-process-recognition' +import { shellReadyMarkerComesFromLineEditor } from './shell-ready-marker-timing' export type StartupCommandDelivery = 'fast' | 'shell-ready' @@ -74,9 +75,23 @@ export function hasCodexNativeDraftFlag(command: string | null | undefined): boo ) } +export function isCodexStartupCommand(command: string | null | undefined): boolean { + return recognizeAgentProcessFromCommandLine(command)?.agent === 'codex' +} + export function shouldUseShellReadyStartupDelivery(args: { command: string | null | undefined startupCommandDelivery?: StartupCommandDelivery + /** The shell that will run the command, when the deciding side knows it. Plain Codex + * waits for the handshake on shells that publish the marker from their line editor: + * there the wait ends at the prompt, while an early write double-echoes the launch. */ + shellPath?: string }): boolean { - return args.startupCommandDelivery === 'shell-ready' || hasCodexNativeDraftFlag(args.command) + return ( + args.startupCommandDelivery === 'shell-ready' || + hasCodexNativeDraftFlag(args.command) || + (args.shellPath !== undefined && + shellReadyMarkerComesFromLineEditor(args.shellPath) && + isCodexStartupCommand(args.command)) + ) } diff --git a/src/shared/ephemeral-vm-recipe-process.ts b/src/shared/ephemeral-vm-recipe-process.ts index b258a1b2704..30d44ac1dbc 100644 --- a/src/shared/ephemeral-vm-recipe-process.ts +++ b/src/shared/ephemeral-vm-recipe-process.ts @@ -1,5 +1,6 @@ import { spawn, type ChildProcessWithoutNullStreams } from 'node:child_process' import type { EphemeralVmRecipeContext } from './ephemeral-vm-recipe-runner' +import { admitProcessTreeKill } from './child-process/process-tree-kill-gate' const DEFAULT_MAX_CAPTURE_BYTES = 1024 * 1024 const CANCEL_FORCE_KILL_DELAY_MS = 5_000 @@ -132,13 +133,26 @@ export async function runRecipeCommand(args: { }) } -function killRecipeProcess(child: ChildProcessWithoutNullStreams, force = false): void { +/** Exported for the refusal-fallback test; the abort path is otherwise unreachable. */ +export function killRecipeProcess(child: ChildProcessWithoutNullStreams, force = false): void { const signal = force ? 'SIGKILL' : 'SIGTERM' if (process.platform === 'win32') { // Recipes run through `cmd.exe /c` (shell: true), so child.kill() would only // terminate the wrapper and orphan the actual recipe subprocess (e.g. a cloud // CLI mid-provision). taskkill /T walks and kills the whole tree. if (child.pid) { + if ( + !admitProcessTreeKill({ + pid: child.pid, + site: 'ephemeral-vm-recipe', + scope: 'win-taskkill-tree' + }) + ) { + // Refusal blocks the tree walk, not the termination: the root kill is + // handle-addressed, so it cannot reach the recycled pid we refused. + child.kill(signal) + return + } const killer = spawn('taskkill', ['/pid', String(child.pid), '/t', '/f'], { windowsHide: true, stdio: 'ignore' diff --git a/src/shared/native-chat-begin-patch.ts b/src/shared/native-chat-begin-patch.ts new file mode 100644 index 00000000000..becd0ece46a --- /dev/null +++ b/src/shared/native-chat-begin-patch.ts @@ -0,0 +1,173 @@ +import { finalizeEditFile, type NativeChatEditFile } from './native-chat-edit-model' +import { editLinesFromUnifiedPatch, editLinesFromWholeFile } from './native-chat-unified-patch' + +const BEGIN = '*** Begin Patch' +const END = '*** End Patch' +const FILE_HEADER = /^\*\*\* (Add|Update|Delete) File: (.+)$/ +const MOVE_HEADER = /^\*\*\* Move to: (.+)$/ +/** Envelope structure that carries no file content of its own. */ +const CONTROL_LINE = /^\*\*\* (?:End of File|Environment ID:)/ + +/** The envelope reaches a command tool as one of its patch or command + * arguments, either whole or as one word of the argument vector it runs. + * Recover its text. Callers must gate this on the tool being one that runs a + * patch: a file's own contents may quote an envelope. + * + * `requireApplyCommand` is for a general command tool, where the envelope + * proves nothing on its own — a command writing documentation quotes one + * without applying it, and the command must say it is applying it. */ +export function unwrapBeginPatch( + input: unknown, + options?: { requireApplyCommand?: boolean } +): string | null { + const source = envelopeSource(input, options?.requireApplyCommand === true) + if (!source) { + return null + } + const start = source.indexOf(BEGIN) + if (start === -1) { + return null + } + const end = source.indexOf(END, start) + if (end === -1) { + // Without the closing marker there is nothing separating the patch body from + // whatever the command line continues with, and trailing shell syntax would + // render as file content the agent never wrote. + return null + } + return source.slice(start, end + END.length) +} + +/** The arguments that carry a patch or the command line that applies one. Only + * these are searched: any other value is data the tool operates on, and a file + * whose own contents quote an envelope would otherwise be read as a patch + * against some other file entirely. */ +const ENVELOPE_ARGUMENTS = ['input', 'command', 'patch', 'arguments', 'script'] as const +/** What a command tool runs to apply an envelope, as opposed to quoting one. + * Both spellings the runner accepts, since either one really applies it. */ +const APPLY_COMMAND = /apply_?patch/ + +/** The call payload may itself be a string holding JSON. Decoding it here, in + * the one consumer that needs its structure, keeps every other reader of the + * call input seeing exactly what the provider sent. */ +function envelopeSource(input: unknown, requireApplyCommand: boolean): string | null { + if (typeof input === 'string') { + const record = jsonRecord(input) + return record + ? envelopeArgument(record, requireApplyCommand) + : applied(input, requireApplyCommand) + } + return typeof input === 'object' && input !== null + ? envelopeArgument(input as Record, requireApplyCommand) + : null +} + +function applied(value: string, requireApplyCommand: boolean): string | null { + return !requireApplyCommand || APPLY_COMMAND.test(value) ? value : null +} + +function jsonRecord(value: string): Record | null { + if (!value.trimStart().startsWith('{')) { + return null + } + try { + const parsed: unknown = JSON.parse(value) + return typeof parsed === 'object' && parsed !== null && !Array.isArray(parsed) + ? (parsed as Record) + : null + } catch { + return null + } +} + +/** A command tool's argument is a vector, so the envelope sits one level in and + * the words that apply it may be a different element than the envelope. */ +function envelopeArgument( + record: Record, + requireApplyCommand: boolean +): string | null { + for (const key of ENVELOPE_ARGUMENTS) { + const value = record[key] + const words = + typeof value === 'string' + ? [value] + : Array.isArray(value) + ? value.filter((entry): entry is string => typeof entry === 'string') + : [] + if (requireApplyCommand && !words.some((word) => APPLY_COMMAND.test(word))) { + continue + } + const word = words.find((entry) => entry.includes(BEGIN)) + if (word) { + return word + } + } + return null +} + +/** Splits a `*** Begin Patch` envelope into one entry per file it touches. */ +export function editFilesFromBeginPatch(envelope: string): NativeChatEditFile[] { + const sections: { kind: 'Add' | 'Update' | 'Delete'; path: string; body: string[] }[] = [] + const moves = new Map() + + // Split on both newline forms once, so every marker below can be matched + // exactly rather than each pattern having to tolerate a trailing `\r`. + for (const raw of envelope.split(/\r?\n/)) { + const header = FILE_HEADER.exec(raw) + if (header) { + sections.push({ + kind: header[1] as 'Add' | 'Update' | 'Delete', + path: header[2]!.trim(), + body: [] + }) + continue + } + const move = MOVE_HEADER.exec(raw) + if (move && sections.length > 0) { + moves.set(sections.length - 1, move[1]!.trim()) + continue + } + if (raw === BEGIN || raw === END || CONTROL_LINE.test(raw) || sections.length === 0) { + continue + } + sections.at(-1)!.body.push(raw) + } + + return sections.flatMap((section, index) => { + const body = section.body.join('\n') + const moved = moves.get(index) ?? null + if (section.kind === 'Add' || section.kind === 'Delete') { + const sign = section.kind === 'Add' ? '+' : '-' + // Add/Delete bodies carry a sign per line but no hunk header. + const stripped = section.body + .map((line) => (line.startsWith(sign) ? line.slice(1) : line)) + .join('\n') + const whole = editLinesFromWholeFile(stripped, section.kind === 'Add' ? 'add' : 'del') + return [ + finalizeEditFile({ + path: section.path, + oldPath: null, + changeKind: section.kind === 'Add' ? 'added' : 'deleted', + lines: whole.lines, + lineNumbersKnown: true, + truncated: whole.truncated + }) + ] + } + // The first chunk of an update may carry no hunk header at all, and a + // section may carry no body either. The envelope named the file, so it is + // reported with whatever rows it has rather than dropped from a multi-file + // envelope with nothing to say it went missing. + const parsed = editLinesFromUnifiedPatch(body, { implicitFirstHunk: true }) + return [ + finalizeEditFile({ + path: moved ?? section.path, + oldPath: moved ? section.path : null, + changeKind: moved ? 'renamed' : 'edited', + lines: parsed?.lines ?? [], + lineNumbersKnown: parsed?.lineNumbersKnown ?? false, + truncated: parsed?.truncated ?? false + }) + ] + }) +} diff --git a/src/shared/native-chat-diff.ts b/src/shared/native-chat-diff.ts index 1df82db4800..a6dde625a14 100644 --- a/src/shared/native-chat-diff.ts +++ b/src/shared/native-chat-diff.ts @@ -14,8 +14,11 @@ const DIFF_TRUNCATED_LINE: NativeChatDiffLine = { } const HUNK_HEADER = /^@@ -\d+(?:,\d+)? \+\d+(?:,\d+)? @@/ -// Lines that open a new file section, so any hunk before them has ended. -const FILE_SECTION_START = +/** Lines that open a new file section, so any hunk before them has ended. + * `--- `/`+++ ` are deliberately absent: inside a hunk they are content — a + * removed `-- comment` is emitted as `--- comment` — so they go through + * `isFileHeaderPair` instead. */ +export const FILE_SECTION_START = /^(?:diff |index |old mode |new mode |new file mode |deleted file mode |similarity index |dissimilarity index |rename |copy |Binary files )/ // Markdown thematic break or YAML document separator, not a marker. const BARE_RULE = /^(?:-{3,}|\+{3,})$/ @@ -28,11 +31,17 @@ type DiffStructure = { } /** - * Locates the `--- ` / `+++ ` file headers. A bare `---`/`+++` prefix - * is not enough to spot one: a removed line whose content began with `--` - * (SQL/Lua `-- comment`, C `--i`) is emitted as `---`. Real headers - * always come as an adjacent pair and never appear inside a hunk. + * True when the row at `index` opens a `--- ` / `+++ ` file header. A + * bare `---`/`+++` prefix is not enough to spot one: a removed line whose + * content began with `--` (SQL/Lua `-- comment`, C `--i`) is emitted as + * `---`. Real headers always come as an adjacent pair and never appear + * inside a hunk, so callers must check this only outside one. */ +export function isFileHeaderPair(lines: readonly string[], index: number): boolean { + return (lines[index] ?? '').startsWith('--- ') && (lines[index + 1] ?? '').startsWith('+++ ') +} + +/** Locates the file headers and rules that are structure rather than content. */ function scanDiffStructure(lines: string[]): DiffStructure { const metaIndices = new Set() let isStructuredDiff = false @@ -57,7 +66,7 @@ function scanDiffStructure(lines: string[]): DiffStructure { metaIndices.add(index) continue } - if (line.startsWith('--- ') && (lines[index + 1] ?? '').startsWith('+++ ')) { + if (isFileHeaderPair(lines, index)) { metaIndices.add(index) metaIndices.add(index + 1) index += 1 diff --git a/src/shared/native-chat-edit-lcs.ts b/src/shared/native-chat-edit-lcs.ts new file mode 100644 index 00000000000..733d5f5c700 --- /dev/null +++ b/src/shared/native-chat-edit-lcs.ts @@ -0,0 +1,105 @@ +import { + MAX_EDIT_DIFF_CELLS, + splitEditContent, + type NativeChatEditLine +} from './native-chat-edit-model' + +/** Line diff between two contents, interleaved with context. Numbers are + * positions within the given contents, so they locate rows in the file only + * when the caller passed whole files. */ +export function editLinesFromContents( + originalContent: string, + modifiedContent: string +): { lines: NativeChatEditLine[]; truncated: boolean } { + const original = splitEditContent(originalContent) + const modified = splitEditContent(modifiedContent) + const lines = + original.lines.length * modified.lines.length <= MAX_EDIT_DIFF_CELLS + ? lcsLines(original.lines, modified.lines) + : prefixSuffixLines(original.lines, modified.lines) + return { lines, truncated: original.truncated || modified.truncated } +} + +function context(text: string, oldNo: number, newNo: number): NativeChatEditLine { + return { kind: 'context', text, oldLineNumber: oldNo, newLineNumber: newNo } +} + +function removal(text: string, oldNo: number): NativeChatEditLine { + return { kind: 'del', text, oldLineNumber: oldNo, newLineNumber: null } +} + +function addition(text: string, newNo: number): NativeChatEditLine { + return { kind: 'add', text, oldLineNumber: null, newLineNumber: newNo } +} + +function lcsLines(original: string[], modified: string[]): NativeChatEditLine[] { + const width = modified.length + 1 + const dp = new Uint32Array((original.length + 1) * width) + for (let i = original.length - 1; i >= 0; i -= 1) { + for (let j = modified.length - 1; j >= 0; j -= 1) { + dp[i * width + j] = + original[i] === modified[j] + ? dp[(i + 1) * width + j + 1]! + 1 + : Math.max(dp[(i + 1) * width + j]!, dp[i * width + j + 1]!) + } + } + + const lines: NativeChatEditLine[] = [] + let oldIndex = 0 + let newIndex = 0 + while (oldIndex < original.length && newIndex < modified.length) { + if (original[oldIndex] === modified[newIndex]) { + lines.push(context(original[oldIndex] ?? '', oldIndex + 1, newIndex + 1)) + oldIndex += 1 + newIndex += 1 + } else if (dp[(oldIndex + 1) * width + newIndex]! >= dp[oldIndex * width + newIndex + 1]!) { + lines.push(removal(original[oldIndex] ?? '', oldIndex + 1)) + oldIndex += 1 + } else { + lines.push(addition(modified[newIndex] ?? '', newIndex + 1)) + newIndex += 1 + } + } + for (; oldIndex < original.length; oldIndex += 1) { + lines.push(removal(original[oldIndex] ?? '', oldIndex + 1)) + } + for (; newIndex < modified.length; newIndex += 1) { + lines.push(addition(modified[newIndex] ?? '', newIndex + 1)) + } + return lines +} + +function prefixSuffixLines(original: string[], modified: string[]): NativeChatEditLine[] { + let prefix = 0 + while ( + prefix < original.length && + prefix < modified.length && + original[prefix] === modified[prefix] + ) { + prefix += 1 + } + let suffix = 0 + while ( + suffix + prefix < original.length && + suffix + prefix < modified.length && + original[original.length - suffix - 1] === modified[modified.length - suffix - 1] + ) { + suffix += 1 + } + + const lines: NativeChatEditLine[] = [] + for (let i = 0; i < prefix; i += 1) { + lines.push(context(original[i] ?? '', i + 1, i + 1)) + } + for (let i = prefix; i < original.length - suffix; i += 1) { + lines.push(removal(original[i] ?? '', i + 1)) + } + for (let i = prefix; i < modified.length - suffix; i += 1) { + lines.push(addition(modified[i] ?? '', i + 1)) + } + for (let i = original.length - suffix; i < original.length; i += 1) { + const newIndex = modified.length - suffix + (i - (original.length - suffix)) + lines.push(context(original[i] ?? '', i + 1, newIndex + 1)) + } + return lines +} diff --git a/src/shared/native-chat-edit-model.ts b/src/shared/native-chat-edit-model.ts new file mode 100644 index 00000000000..6a26f87c599 --- /dev/null +++ b/src/shared/native-chat-edit-model.ts @@ -0,0 +1,107 @@ +/** One rendered diff row. Numbers are per side: a removed row has no new-side + * number and an added row has no old-side number. `gap` marks the break + * between two regions of the file, which are otherwise concatenated and read + * as one continuous block even as the gutter jumps hundreds of lines. */ +export type NativeChatEditLineKind = 'context' | 'add' | 'del' | 'gap' + +export type NativeChatEditLine = { + kind: NativeChatEditLineKind + text: string + oldLineNumber: number | null + newLineNumber: number | null +} + +export type NativeChatEditChangeKind = 'added' | 'deleted' | 'edited' | 'renamed' + +export type NativeChatEditFile = { + path: string + /** Set only when the change moved the file. */ + oldPath: string | null + changeKind: NativeChatEditChangeKind + lines: NativeChatEditLine[] + added: number + removed: number + /** False when the numbers locate a row inside a snippet rather than the file, + * which is the case whenever the provider gave us no resolved hunk ranges. */ + lineNumbersKnown: boolean + truncated: boolean +} + +export const MAX_EDIT_LINES = 2_000 +export const MAX_EDIT_CHARS = 96_000 +/** The LCS table is quadratic; above this a linear prefix/suffix diff is used. */ +export const MAX_EDIT_DIFF_CELLS = 200_000 + +/** Rows of a source string, plus whether it was clipped before splitting. */ +export type EditContentLines = { lines: string[]; truncated: boolean } + +/** The one row splitter for every edit shape. Splits on both newline forms so a + * CRLF file never carries a trailing `\r` into a row, where it would render as + * a stray character, defeat the phantom-row guard, and reach the clipboard. */ +export function splitEditContent(content: string): EditContentLines { + if (content.length === 0) { + return { lines: [], truncated: false } + } + const truncated = content.length > MAX_EDIT_CHARS + const body = truncated ? content.slice(0, MAX_EDIT_CHARS) : content + const lines = body.split(/\r?\n/) + // Tested against the clipped body: on the un-clipped string this popped a + // real line whenever the slice fired. + if (body.endsWith('\n')) { + lines.pop() + } + return { lines, truncated } +} + +/** The break between two regions of a file. Carries no text and no position. */ +function editGapLine(): NativeChatEditLine { + return { kind: 'gap', text: '', oldLineNumber: null, newLineNumber: null } +} + +/** Appends a gap when rows already exist, so the break never opens a diff or + * doubles up behind an empty region. */ +export function pushEditGap(lines: NativeChatEditLine[]): void { + if (lines.length > 0 && lines.at(-1)?.kind !== 'gap') { + lines.push(editGapLine()) + } +} + +/** Unified line numbering: a removed row is located on the old side, everything + * else on the new side. One column, so a replaced line repeats its number. */ +export function unifiedLineNumber(line: NativeChatEditLine): number | null { + return line.kind === 'del' ? line.oldLineNumber : (line.newLineNumber ?? line.oldLineNumber) +} + +export function finalizeEditFile( + input: Omit & { + /** Set when the source text was clipped before it became rows. */ + truncated?: boolean + } +): NativeChatEditFile { + const overLineCap = input.lines.length > MAX_EDIT_LINES + const truncated = overLineCap || input.truncated === true + const capped = overLineCap ? input.lines.slice(0, MAX_EDIT_LINES) : input.lines + // A gap marks a break between regions, so one at the end marks nothing. The + // row cap can leave one behind even when the source did not. + let end = capped.length + while (end > 0 && capped[end - 1]?.kind === 'gap') { + end -= 1 + } + const trimmed = end === capped.length ? capped : capped.slice(0, end) + // Without resolved ranges the numbers locate a row inside a snippet; dropping + // them keeps a plausible-looking wrong position out of the gutter, the copy + // text, and the row keys. + const lines = input.lineNumbersKnown + ? trimmed + : trimmed.map((line) => ({ ...line, oldLineNumber: null, newLineNumber: null })) + let added = 0 + let removed = 0 + for (const line of lines) { + if (line.kind === 'add') { + added += 1 + } else if (line.kind === 'del') { + removed += 1 + } + } + return { ...input, lines, added, removed, truncated } +} diff --git a/src/shared/native-chat-edit-normalize.test.ts b/src/shared/native-chat-edit-normalize.test.ts new file mode 100644 index 00000000000..f49e23be296 --- /dev/null +++ b/src/shared/native-chat-edit-normalize.test.ts @@ -0,0 +1,609 @@ +import { describe, expect, it } from 'vitest' +import { editFilesFromToolPair, isEditToolName } from './native-chat-edit-normalize' +import { MAX_EDIT_CHARS, unifiedLineNumber } from './native-chat-edit-model' +import { editLinesFromUnifiedPatch } from './native-chat-unified-patch' +import { unwrapBeginPatch } from './native-chat-begin-patch' + +const gutter = (files: ReturnType): (number | null)[] => + (files ?? []).flatMap((file) => file.lines.map((line) => unifiedLineNumber(line))) + +/** A card takes evidence the edit landed, so these cases report the call as + * complete. Cases about the lifecycle itself pass their own state. */ +const settledFiles = ( + pair: Parameters[0] +): ReturnType => + editFilesFromToolPair({ state: 'completed', ...pair }) + +describe('editLinesFromUnifiedPatch', () => { + it('numbers deletes from the old side and adds from the new side', () => { + const parsed = editLinesFromUnifiedPatch('@@ -12,3 +12,3 @@\n ctx\n-was\n+now\n tail') + expect(parsed?.lineNumbersKnown).toBe(true) + expect(parsed?.lines.map((line) => [line.kind, unifiedLineNumber(line)])).toEqual([ + ['context', 12], + ['del', 13], + ['add', 13], + ['context', 14] + ]) + }) + + it('leaves rows unnumbered when the hunk header carries no ranges', () => { + const parsed = editLinesFromUnifiedPatch('@@\n ctx\n-was\n+now') + expect(parsed?.lineNumbersKnown).toBe(false) + expect(parsed?.lines.every((line) => unifiedLineNumber(line) === null)).toBe(true) + }) + + it('returns null for text with no hunk header', () => { + expect(editLinesFromUnifiedPatch('just prose\n- a bullet')).toBeNull() + }) + + it('keeps the hunk open across a mid-hunk no-newline marker', () => { + const parsed = editLinesFromUnifiedPatch( + '@@ -1,2 +1,2 @@\n keep\n-old\n\\ No newline at end of file\n+new\n\\ No newline at end of file' + ) + expect(parsed?.lines.map((line) => [line.kind, line.text])).toEqual([ + ['context', 'keep'], + ['del', 'old'], + ['add', 'new'] + ]) + }) + + it('reads a removed line that starts with `--` as content, not a file header', () => { + const parsed = editLinesFromUnifiedPatch('@@ -1,4 +1,3 @@\n keep\n--- comment\n-gone\n tail') + expect(parsed?.lines.map((line) => [line.kind, line.text])).toEqual([ + ['context', 'keep'], + ['del', '-- comment'], + ['del', 'gone'], + ['context', 'tail'] + ]) + }) + + it('skips a real file header pair, which only appears outside a hunk', () => { + const parsed = editLinesFromUnifiedPatch( + 'diff --git a/a.ts b/a.ts\n--- a/a.ts\n+++ b/a.ts\n@@ -1,1 +1,1 @@\n-was\n+now' + ) + expect(parsed?.lines.map((line) => [line.kind, line.text])).toEqual([ + ['del', 'was'], + ['add', 'now'] + ]) + }) + + it('splits CRLF rows without leaving a carriage return or a phantom row', () => { + const parsed = editLinesFromUnifiedPatch('@@ -1,2 +1,2 @@\r\n ctx\r\n-was\r\n+now\r\n') + expect(parsed?.lines.map((line) => line.text)).toEqual(['ctx', 'was', 'now']) + }) + + it('reports truncation when the patch text runs past the character cap', () => { + const body = `@@ -1,1 +1,1 @@\n${'+x\n'.repeat(MAX_EDIT_CHARS)}` + expect(editLinesFromUnifiedPatch(body)?.truncated).toBe(true) + }) + + it('marks the break between hunks, and only between them', () => { + const parsed = editLinesFromUnifiedPatch( + '@@ -40,2 +40,2 @@\n keep\n-was\n@@ -310,2 +310,2 @@\n+now\n tail' + ) + expect(parsed?.lines.map((line) => [line.kind, unifiedLineNumber(line)])).toEqual([ + ['context', 40], + ['del', 41], + ['gap', null], + ['add', 310], + ['context', 311] + ]) + }) + + it('reads a body that opens with no hunk header as an unlocatable hunk', () => { + const parsed = editLinesFromUnifiedPatch('-was\n+now', { implicitFirstHunk: true }) + expect(parsed?.lines.map((line) => line.kind)).toEqual(['del', 'add']) + expect(parsed?.lineNumbersKnown).toBe(false) + // Without the option the same body is not a patch at all. + expect(editLinesFromUnifiedPatch('-was\n+now')).toBeNull() + }) +}) + +describe('unwrapBeginPatch', () => { + it('recovers an envelope carried in one word of an argument vector', () => { + const envelope = '*** Begin Patch\n*** Update File: a.ts\n@@\n-x\n+y\n*** End Patch' + expect( + unwrapBeginPatch({ + command: ['bash', '-lc', `apply_patch <<'EOF'\n${envelope}\nEOF`], + workdir: '/repo' + }) + ).toBe(envelope) + }) + + it('leaves an already-decoded envelope alone', () => { + const plain = '*** Begin Patch\n*** Update File: a.ts\n@@\n-x\n+y\n*** End Patch' + expect(unwrapBeginPatch(plain)).toBe(plain) + }) + + it('ignores input with no envelope', () => { + expect(unwrapBeginPatch('ls -la')).toBeNull() + }) + + it('declines an envelope with no closing marker rather than swallowing the command line', () => { + const command = 'bash -c "*** Begin Patch\n*** Update File: a.ts\n@@\n-x\n+y" && echo ok' + expect(unwrapBeginPatch(command)).toBeNull() + expect(settledFiles({ name: 'shell', input: command })).toBeNull() + }) +}) + +describe('editFilesFromToolPair', () => { + it('renders an apply_patch run through a command tool, which produced no diff', () => { + const files = settledFiles({ + name: 'exec', + input: { + command: [ + 'bash', + '-lc', + "apply_patch <<'EOF'\n*** Begin Patch\n*** Update File: src/a.ts\n@@\n ctx\n-was\n+now\n*** End Patch\nEOF" + ] + } + }) + expect(files).toHaveLength(1) + expect(files?.[0]?.path).toBe('src/a.ts') + expect(files?.[0]?.changeKind).toBe('edited') + expect(files?.[0]?.added).toBe(1) + expect(files?.[0]?.removed).toBe(1) + // Codex hunk headers are context anchors, so no row may claim a file position. + expect(files?.[0]?.lineNumbersKnown).toBe(false) + }) + + it('numbers an added file from 1', () => { + const files = settledFiles({ + name: 'exec', + input: + "apply_patch <<'EOF'\n*** Begin Patch\n*** Add File: new.ts\n+one\n+two\n*** End Patch\nEOF" + }) + expect(files?.[0]?.changeKind).toBe('added') + expect(files?.[0]?.lineNumbersKnown).toBe(true) + expect(gutter(files)).toEqual([1, 2]) + }) + + it('keeps a file whose update body carries no hunk header', () => { + const files = settledFiles({ + name: 'apply_patch', + input: { + input: + '*** Begin Patch\n*** Update File: first.ts\n ctx\n-was\n+now\n*** Update File: second.ts\n@@ -1,1 +1,1 @@\n-a\n+b\n*** End Patch' + } + }) + expect(files?.map((file) => file.path)).toEqual(['first.ts', 'second.ts']) + expect(files?.[0]?.lines.map((line) => line.kind)).toEqual(['context', 'del', 'add']) + expect(files?.[0]?.lineNumbersKnown).toBe(false) + }) + + it('does not render envelope control lines as file content', () => { + const files = settledFiles({ + name: 'apply_patch', + input: { + input: + '*** Begin Patch\n*** Environment ID: abc123\n*** Update File: a.ts\n@@\n-was\n+now\n*** End of File\n*** End Patch' + } + }) + expect(files?.[0]?.lines.map((line) => line.text)).toEqual(['was', 'now']) + }) + + it('reports a delete that names the file and carries no body', () => { + const files = settledFiles({ + name: 'apply_patch', + input: { input: '*** Begin Patch\n*** Delete File: gone.ts\n*** End Patch' } + }) + expect(files).toHaveLength(1) + expect(files?.[0]?.changeKind).toBe('deleted') + expect(files?.[0]?.path).toBe('gone.ts') + expect(files?.[0]?.lines).toEqual([]) + }) + + it('reads a CRLF envelope, whose markers otherwise match nothing', () => { + const files = settledFiles({ + name: 'apply_patch', + input: { + input: + '*** Begin Patch\r\n*** Update File: a.ts\r\n@@ -1,2 +1,2 @@\r\n-was\r\n+now\r\n*** End Patch' + } + }) + expect(files).toHaveLength(1) + expect(files?.[0]?.path).toBe('a.ts') + expect(files?.[0]?.lines.map((line) => line.text)).toEqual(['was', 'now']) + }) + + it('marks the break between resolved hunks that sit far apart', () => { + const files = settledFiles({ + name: 'Edit', + input: { file_path: '/repo/a.ts' }, + result: { + editPatch: { + filePath: '/repo/a.ts', + hunks: [ + { oldStart: 42, oldLines: 1, newStart: 42, newLines: 1, lines: ['-was', '+now'] }, + { oldStart: 310, oldLines: 1, newStart: 310, newLines: 1, lines: ['-old', '+new'] } + ] + } + } + }) + expect(files?.[0]?.lines.map((line) => line.kind)).toEqual(['del', 'add', 'gap', 'del', 'add']) + // A break marks nothing at either end, and counts no change of its own. + expect(files?.[0]?.added).toBe(2) + expect(files?.[0]?.removed).toBe(2) + expect(gutter(files)).toEqual([42, 42, null, 310, 310]) + }) + + it('reads a move header as a rename', () => { + const files = settledFiles({ + name: 'exec', + input: + "apply_patch <<'EOF'\n*** Begin Patch\n*** Update File: old.ts\n*** Move to: new.ts\n@@\n-a\n+b\n*** End Patch\nEOF" + }) + expect(files?.[0]?.changeKind).toBe('renamed') + expect(files?.[0]?.oldPath).toBe('old.ts') + expect(files?.[0]?.path).toBe('new.ts') + }) + + it('interleaves a Claude snippet pair without claiming line positions', () => { + const files = settledFiles({ + name: 'Edit', + input: { + file_path: '/repo/a.ts', + old_string: 'keep\nwas\ntail', + new_string: 'keep\nnow\ntail' + } + }) + expect(files?.[0]?.lines.map((line) => line.kind)).toEqual(['context', 'del', 'add', 'context']) + expect(files?.[0]?.lineNumbersKnown).toBe(false) + }) + + it('prefers the resolved hunks on the result over the snippet pair', () => { + const files = settledFiles({ + name: 'Edit', + input: { file_path: '/repo/a.ts', old_string: 'was', new_string: 'now' }, + result: { + editPatch: { + filePath: '/repo/a.ts', + hunks: [ + { + oldStart: 12, + oldLines: 3, + newStart: 12, + newLines: 3, + lines: [' ctx', '-was', '+now', ' tail'] + } + ] + } + } + }) + expect(files?.[0]?.lineNumbersKnown).toBe(true) + expect(gutter(files)).toEqual([12, 13, 13, 14]) + }) + + it('treats a Write the provider reported as a creation as an added file', () => { + const files = settledFiles({ + name: 'Write', + input: { file_path: '/repo/new.ts', content: 'one\ntwo\n' }, + result: { output: 'File created successfully at: /repo/new.ts' } + }) + expect(files?.[0]?.changeKind).toBe('added') + expect(gutter(files)).toEqual([1, 2]) + }) + + it('does not claim a creation for a Write over an existing file', () => { + const overwrite = settledFiles({ + name: 'Write', + input: { file_path: '/repo/a.ts', content: 'one\ntwo\n' }, + result: { output: 'The file /repo/a.ts has been updated.' } + }) + expect(overwrite?.[0]?.changeKind).toBe('edited') + // With no result at all there is no evidence of a creation either. + const unreported = settledFiles({ + name: 'Write', + input: { file_path: '/repo/a.ts', content: 'one\n' } + }) + expect(unreported?.[0]?.changeKind).toBe('edited') + }) + + it('reads a MultiEdit, whose snippet pairs sit in edits[]', () => { + const files = settledFiles({ + name: 'MultiEdit', + input: { + file_path: '/repo/a.ts', + edits: [ + { old_string: 'was', new_string: 'now' }, + { old_string: 'gone', new_string: 'kept' } + ] + } + }) + expect(files).toHaveLength(1) + expect(files?.[0]?.path).toBe('/repo/a.ts') + // Each entry is its own region, so a break separates them. + expect(files?.[0]?.lines.map((line) => [line.kind, line.text])).toEqual([ + ['del', 'was'], + ['add', 'now'], + ['gap', ''], + ['del', 'gone'], + ['add', 'kept'] + ]) + expect(files?.[0]?.added).toBe(2) + expect(files?.[0]?.removed).toBe(2) + }) + + it('leaves NotebookEdit to the generic tool view', () => { + expect(isEditToolName('NotebookEdit')).toBe(false) + }) + + it('drops the gutter numbers whenever they locate a snippet rather than the file', () => { + const files = settledFiles({ + name: 'Edit', + input: { file_path: '/repo/a.ts', old_string: 'keep\nwas', new_string: 'keep\nnow' } + }) + expect(files?.[0]?.lineNumbersKnown).toBe(false) + expect(gutter(files)).toEqual([null, null, null]) + expect( + files?.[0]?.lines.every((line) => line.oldLineNumber === null && line.newLineNumber === null) + ).toBe(true) + }) + + it('renders no card for an edit the provider rejected or has not landed', () => { + const failedInput = { file_path: '/repo/a.ts', old_string: 'missing', new_string: 'now' } + expect( + settledFiles({ + name: 'Edit', + input: failedInput, + result: { output: 'String to replace not found in file.', isError: true } + }) + ).toBeNull() + expect( + settledFiles({ + name: 'apply_patch', + input: { + changes: [{ path: 'a.ts', kind: { type: 'update' }, diff: '@@ -1 +1 @@\n-a\n+b' }] + }, + state: 'failed' + }) + ).toBeNull() + expect(settledFiles({ name: 'Edit', input: failedInput, state: 'running' })).toBeNull() + }) + + it('does not read a command tool result as a file edit', () => { + const patch = 'diff --git a/a.ts b/a.ts\n--- a/a.ts\n+++ b/a.ts\n@@ -1 +1 @@\n-was\n+now' + expect( + settledFiles({ + name: 'exec', + input: { command: 'git diff' }, + result: { output: patch } + }) + ).toBeNull() + // The structured journal's `Diff` item carries its patch only on the result. + const diffed = settledFiles({ + name: 'Diff', + input: { path: '/repo/a.ts' }, + result: { output: patch } + }) + expect(diffed?.[0]?.path).toBe('/repo/a.ts') + expect(diffed?.[0]?.added).toBe(1) + }) + + it('reports truncation when the content runs past the character cap', () => { + const files = settledFiles({ + name: 'Write', + input: { file_path: '/repo/a.ts', content: `${'x'.repeat(MAX_EDIT_CHARS)}\nlast\n` } + }) + expect(files?.[0]?.truncated).toBe(true) + // The clipped body ends mid-line, so its one row is real and must survive. + expect(files?.[0]?.lines).toHaveLength(1) + }) + + it('reads Codex structured changes, stripping the move marker from the body', () => { + const files = settledFiles({ + name: 'apply_patch', + input: { + changes: [ + { + path: 'old.ts', + kind: { type: 'update', move_path: 'new.ts' }, + diff: '@@ -1,2 +1,2 @@\n-a\n+b\n\nMoved to: new.ts' + } + ] + } + }) + expect(files?.[0]?.changeKind).toBe('renamed') + expect(files?.[0]?.lines.some((line) => line.text.includes('Moved to'))).toBe(false) + expect(gutter(files)).toEqual([1, 1]) + }) + + it('reads a Codex add change, which arrives as raw content with no hunk header', () => { + const files = settledFiles({ + name: 'apply_patch', + input: { changes: [{ path: 'new.ts', kind: { type: 'add' }, diff: 'one\ntwo' }] } + }) + expect(files?.[0]?.changeKind).toBe('added') + expect(files?.[0]?.added).toBe(2) + }) + + it('renders the file a write actually wrote, not one its content quotes', () => { + const files = settledFiles({ + name: 'Write', + input: { + file_path: 'docs/patch-format.md', + content: + 'Example:\n\n*** Begin Patch\n*** Update File: src/victim.ts\n@@\n-a\n+b\n*** End Patch\n' + }, + result: { output: 'File created successfully at: docs/patch-format.md' } + }) + expect(files?.map((file) => file.path)).toEqual(['docs/patch-format.md']) + expect(files?.[0]?.lines.some((line) => line.text.includes('Begin Patch'))).toBe(true) + }) + + it('finds an envelope in a command payload that arrived as JSON text', () => { + const envelope = '*** Begin Patch\n*** Update File: src/a.ts\n@@\n-was\n+now\n*** End Patch' + const files = settledFiles({ + name: 'shell', + input: JSON.stringify({ + command: ['bash', '-lc', `apply_patch <<'EOF'\n${envelope}\nEOF`], + workdir: '/repo' + }) + }) + expect(files?.[0]?.path).toBe('src/a.ts') + expect(files?.[0]?.added).toBe(1) + }) + + it('renders no card for a call the turn never answered', () => { + const input = { file_path: '/repo/a.ts', old_string: 'was', new_string: 'now' } + // No lifecycle and no result: nothing says the edit was applied. + expect(editFilesFromToolPair({ name: 'Edit', input })).toBeNull() + expect(editFilesFromToolPair({ name: 'Edit', input, state: 'completed' })).toHaveLength(1) + expect(editFilesFromToolPair({ name: 'Edit', input, result: { output: 'ok' } })).toHaveLength(1) + }) + + it('splits a multi-file patch into one card per file', () => { + const files = settledFiles({ + name: 'Diff', + input: { path: '/repo/one.ts' }, + result: { + output: + 'diff --git a/one.ts b/one.ts\n--- a/one.ts\n+++ b/one.ts\n@@ -1,1 +1,1 @@\n-first\n+FIRST\n' + + 'diff --git a/two.ts b/two.ts\n--- a/two.ts\n+++ b/two.ts\n@@ -10,1 +10,1 @@\n-second\n+SECOND' + } + }) + expect(files?.map((file) => file.path)).toEqual(['one.ts', 'two.ts']) + expect(files?.[0]?.lines.map((line) => line.text)).toEqual(['first', 'FIRST']) + expect(gutter(files?.slice(1) ?? null)).toEqual([10, 10]) + }) + + it('keeps a file whose envelope section carries no body at all', () => { + const files = settledFiles({ + name: 'apply_patch', + input: { + input: + '*** Begin Patch\n*** Update File: first.ts\n@@\n-a\n+b\n*** Update File: second.ts\n*** Update File: third.ts\n@@\n-c\n+d\n*** End Patch' + } + }) + expect(files?.map((file) => file.path)).toEqual(['first.ts', 'second.ts', 'third.ts']) + expect(files?.[1]?.lines).toEqual([]) + }) + + it('splits a multi-file patch written without per-file preamble lines', () => { + const files = settledFiles({ + name: 'Diff', + input: { path: '/repo/one.ts' }, + result: { + output: + '--- a/one.ts\n+++ b/one.ts\n@@ -1,1 +1,1 @@\n-first\n+FIRST\n' + + '--- a/two.ts\n+++ b/two.ts\n@@ -10,1 +10,1 @@\n-second\n+SECOND' + } + }) + expect(files?.map((file) => file.path)).toEqual(['one.ts', 'two.ts']) + expect(gutter(files?.slice(1) ?? null)).toEqual([10, 10]) + }) + + it('refuses a card when the call names a file count instead of a file', () => { + expect( + settledFiles({ + name: 'Diff', + input: { path: '2 files' }, + result: { output: '@@\n-a\n+b\n@@\n-c\n+d' } + }) + ).toBeNull() + }) + + it('reports a clipped patch as truncated instead of rendering its marker', () => { + const files = settledFiles({ + name: 'Diff', + input: { path: 'src/a.ts' }, + result: { output: '@@ -1,3 +1,3 @@\n ctx\n-was\n+now\n… (48210 bytes)' } + }) + expect(files?.[0]?.truncated).toBe(true) + expect(files?.[0]?.lines.map((line) => line.text)).toEqual(['ctx', 'was', 'now']) + }) + + it('reads a move appended to the patch body as a rename, as the other lane does', () => { + const files = settledFiles({ + name: 'Diff', + input: { path: 'src/old.ts' }, + result: { output: '@@ -1,1 +1,1 @@\n-a\n+b\n\nMoved to: src/new.ts' } + }) + expect(files?.[0]?.changeKind).toBe('renamed') + expect(files?.[0]?.path).toBe('src/new.ts') + expect(files?.[0]?.oldPath).toBe('src/old.ts') + expect(files?.[0]?.lines.some((line) => line.text.includes('Moved to'))).toBe(false) + }) + + it('does not read a row that merely mentions a move as one', () => { + const body = '@@ -1,2 +1,2 @@\n ctx\n+See Moved to: docs/archive/index.md' + const fromPatch = settledFiles({ + name: 'Diff', + input: { path: 'docs/index.md' }, + result: { output: body } + }) + const fromChanges = settledFiles({ + name: 'apply_patch', + input: { changes: [{ path: 'docs/index.md', kind: { type: 'update' }, diff: body }] } + }) + for (const files of [fromPatch, fromChanges]) { + expect(files?.[0]?.changeKind).toBe('edited') + expect(files?.[0]?.oldPath).toBeNull() + expect(files?.[0]?.path).toBe('docs/index.md') + expect(files?.[0]?.lines.at(-1)?.text).toBe('See Moved to: docs/archive/index.md') + } + }) + + it('accepts either spelling of the command that applies an envelope', () => { + const envelope = '*** Begin Patch\n*** Update File: src/a.ts\n@@\n-was\n+now\n*** End Patch' + const files = settledFiles({ + name: 'shell', + input: { command: ['bash', '-lc', `applypatch <<'EOF'\n${envelope}\nEOF`] } + }) + expect(files?.[0]?.path).toBe('src/a.ts') + }) + + it('keeps the header destination for a rename the call names by its old path', () => { + const files = settledFiles({ + name: 'Diff', + input: { path: 'old.txt' }, + result: { + output: + 'diff --git a/old.txt b/new.txt\n--- a/old.txt\n+++ b/new.txt\n@@ -1,1 +1,1 @@\n-a\n+b' + } + }) + expect(files?.[0]?.path).toBe('new.txt') + expect(files?.[0]?.oldPath).toBe('old.txt') + }) + + it('does not read a command that only quotes an envelope as an edit', () => { + const envelope = '*** Begin Patch\n*** Update File: src/real.ts\n@@\n-a\n+b\n*** End Patch' + expect( + settledFiles({ + name: 'shell', + input: { command: ['bash', '-lc', `cat > notes.md <<'EOF'\n${envelope}\nEOF`] } + }) + ).toBeNull() + }) + + it('does not call two compared directories a rename', () => { + const files = settledFiles({ + name: 'Diff', + input: {}, + result: { output: '--- d1/x.ts\n+++ d2/x.ts\n@@ -1,1 +1,1 @@\n-a\n+b' } + }) + expect(files?.[0]?.changeKind).toBe('edited') + expect(files?.[0]?.oldPath).toBeNull() + expect(files?.[0]?.path).toBe('d2/x.ts') + }) + + it('lets the call name the file when a preamble precedes the only header', () => { + const files = settledFiles({ + name: 'Diff', + input: { path: '/repo/one.ts' }, + result: { + output: + 'warning: something\ndiff --git a/one.ts b/one.ts\n--- a/one.ts\n+++ b/one.ts\n@@ -1,1 +1,1 @@\n-a\n+b' + } + }) + // The preamble is its own nameless section, and must not make this look + // like a patch over several files. + expect(files?.map((file) => file.path)).toEqual(['/repo/one.ts']) + }) + + it('returns null for a tool that did not edit a file', () => { + expect(settledFiles({ name: 'Bash', input: { command: 'ls' } })).toBeNull() + expect(isEditToolName('Bash')).toBe(false) + expect(isEditToolName('Edit')).toBe(true) + }) +}) diff --git a/src/shared/native-chat-edit-normalize.ts b/src/shared/native-chat-edit-normalize.ts new file mode 100644 index 00000000000..95c2ca18bd5 --- /dev/null +++ b/src/shared/native-chat-edit-normalize.ts @@ -0,0 +1,349 @@ +import { editFilesFromBeginPatch, unwrapBeginPatch } from './native-chat-begin-patch' +import { editLinesFromContents } from './native-chat-edit-lcs' +import { + finalizeEditFile, + pushEditGap, + type NativeChatEditFile, + type NativeChatEditLine +} from './native-chat-edit-model' +import { stripBoundedTextMarker } from './structured-agent-session-projection' +import { + editLinesFromUnifiedPatch, + editLinesFromWholeFile, + unifiedPatchSections, + type UnifiedPatchSection +} from './native-chat-unified-patch' +import type { NativeChatEditPatch } from './native-chat-types' + +// `NotebookEdit` is deliberately absent: its input carries only the new cell +// source, so a card would render an unchanged cell as wholly added. It falls +// through to the generic tool view instead. +const CLAUDE_EDIT_TOOLS = new Set(['Edit', 'MultiEdit', 'Write', 'str_replace']) +/** Command tools, which run a patch as one of many things they can run, so a + * quoted envelope is not evidence that one was applied. */ +const COMMAND_PATCH_TOOLS = new Set(['exec', 'shell', 'local_shell']) +/** Tools whose input may wrap a `*** Begin Patch` envelope. The dedicated patch + * tool applies whatever it is given; a command tool must say that it is. */ +const PATCH_ENVELOPE_TOOLS = new Set(['apply_patch', ...COMMAND_PATCH_TOOLS]) +/** A count standing in for a path, from a producer that joined several files' + * patches and kept no per-file path. */ +const FILE_COUNT_PATH = /^\d+ files?$/ +/** Tools whose whole payload is patch text. `Diff` reaches its patch only + * through the result, because the structured journal projects a diff item as a + * call carrying just the path. */ +const PATCH_TEXT_TOOLS = new Set(['apply_patch', 'Diff']) + +export function isEditToolName(name: string): boolean { + return CLAUDE_EDIT_TOOLS.has(name) || PATCH_ENVELOPE_TOOLS.has(name) || PATCH_TEXT_TOOLS.has(name) +} + +function record(value: unknown): Record | null { + return typeof value === 'object' && value !== null ? (value as Record) : null +} + +function text(value: unknown): string | null { + return typeof value === 'string' ? value : null +} + +/** Rows straight from resolved hunks, which is the only path with true numbers + * for a provider that reports its edits as a snippet pair. */ +function linesFromEditPatch(patch: NativeChatEditPatch): NativeChatEditLine[] { + const lines: NativeChatEditLine[] = [] + for (const hunk of patch.hunks) { + // Hunks are separate regions of the file; run together the gutter jumps + // from one to the next with nothing marking the skipped span. + pushEditGap(lines) + let oldNo = hunk.oldStart + let newNo = hunk.newStart + for (const raw of hunk.lines) { + if (raw.startsWith('+')) { + lines.push({ kind: 'add', text: raw.slice(1), oldLineNumber: null, newLineNumber: newNo }) + newNo += 1 + } else if (raw.startsWith('-')) { + lines.push({ kind: 'del', text: raw.slice(1), oldLineNumber: oldNo, newLineNumber: null }) + oldNo += 1 + } else { + lines.push({ + kind: 'context', + text: raw.startsWith(' ') ? raw.slice(1) : raw, + oldLineNumber: oldNo, + newLineNumber: newNo + }) + oldNo += 1 + newNo += 1 + } + } + } + return lines +} + +/** A whole-content write looks identical whether it created the file or + * overwrote one, so only positive evidence may claim a creation. With no + * evidence either way this errs toward the weaker claim: calling a creation an + * edit is imprecise, while calling an overwrite a creation is false and paints + * an existing file as wholly new. */ +const CREATED_FILE_RESULT = /^\s*File created successfully/ + +function wholeContentChangeKind( + input: Record, + output: string | undefined +): 'added' | 'edited' { + if (text(input.command) === 'create') { + return 'added' + } + return output !== undefined && CREATED_FILE_RESULT.test(output) ? 'added' : 'edited' +} + +/** `MultiEdit` carries its snippet pairs in `edits[]`, not at the top level. */ +function multiEditFiles(input: Record, path: string): NativeChatEditFile[] | null { + if (!Array.isArray(input.edits)) { + return null + } + const lines: NativeChatEditLine[] = [] + let truncated = false + for (const entry of input.edits) { + const edit = record(entry) + const oldString = text(edit?.old_string) ?? text(edit?.oldString) + const newString = text(edit?.new_string) ?? text(edit?.newString) + if (oldString === null && newString === null) { + continue + } + // Each entry is its own snippet, so it starts a new region. + pushEditGap(lines) + const diffed = editLinesFromContents(oldString ?? '', newString ?? '') + lines.push(...diffed.lines) + truncated ||= diffed.truncated + } + if (lines.length === 0) { + return null + } + return [ + finalizeEditFile({ + path, + oldPath: null, + changeKind: 'edited', + lines, + // A snippet pair cannot say where in the file it sits. + lineNumbersKnown: false, + truncated + }) + ] +} + +function claudeEditFiles( + name: string, + input: Record, + output: string | undefined +): NativeChatEditFile[] | null { + const path = text(input.file_path) ?? text(input.path) ?? 'file' + if (name === 'MultiEdit') { + return multiEditFiles(input, path) + } + const oldString = text(input.old_string) ?? text(input.oldString) + const newString = text(input.new_string) ?? text(input.newString) + const content = text(input.content) ?? text(input.file_text) + if (oldString === null && content !== null) { + const whole = editLinesFromWholeFile(content, 'add') + return [ + finalizeEditFile({ + path, + oldPath: null, + changeKind: wholeContentChangeKind(input, output), + lines: whole.lines, + lineNumbersKnown: true, + truncated: whole.truncated + }) + ] + } + if (oldString === null && newString === null) { + return null + } + const diffed = editLinesFromContents(oldString ?? '', newString ?? content ?? '') + return [ + finalizeEditFile({ + path, + oldPath: null, + changeKind: 'edited', + lines: diffed.lines, + // A snippet pair cannot say where in the file it sits. + lineNumbersKnown: false, + truncated: diffed.truncated + }) + ] +} + +/** A move is appended to the patch body as prose rather than a header field, on + * every lane that carries the body as text. Left in place it renders as a + * numbered line of the file it moved. + * + * Anchored to the start of the final line: unanchored, a row whose own content + * mentions a move was cut in half and the file it names claimed as a rename + * that never happened. */ +const MOVE_MARKER = /(?:^|\n)Moved to: (.+)$/ + +function splitMoveMarker(patch: string): { body: string; movedTo: string | null } { + const match = MOVE_MARKER.exec(patch) + return match + ? { body: patch.slice(0, match.index), movedTo: match[1]!.trim() } + : { body: patch, movedTo: null } +} + +function codexChangeFiles(changes: unknown[]): NativeChatEditFile[] { + return changes.flatMap((entry) => { + const change = record(entry) + const path = text(change?.path) + const diff = text(change?.diff) + if (!change || !path || !diff) { + return [] + } + const kind = record(change.kind) + const kindType = text(kind?.type) ?? text(change.kind) ?? 'update' + const movePath = text(kind?.move_path) ?? text(change.movePath) + if (kindType === 'add' || kindType === 'delete') { + // Add and delete arrive as raw file content, with no hunk header or signs. + const whole = editLinesFromWholeFile(diff, kindType === 'add' ? 'add' : 'del') + return [ + finalizeEditFile({ + path, + oldPath: null, + changeKind: kindType === 'add' ? 'added' : 'deleted', + lines: whole.lines, + lineNumbersKnown: true, + truncated: whole.truncated + }) + ] + } + const parsed = editLinesFromUnifiedPatch(splitMoveMarker(diff).body) + if (!parsed) { + return [] + } + return [ + finalizeEditFile({ + path: movePath ?? path, + oldPath: movePath ? path : null, + changeKind: movePath ? 'renamed' : 'edited', + lines: parsed.lines, + lineNumbersKnown: parsed.lineNumbersKnown, + truncated: parsed.truncated + }) + ] + }) +} + +/** One diff model for a tool call and its result, across every shape the + * supported agents use to report a file edit. */ +export function editFilesFromToolPair(pair: { + name: string + input: unknown + /** Provider lifecycle for the call, when the lane reports one. */ + state?: 'running' | 'completed' | 'failed' + result?: { output?: string; isError?: boolean; editPatch?: NativeChatEditPatch } +}): NativeChatEditFile[] | null { + // A card states the edit as made, so it takes evidence that it landed: the + // provider reporting the call complete, or a result that is not an error. + // Anything else — failed, still running, or a turn that stopped before the + // call was answered — keeps the generic tool view and its error body. + if (pair.state === 'failed' || pair.state === 'running' || pair.result?.isError === true) { + return null + } + if (pair.state !== 'completed' && pair.result === undefined) { + return null + } + const input = record(pair.input) + const patch = pair.result?.editPatch + if (patch && patch.hunks.length > 0) { + return [ + finalizeEditFile({ + path: patch.filePath ?? text(input?.file_path) ?? 'file', + oldPath: null, + changeKind: 'edited', + lines: linesFromEditPatch(patch), + lineNumbersKnown: true + }) + ] + } + + // Only a tool that runs a patch may be searched for an envelope: a file's own + // contents can quote one, and scanning a write's payload rendered a card for + // the quoted file while the file actually written never appeared. + if (PATCH_ENVELOPE_TOOLS.has(pair.name)) { + const envelope = unwrapBeginPatch(pair.input, { + requireApplyCommand: COMMAND_PATCH_TOOLS.has(pair.name) + }) + const files = envelope ? editFilesFromBeginPatch(envelope) : [] + if (files.length > 0) { + return files + } + } + + if (input && Array.isArray(input.changes)) { + const files = codexChangeFiles(input.changes) + if (files.length > 0) { + return files + } + } + + if (input && CLAUDE_EDIT_TOOLS.has(pair.name)) { + return claudeEditFiles(pair.name, input, pair.result?.output) + } + + if (!PATCH_TEXT_TOOLS.has(pair.name)) { + return null + } + // The result fallback is scoped to `Diff`, whose call carries only a path. + // Reading any command tool's output as a patch reclassified `git diff` as a + // file edit and swallowed the command line with it. + const patchText = + text(input?.patch) ?? text(input?.diff) ?? (pair.name === 'Diff' ? pair.result?.output : null) + if (!patchText) { + return null + } + // The body carries its own marker when the journal clipped it. Read as + // content it becomes a numbered line of the file, and the rows that follow + // are reported complete. + const bounded = stripBoundedTextMarker(patchText) + const moved = splitMoveMarker(bounded.text) + // One card per file the patch touches: run together, the later files' rows + // and gutter numbers sit under the first file's name. + const split = unifiedPatchSections(moved.body) + const callerPath = text(input?.path) ?? text(input?.file_path) + if (callerPath !== null && FILE_COUNT_PATH.test(callerPath)) { + // The producer joined several files' patches and kept a count in place of a + // path, so nothing here can name a file. Naming the card after the count + // would assert a file that does not exist. + return null + } + // A patch that names one file is the file the call is reporting on, so the + // call's own path wins — it is the provider's, where the header's is relative + // to the patch. A patch naming several has no one path, and a rename's + // destination is only ever in the header. Sections that name nothing are + // preamble and must not change that count. + const namedSections = split.sections.filter((section) => section.path !== null).length + const named = (section: UnifiedPatchSection): string => + (namedSections <= 1 && section.oldPath === null + ? (callerPath ?? section.path) + : (section.path ?? callerPath)) ?? 'file' + const files = split.sections.flatMap((section) => { + const parsed = editLinesFromUnifiedPatch(section.body) + if (!parsed && section.path === null) { + return [] + } + return [ + finalizeEditFile({ + path: named(section), + oldPath: section.oldPath, + changeKind: section.changeKind, + lines: parsed?.lines ?? [], + lineNumbersKnown: parsed?.lineNumbersKnown ?? false, + truncated: bounded.truncated || split.truncated || (parsed?.truncated ?? false) + }) + ] + }) + // The move marker names where the whole patch moved, so it can only speak for + // a patch describing one file. + if (moved.movedTo !== null && files.length === 1 && files[0]) { + const only = files[0] + return [{ ...only, path: moved.movedTo, oldPath: only.path, changeKind: 'renamed' }] + } + return files.length > 0 ? files : null +} diff --git a/src/shared/native-chat-href-routing.test.ts b/src/shared/native-chat-href-routing.test.ts index 1b7c276ffd3..5f6e7ad4595 100644 --- a/src/shared/native-chat-href-routing.test.ts +++ b/src/shared/native-chat-href-routing.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from 'vitest' -import { routeNativeChatHref } from './native-chat-href-routing' +import { createNativeChatFileHref, routeNativeChatHref } from './native-chat-href-routing' describe('routeNativeChatHref', () => { it('classifies web and mail links', () => { @@ -47,6 +47,35 @@ describe('routeNativeChatHref', () => { }) }) + it('routes encoded renderer file targets without treating Windows drives as schemes', () => { + expect( + routeNativeChatHref(createNativeChatFileHref(String.raw`C:\repo\report.docx:12`)) + ).toEqual({ + kind: 'file', + pathText: String.raw`C:\repo\report.docx:12`, + line: null + }) + expect(routeNativeChatHref(createNativeChatFileHref('/tmp/report.html'))).toEqual({ + kind: 'file', + pathText: '/tmp/report.html', + line: null + }) + }) + + it('bounds nested renderer file target decoding', () => { + let href = '/tmp/report.html' + for (let depth = 0; depth < 4; depth += 1) { + href = createNativeChatFileHref(` ${href}`) + } + expect(routeNativeChatHref(href)).toEqual({ + kind: 'file', + pathText: '/tmp/report.html', + line: null + }) + + expect(routeNativeChatHref(createNativeChatFileHref(` ${href}`))).toEqual({ kind: 'none' }) + }) + it('drops anchors, unknown schemes, malformed file URIs, and empty hrefs', () => { expect(routeNativeChatHref('#section')).toEqual({ kind: 'none' }) expect(routeNativeChatHref(undefined)).toEqual({ kind: 'none' }) diff --git a/src/shared/native-chat-href-routing.ts b/src/shared/native-chat-href-routing.ts index 005ab957f65..edf36565dab 100644 --- a/src/shared/native-chat-href-routing.ts +++ b/src/shared/native-chat-href-routing.ts @@ -8,6 +8,24 @@ export type NativeChatHrefRoute = const WEB_SCHEME_PATTERN = /^(?:https?|mailto):/i const SCHEME_PATTERN = /^[A-Za-z][A-Za-z0-9+.-]*:/ +export const NATIVE_CHAT_FILE_HREF_PREFIX = '#orca-native-chat-file=' +const MAX_NATIVE_CHAT_FILE_HREF_DECODES = 4 + +export function createNativeChatFileHref(pathText: string): string { + return `${NATIVE_CHAT_FILE_HREF_PREFIX}${encodeURIComponent(pathText)}` +} + +function decodeNativeChatFileHref(href: string): string | null { + if (!href.startsWith(NATIVE_CHAT_FILE_HREF_PREFIX)) { + return null + } + try { + const decoded = decodeURIComponent(href.slice(NATIVE_CHAT_FILE_HREF_PREFIX.length)) + return decoded && !decoded.startsWith(NATIVE_CHAT_FILE_HREF_PREFIX) ? decoded : null + } catch { + return null + } +} function parseLineFragment(hash: string): number | null { if (!hash) { @@ -45,8 +63,21 @@ function maybeDecodeHrefPath(value: string): string { } export function routeNativeChatHref(href: string | null | undefined): NativeChatHrefRoute { - const trimmed = href?.trim() - if (!trimmed || trimmed.startsWith('#')) { + let trimmed = href?.trim() + if (!trimmed) { + return { kind: 'none' } + } + for (let depth = 0; depth < MAX_NATIVE_CHAT_FILE_HREF_DECODES; depth += 1) { + const encodedFileHref = decodeNativeChatFileHref(trimmed) + if (!encodedFileHref) { + break + } + trimmed = encodedFileHref.trim() + } + if (!trimmed || trimmed.startsWith(NATIVE_CHAT_FILE_HREF_PREFIX)) { + return { kind: 'none' } + } + if (trimmed.startsWith('#')) { return { kind: 'none' } } if (WEB_SCHEME_PATTERN.test(trimmed)) { diff --git a/src/shared/native-chat-session-option-snapshot.test.ts b/src/shared/native-chat-session-option-snapshot.test.ts index 5286ad9154b..c9980be3b3b 100644 --- a/src/shared/native-chat-session-option-snapshot.test.ts +++ b/src/shared/native-chat-session-option-snapshot.test.ts @@ -1,5 +1,8 @@ import { describe, expect, it } from 'vitest' -import type { SessionOptionDescriptor } from './native-chat-session-options' +import { + sessionOptionDispatchUnconfirmed, + type SessionOptionDescriptor +} from './native-chat-session-options' import { mergeDiscoveredAuthoritativeModels } from './agent-session-option-catalog' import { CLAUDE_SESSION_OPTION_CATALOG, @@ -22,13 +25,37 @@ function claudeRecord(): NativeChatSessionOptionRecord { } describe('buildNativeChatSessionOptionSnapshot', () => { + // The producer names its lane once, here; `dispatched` is emitted by both and + // is not evidence of which one, so the descriptor has to carry the answer. + it.each(['catalog', 'agent-session'] as const)( + 'stamps every descriptor with the %s transport it was built for', + (liveTransport) => { + const record = claudeRecord() + record.model = { value: 'sonnet', source: 'dispatched' } + const snapshot = buildNativeChatSessionOptionSnapshot({ + catalog: CLAUDE_SESSION_OPTION_CATALOG, + models: CLAUDE_SESSION_OPTION_CATALOG.models, + record, + mode: 'live', + modelLabel: 'Model', + liveTransport + }) + expect(snapshot.length).toBeGreaterThan(1) + expect(snapshot.every((descriptor) => descriptor.transport === liveTransport)).toBe(true) + const dispatched = snapshot.filter((descriptor) => descriptor.valueSource === 'dispatched') + expect(dispatched.length).toBeGreaterThan(0) + expect(dispatched.every(sessionOptionDispatchUnconfirmed)).toBe(liveTransport === 'catalog') + } + ) + it('offers every catalog model with the current value unknown', () => { const snapshot = buildNativeChatSessionOptionSnapshot({ catalog: CLAUDE_SESSION_OPTION_CATALOG, models: CLAUDE_SESSION_OPTION_CATALOG.models, record: claudeRecord(), mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) expect(snapshot).toHaveLength(1) const model = snapshot[0]! @@ -50,7 +77,8 @@ describe('buildNativeChatSessionOptionSnapshot', () => { models: CLAUDE_SESSION_OPTION_CATALOG.models, record, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) expect(snapshot.map((descriptor) => descriptor.id)).toEqual(['model', 'effort']) expect(snapshot[0]).toMatchObject({ valueSource: 'dispatched' }) @@ -66,7 +94,8 @@ describe('buildNativeChatSessionOptionSnapshot', () => { models: CLAUDE_SESSION_OPTION_CATALOG.models, record, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) const model = snapshot[0]! if (model.kind.type !== 'select') { @@ -84,7 +113,8 @@ describe('buildNativeChatSessionOptionSnapshot', () => { models: [], record: claudeRecord(), mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) ).toEqual([]) }) @@ -136,7 +166,8 @@ describe('buildNativeChatSessionOptionSnapshot', () => { models: reconciled, record, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) const model = snapshot[0]! if (model.kind.type !== 'select') { @@ -197,7 +228,8 @@ describe('buildNativeChatSessionOptionSnapshot', () => { models: reconciled, record, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) expect(snapshot.map((descriptor) => descriptor.id)).toEqual(['model', 'effort']) expect(resolveAgentSessionOptionLaunch('grok', { model: 'grok-4.5' }).args).toEqual([ @@ -215,7 +247,8 @@ describe('buildNativeChatSessionOptionSnapshot', () => { models: CODEX_SESSION_OPTION_CATALOG.models, record: createNativeChatSessionOptionRecord('codex'), mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) expect(snapshot[0]).toMatchObject({ settable: true }) expect(snapshot[0]?.action).toEqual({ type: 'agent-picker' }) @@ -233,7 +266,8 @@ describe('buildNativeChatSessionOptionSnapshot', () => { models: CLAUDE_SESSION_OPTION_CATALOG.models, record, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) const fastMode = snapshot.find((descriptor) => descriptor.id === 'fastMode') expect(fastMode).toMatchObject({ action: { type: 'toggle-command' } }) @@ -250,7 +284,8 @@ describe('defaults on load', () => { models, record: createNativeChatSessionOptionRecord('grok'), mode, - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) it('shows the default model before anything is picked', () => { @@ -299,7 +334,8 @@ describe('defaults on load', () => { ], record, mode: 'draft', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) expect(snapshot[0]).toMatchObject({ valueSource: 'dispatched' }) expect(snapshot[0]!.kind.type === 'select' ? snapshot[0]!.kind.currentValue : null).toBe( @@ -315,7 +351,8 @@ describe('defaults on load', () => { models: CLAUDE_SESSION_OPTION_CATALOG.models, record: claudeRecord(), mode: 'draft', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) expect(CLAUDE_SESSION_OPTION_CATALOG.models.some((model) => model.isDefault)).toBe(true) expect(CLAUDE_SESSION_OPTION_CATALOG.defaultModelIsCliDefault).toBeUndefined() @@ -332,7 +369,8 @@ describe('defaults on load', () => { models: CLAUDE_SESSION_OPTION_CATALOG.models, record, mode: 'draft', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) const effort = snapshot.find((descriptor) => descriptor.id === 'effort') expect(effort).toMatchObject({ valueSource: 'default' }) diff --git a/src/shared/native-chat-session-option-snapshot.ts b/src/shared/native-chat-session-option-snapshot.ts index 76556211ab0..9fba42cf897 100644 --- a/src/shared/native-chat-session-option-snapshot.ts +++ b/src/shared/native-chat-session-option-snapshot.ts @@ -5,6 +5,7 @@ import type { CatalogOption } from './agent-session-option-catalog' import type { + NativeChatLiveOptionTransport, SessionOptionDescriptor, SessionOptionSelectChoice } from './native-chat-session-options' @@ -15,7 +16,7 @@ import { } from './native-chat-session-option-state' export type NativeChatSessionOptionMode = 'draft' | 'live' -export type NativeChatLiveOptionTransport = 'catalog' | 'agent-session' +export type { NativeChatLiveOptionTransport } function choiceWithCurrent( choices: readonly SessionOptionSelectChoice[], @@ -107,6 +108,7 @@ function optionDescriptor(args: { choices }, valueSource, + transport: liveTransport, ...settable, ...(action ? { action } : {}) } @@ -127,6 +129,7 @@ function optionDescriptor(args: { ...(currentValue === undefined ? {} : { currentValue }) }, valueSource, + transport: liveTransport, ...settable, ...(action ? { action } : {}) } @@ -202,9 +205,11 @@ export function buildNativeChatSessionOptionSnapshot(args: { record: NativeChatSessionOptionRecord mode: NativeChatSessionOptionMode modelLabel: string - liveTransport?: NativeChatLiveOptionTransport + /** Required, not defaulted: this is the only place a descriptor is built, so a + * producer that must state its lane here cannot silently inherit the other's. */ + liveTransport: NativeChatLiveOptionTransport }): SessionOptionDescriptor[] { - const { catalog, models, record, mode, modelLabel, liveTransport = 'catalog' } = args + const { catalog, models, record, mode, modelLabel, liveTransport } = args if (models.length === 0) { return [] } @@ -232,6 +237,7 @@ export function buildNativeChatSessionOptionSnapshot(args: { choices: modelChoices }, valueSource: modelTracked?.source ?? (defaultModelId ? 'default' : 'unknown'), + transport: liveTransport, ...settableState({ mode, liveTransport, apply: catalog.modelApply }), ...(modelAction ? { action: modelAction } : {}) } diff --git a/src/shared/native-chat-session-option-state.ts b/src/shared/native-chat-session-option-state.ts index 78b70980270..5d070969775 100644 --- a/src/shared/native-chat-session-option-state.ts +++ b/src/shared/native-chat-session-option-state.ts @@ -122,25 +122,30 @@ export function flattenNativeChatSessionOptionRecord( export function applyNativeChatReportedSessionOptions( record: NativeChatSessionOptionRecord, - values: Record + values: Record, + /** Ids the provider reported back. Omitted means every value is a report, which + * is what a surface that only ever learns values by reading them sends. */ + confirmed?: readonly string[] ): boolean { + const sourceFor = (id: string): TrackedNativeChatSessionOption['source'] => + confirmed === undefined || confirmed.includes(id) ? 'reported' : 'dispatched' const modelId = typeof values.model === 'string' ? values.model : null if (!modelId) { return false } const modelChanged = record.model?.value !== modelId - let changed = modelChanged || record.model?.source !== 'reported' - record.model = { value: modelId, source: 'reported' } + let changed = modelChanged || record.model?.source !== sourceFor('model') + record.model = { value: modelId, source: sourceFor('model') } const modelValues = modelChanged ? {} : { ...record.valuesByModel[modelId] } for (const [id, value] of Object.entries(values)) { if (id === 'model') { continue } const current = modelValues[id] - if (current?.value !== value || current.source !== 'reported') { + if (current?.value !== value || current.source !== sourceFor(id)) { changed = true } - modelValues[id] = { value, source: 'reported' } + modelValues[id] = { value, source: sourceFor(id) } } record.valuesByModel[modelId] = modelValues return changed diff --git a/src/shared/native-chat-session-options.ts b/src/shared/native-chat-session-options.ts index 84b23dcba06..33567d39131 100644 --- a/src/shared/native-chat-session-options.ts +++ b/src/shared/native-chat-session-options.ts @@ -7,9 +7,17 @@ export type SessionOptionSelectChoice = { } /** `default` is the catalog's own value shown before anything is observed — - * truthful to display, but never evidence about a running agent. */ + * truthful to display, but never evidence about a running agent. `dispatched` + * is sent-but-unread: the pill shows it, and a later report that disagrees is + * what corrects it. Both transports emit it, so it alone names neither — see + * `transport` on the descriptor. */ export type SessionOptionValueSource = 'applied' | 'dispatched' | 'reported' | 'default' | 'unknown' +/** How a live value reaches the agent. `catalog` types the catalog's command into + * the agent's terminal and can only learn the outcome by parsing the screen back; + * `agent-session` writes over the structured protocol, which reports every turn. */ +export type NativeChatLiveOptionTransport = 'catalog' | 'agent-session' + /** Closed set of reasons an option is not settable in the current mode. A key * (not free English) so the producer and the localized label stay in sync — * an exhaustive switch turns any drift into a type error instead of leaking @@ -34,6 +42,9 @@ export type SessionOptionDescriptor = { currentValue?: boolean } valueSource: SessionOptionValueSource + /** Required so a new producer cannot inherit the wrong lane's rendering by + * omission — `dispatched` is emitted identically by both and cannot discriminate. */ + transport: NativeChatLiveOptionTransport settable: boolean disabledReason?: SessionOptionDisabledReason /** Why: picker-only and toggle-only PTY commands cannot be represented as @@ -41,6 +52,15 @@ export type SessionOptionDescriptor = { action?: { type: 'agent-picker' | 'toggle-command' } } +/** A value we typed at the agent and have never read back. Only the terminal + * transport can be in this state: the structured lane's own per-turn report is + * what moves a value off `dispatched`, and until it lands nothing else has. */ +export function sessionOptionDispatchUnconfirmed( + descriptor: Pick +): boolean { + return descriptor.valueSource === 'dispatched' && descriptor.transport === 'catalog' +} + export type SessionOptionSetResult = { snapshot: SessionOptionDescriptor[] } diff --git a/src/shared/native-chat-tool-activity.test.ts b/src/shared/native-chat-tool-activity.test.ts new file mode 100644 index 00000000000..915e2b69729 --- /dev/null +++ b/src/shared/native-chat-tool-activity.test.ts @@ -0,0 +1,109 @@ +import { describe, expect, it } from 'vitest' +import type { NativeChatBlock } from './native-chat-types' +import { + describeActiveToolCall, + formatActiveToolLabel, + formatToolCallCount, + isCommandToolName, + selectActiveToolCall +} from './native-chat-tool-activity' + +function call( + name: string, + input: unknown, + state?: 'running' | 'completed' | 'failed' +): Extract { + return { type: 'tool-call', name, input, ...(state ? { state } : {}) } as Extract< + NativeChatBlock, + { type: 'tool-call' } + > +} + +describe('isCommandToolName', () => { + it('matches the shell-running tools regardless of case or padding', () => { + expect(isCommandToolName('bash')).toBe(true) + expect(isCommandToolName(' Bash ')).toBe(true) + expect(isCommandToolName('run_terminal_cmd')).toBe(true) + }) + + it('does not match a named tool', () => { + expect(isCommandToolName('Read')).toBe(false) + expect(isCommandToolName('')).toBe(false) + }) +}) + +describe('describeActiveToolCall / formatActiveToolLabel', () => { + it('drops the tool name for a shell command with a preview', () => { + const descriptor = describeActiveToolCall(call('Bash', { command: 'npm test' })) + expect(descriptor.key).toBe('runningPreview') + expect(descriptor.isCommand).toBe(true) + expect(formatActiveToolLabel(descriptor)).toBe('Running npm test') + }) + + it('falls back to a generic command label when there is no preview', () => { + const descriptor = describeActiveToolCall(call('bash', null)) + expect(descriptor.key).toBe('runningCommand') + expect(formatActiveToolLabel(descriptor)).toBe('Running command') + }) + + it('keeps the tool name for a non-command tool', () => { + const descriptor = describeActiveToolCall(call('Read', { file_path: 'a/b.ts' })) + expect(descriptor.key).toBe('runningNamedPreview') + expect(descriptor.isCommand).toBe(false) + expect(formatActiveToolLabel(descriptor)).toBe('Running Read a/b.ts') + }) + + it('names a previewless tool on its own', () => { + const descriptor = describeActiveToolCall(call('Think', null)) + expect(descriptor.key).toBe('runningNamed') + expect(formatActiveToolLabel(descriptor)).toBe('Running Think') + }) +}) + +describe('selectActiveToolCall', () => { + it('returns nothing once the turn is known to have ended', () => { + const blocks = [call('Bash', { command: 'x' }, 'running')] + expect(selectActiveToolCall(blocks, { activeTurnIsWorking: false })).toBeNull() + }) + + it('picks the latest explicitly running call', () => { + const blocks = [ + call('Read', { file_path: 'a' }, 'completed'), + call('Bash', { command: 'x' }, 'running'), + call('Grep', { pattern: 'y' }, 'running') + ] + expect(selectActiveToolCall(blocks, { activeTurnIsWorking: true })?.name).toBe('Grep') + }) + + it('ignores settled calls even while the turn works', () => { + const blocks = [call('Read', { file_path: 'a' }, 'completed')] + expect(selectActiveToolCall(blocks, { activeTurnIsWorking: true })).toBeNull() + }) + + it('treats a lifecycle-less call as running only while the turn works', () => { + const blocks = [call('Read', { file_path: 'a' })] + expect(selectActiveToolCall(blocks, { activeTurnIsWorking: true })?.name).toBe('Read') + expect(selectActiveToolCall(blocks, { activeTurnIsWorking: undefined })).toBeNull() + }) + + it('still surfaces an explicitly running call when the turn state is unknown', () => { + const blocks = [call('Bash', { command: 'x' }, 'running')] + expect(selectActiveToolCall(blocks, { activeTurnIsWorking: undefined })?.name).toBe('Bash') + }) + + it('skips non-tool-call blocks', () => { + const blocks: NativeChatBlock[] = [ + { type: 'text', text: 'hello' }, + call('Bash', { command: 'x' }, 'running') + ] + expect(selectActiveToolCall(blocks, { activeTurnIsWorking: true })?.name).toBe('Bash') + }) +}) + +describe('formatToolCallCount', () => { + it('singularizes one call', () => { + expect(formatToolCallCount(1)).toBe('1 tool call') + expect(formatToolCallCount(4)).toBe('4 tool calls') + expect(formatToolCallCount(0)).toBe('0 tool calls') + }) +}) diff --git a/src/shared/native-chat-tool-activity.ts b/src/shared/native-chat-tool-activity.ts new file mode 100644 index 00000000000..533dd75aecf --- /dev/null +++ b/src/shared/native-chat-tool-activity.ts @@ -0,0 +1,97 @@ +// Live tool-activity derivation and copy for the native-chat "Running …" row, +// shared by the desktop renderer (as its i18n fallback strings) and the mobile +// app (used directly — mobile ships English only) so the two surfaces never drift. + +import { createToolInputDisplay } from './native-chat-tool-summary' +import { isToolCallBlock, type NativeChatBlock } from './native-chat-types' + +type NativeChatToolCallBlock = Extract + +export const NATIVE_CHAT_TOOL_ACTIVITY_COPY = { + runningPreview: 'Running {{preview}}', + runningCommand: 'Running command', + runningNamedPreview: 'Running {{toolName}} {{preview}}', + runningNamed: 'Running {{toolName}}', + countOne: '1 tool call', + countN: '{{value0}} tool calls' +} as const + +/** Tools whose call is a shell command, so the row reads as terminal activity + * (and takes the terminal glyph) rather than a named tool invocation. */ +export const COMMAND_TOOL_NAMES: ReadonlySet = new Set([ + 'bash', + 'shell', + 'powershell', + 'terminal', + 'execute', + 'run_command', + 'run_shell_command', + 'shell_command', + 'exec_command', + 'run_terminal_cmd', + 'run_terminal_command' +]) + +export function isCommandToolName(name: string): boolean { + return COMMAND_TOOL_NAMES.has(name.trim().toLowerCase()) +} + +export type NativeChatActiveToolDescriptor = { + key: 'runningPreview' | 'runningCommand' | 'runningNamedPreview' | 'runningNamed' + toolName: string + preview: string + isCommand: boolean +} + +/** Which copy key and arguments the active-tool row renders for a running call. */ +export function describeActiveToolCall( + call: NativeChatToolCallBlock +): NativeChatActiveToolDescriptor { + const preview = createToolInputDisplay(call.input).label + const isCommand = isCommandToolName(call.name) + const key = isCommand + ? preview + ? 'runningPreview' + : 'runningCommand' + : preview + ? 'runningNamedPreview' + : 'runningNamed' + return { key, toolName: call.name, preview, isCommand } +} + +/** Resolve the active-tool label in English. For platforms without i18n (mobile). */ +export function formatActiveToolLabel(descriptor: NativeChatActiveToolDescriptor): string { + return NATIVE_CHAT_TOOL_ACTIVITY_COPY[descriptor.key] + .replaceAll('{{preview}}', descriptor.preview) + .replaceAll('{{toolName}}', descriptor.toolName) +} + +/** The most recent still-running call in a run, or null once the run is settled. + * A block without lifecycle `state` only counts while the turn is known to be + * working, so a restored transcript never spins on an orphaned call. */ +export function selectActiveToolCall( + blocks: readonly NativeChatBlock[], + { activeTurnIsWorking }: { activeTurnIsWorking?: boolean } +): NativeChatToolCallBlock | null { + if (activeTurnIsWorking === false) { + return null + } + const calls = blocks.filter(isToolCallBlock) + for (let index = calls.length - 1; index >= 0; index--) { + const call = calls[index] + if ( + call && + (call.state === 'running' || (call.state == null && activeTurnIsWorking === true)) + ) { + return call + } + } + return null +} + +/** Fallback summary when no per-tool summary is available. */ +export function formatToolCallCount(callCount: number): string { + return callCount === 1 + ? NATIVE_CHAT_TOOL_ACTIVITY_COPY.countOne + : NATIVE_CHAT_TOOL_ACTIVITY_COPY.countN.replaceAll('{{value0}}', String(callCount)) +} diff --git a/src/shared/native-chat-turn-status.test.ts b/src/shared/native-chat-turn-status.test.ts new file mode 100644 index 00000000000..3fdcccb7bf3 --- /dev/null +++ b/src/shared/native-chat-turn-status.test.ts @@ -0,0 +1,336 @@ +import { describe, expect, it } from 'vitest' +import type { NativeChatMessage } from './native-chat-types' +import { + describeNativeChatTurnStatus, + formatNativeChatDuration, + formatNativeChatTurnStatusLabel, + nativeChatElapsedSeconds, + nativeChatTurnHasResponse, + reduceNativeChatTurnTiming, + selectNativeChatTurnStatuses, + type NativeChatTurnTimingByTurn +} from './native-chat-turn-status' + +function message( + id: string, + role: NativeChatMessage['role'], + blocks: NativeChatMessage['blocks'] +): NativeChatMessage { + return { id, role, blocks, timestamp: null, source: 'transcript' } +} + +describe('formatNativeChatDuration', () => { + it.each([ + [0, '0s'], + [12, '12s'], + [59, '59s'], + [60, '1m 0s'], + [184, '3m 4s'], + [3600, '1h 0m 0s'], + [3723, '1h 2m 3s'] + ])('formats %i seconds as %s', (seconds, expected) => { + expect(formatNativeChatDuration(seconds)).toBe(expected) + }) + + it('floors a fractional count and clamps a negative or non-finite one', () => { + expect(formatNativeChatDuration(12.9)).toBe('12s') + expect(formatNativeChatDuration(-5)).toBe('0s') + expect(formatNativeChatDuration(Number.NaN)).toBe('0s') + }) +}) + +describe('describeNativeChatTurnStatus', () => { + it('prefers the settled duration over the thinking and counting labels', () => { + expect( + describeNativeChatTurnStatus({ thinking: true, workedSeconds: 184, elapsedSeconds: 9 }) + ).toEqual({ key: 'workedFor', duration: '3m 4s' }) + }) + + it('reports thinking before the turn produces output', () => { + expect( + describeNativeChatTurnStatus({ thinking: true, workedSeconds: null, elapsedSeconds: 9 }) + ).toEqual({ key: 'thinking', duration: null }) + }) + + it('counts once the turn has output', () => { + expect( + describeNativeChatTurnStatus({ thinking: false, workedSeconds: null, elapsedSeconds: 12 }) + ).toEqual({ key: 'workingFor', duration: '12s' }) + }) +}) + +describe('formatNativeChatTurnStatusLabel', () => { + it('renders each state in English for platforms without i18n', () => { + expect( + formatNativeChatTurnStatusLabel({ thinking: true, workedSeconds: null, elapsedSeconds: 0 }) + ).toBe('Thinking') + expect( + formatNativeChatTurnStatusLabel({ thinking: false, workedSeconds: null, elapsedSeconds: 12 }) + ).toBe('Working for 12s') + expect( + formatNativeChatTurnStatusLabel({ thinking: false, workedSeconds: 184, elapsedSeconds: 0 }) + ).toBe('Worked for 3m 4s') + }) +}) + +describe('nativeChatTurnHasResponse', () => { + const user = message('u1', 'user', [{ type: 'text', text: 'go' }]) + + it('is false while the turn has produced nothing', () => { + expect(nativeChatTurnHasResponse([user], 0)).toBe(false) + }) + + it('ignores a whitespace-only assistant block', () => { + const blank = message('a1', 'assistant', [{ type: 'text', text: ' \n ' }]) + expect(nativeChatTurnHasResponse([user, blank], 0)).toBe(false) + }) + + it('is true on the first real text, tool call, or tool result', () => { + expect( + nativeChatTurnHasResponse( + [user, message('a1', 'assistant', [{ type: 'text', text: 'hi' }])], + 0 + ) + ).toBe(true) + expect( + nativeChatTurnHasResponse( + [user, message('t1', 'tool', [{ type: 'tool-call', name: 'Read', input: {} }])], + 0 + ) + ).toBe(true) + expect( + nativeChatTurnHasResponse( + [user, message('t1', 'tool', [{ type: 'tool-result', output: 'ok' }])], + 0 + ) + ).toBe(true) + }) + + it('does not count output that preceded the latest user turn', () => { + const earlier = message('a0', 'assistant', [{ type: 'text', text: 'old' }]) + expect(nativeChatTurnHasResponse([earlier, user], 1)).toBe(false) + }) +}) + +describe('reduceNativeChatTurnTiming', () => { + const validTurnKeys = new Set(['u1']) + + it('stamps a start when a turn begins working', () => { + const next = reduceNativeChatTurnTiming( + {}, + { activeTurnKey: 'u1', validTurnKeys, isWorking: true, now: 1_000 } + ) + expect(next).toEqual({ u1: { startedAt: 1_000, workedSeconds: null } }) + }) + + it('keeps the original start across later working ticks', () => { + const first = reduceNativeChatTurnTiming( + {}, + { activeTurnKey: 'u1', validTurnKeys, isWorking: true, now: 1_000 } + ) + const second = reduceNativeChatTurnTiming(first, { + activeTurnKey: 'u1', + validTurnKeys, + isWorking: true, + now: 9_000 + }) + expect(second).toBe(first) + }) + + it('prefers an authoritative host start over the local stamp', () => { + const next = reduceNativeChatTurnTiming( + {}, + { + activeTurnKey: 'u1', + validTurnKeys, + isWorking: true, + workingStartedAt: 500, + now: 1_000 + } + ) + expect(next.u1?.startedAt).toBe(500) + }) + + it('settles the turn to whole elapsed seconds when work stops', () => { + const working = reduceNativeChatTurnTiming( + {}, + { activeTurnKey: 'u1', validTurnKeys, isWorking: true, now: 1_000 } + ) + const settled = reduceNativeChatTurnTiming(working, { + activeTurnKey: 'u1', + validTurnKeys, + isWorking: false, + now: 13_400 + }) + expect(settled.u1).toEqual({ startedAt: 1_000, workedSeconds: 12 }) + }) + + it('never re-settles an already settled turn', () => { + const settled: NativeChatTurnTimingByTurn = { u1: { startedAt: 1_000, workedSeconds: 12 } } + expect( + reduceNativeChatTurnTiming(settled, { + activeTurnKey: 'u1', + validTurnKeys, + isWorking: false, + now: 99_000 + }) + ).toBe(settled) + }) + + it('does not invent a settled turn that never started', () => { + expect( + reduceNativeChatTurnTiming( + {}, + { activeTurnKey: 'u1', validTurnKeys, isWorking: false, now: 1_000 } + ) + ).toEqual({}) + }) + + it('carries the elapsed start across an optimistic echo becoming a transcript row', () => { + // The mobile composer renders an accepted send as `pending-N` until the + // transcript echo lands under its real id. Without the carry-over the active + // turn key flips mid-turn and "Working for 8s" restarts at 0s. + const working = reduceNativeChatTurnTiming( + {}, + { + activeTurnKey: 'pending-1', + validTurnKeys: new Set(), + isWorking: true, + now: 1_000 + } + ) + expect(working['pending-1']?.startedAt).toBe(1_000) + const swapped = reduceNativeChatTurnTiming(working, { + activeTurnKey: 'u9', + previousActiveTurnKey: 'pending-1', + validTurnKeys: new Set(['u9']), + isWorking: true, + now: 9_000 + }) + expect(swapped.u9).toEqual({ startedAt: 1_000, workedSeconds: null }) + expect(swapped['pending-1']).toBeUndefined() + }) + + it('keeps a settled turn visible when the echo is replaced after it finished', () => { + // The swap can land after the turn settles. Re-keying (rather than only + // carrying a start) is what keeps the "Worked for N" row from vanishing. + const settled: NativeChatTurnTimingByTurn = { + 'pending-1': { startedAt: 1_000, workedSeconds: 12 } + } + const swapped = reduceNativeChatTurnTiming(settled, { + activeTurnKey: 'u9', + previousActiveTurnKey: 'pending-1', + validTurnKeys: new Set(['u9']), + isWorking: false, + now: 20_000 + }) + expect(swapped.u9).toEqual({ startedAt: 1_000, workedSeconds: 12 }) + expect(swapped['pending-1']).toBeUndefined() + }) + + it('settles a re-keyed in-flight turn from its original start', () => { + const working = reduceNativeChatTurnTiming( + {}, + { activeTurnKey: 'pending-1', validTurnKeys: new Set(), isWorking: true, now: 1_000 } + ) + const settled = reduceNativeChatTurnTiming(working, { + activeTurnKey: 'u9', + previousActiveTurnKey: 'pending-1', + validTurnKeys: new Set(['u9']), + isWorking: false, + now: 13_400 + }) + expect(settled.u9).toEqual({ startedAt: 1_000, workedSeconds: 12 }) + }) + + it('does not carry the start into a genuinely new turn', () => { + // The previous turn is still in the transcript, so this is the user sending + // again — that turn starts its own clock. + const working = reduceNativeChatTurnTiming( + {}, + { activeTurnKey: 'u1', validTurnKeys: new Set(['u1']), isWorking: true, now: 1_000 } + ) + const next = reduceNativeChatTurnTiming(working, { + activeTurnKey: 'u2', + previousActiveTurnKey: 'u1', + validTurnKeys: new Set(['u1', 'u2']), + isWorking: true, + now: 9_000 + }) + expect(next.u2?.startedAt).toBe(9_000) + }) + + it('does not carry a start from a turn that had already settled', () => { + const settled: NativeChatTurnTimingByTurn = { + 'pending-1': { startedAt: 1_000, workedSeconds: 5 } + } + const next = reduceNativeChatTurnTiming(settled, { + activeTurnKey: 'u9', + previousActiveTurnKey: 'pending-1', + validTurnKeys: new Set(['u9']), + isWorking: true, + now: 9_000 + }) + expect(next.u9?.startedAt).toBe(9_000) + }) + + it('drops timings for turns that left the transcript, keeping the active one', () => { + const current: NativeChatTurnTimingByTurn = { + gone: { startedAt: 1, workedSeconds: 2 }, + u1: { startedAt: 1_000, workedSeconds: 12 } + } + const next = reduceNativeChatTurnTiming(current, { + activeTurnKey: 'u1', + validTurnKeys, + isWorking: false, + now: 2_000 + }) + expect(Object.keys(next)).toEqual(['u1']) + }) +}) + +describe('selectNativeChatTurnStatuses', () => { + it('reports the working turn as thinking until it produces output', () => { + const { active } = selectNativeChatTurnStatuses( + { u1: { startedAt: 1_000, workedSeconds: null } }, + { activeTurnKey: 'u1', isWorking: true, hasCurrentTurnResponse: false } + ) + expect(active).toEqual({ startedAt: 1_000, thinking: true, workedSeconds: null }) + }) + + it('stops thinking once the turn has output', () => { + const { active } = selectNativeChatTurnStatuses( + { u1: { startedAt: 1_000, workedSeconds: null } }, + { activeTurnKey: 'u1', isWorking: true, hasCurrentTurnResponse: true } + ) + expect(active?.thinking).toBe(false) + }) + + it('exposes settled turns and resolves the active one from them when idle', () => { + const { active, completedByTurn } = selectNativeChatTurnStatuses( + { u1: { startedAt: 1_000, workedSeconds: 12 } }, + { activeTurnKey: 'u1', isWorking: false, hasCurrentTurnResponse: true } + ) + expect(completedByTurn.u1).toEqual({ startedAt: 1_000, thinking: false, workedSeconds: 12 }) + expect(active).toEqual(completedByTurn.u1) + }) + + it('omits an in-flight turn from the completed map', () => { + const { completedByTurn } = selectNativeChatTurnStatuses( + { u1: { startedAt: 1_000, workedSeconds: null } }, + { activeTurnKey: 'u1', isWorking: true, hasCurrentTurnResponse: true } + ) + expect(completedByTurn).toEqual({}) + }) +}) + +describe('nativeChatElapsedSeconds', () => { + it('falls back to the mount epoch before the turn start lands', () => { + expect(nativeChatElapsedSeconds(null, 1_000, 5_400)).toBe(4) + expect(nativeChatElapsedSeconds(2_000, 1_000, 5_400)).toBe(3) + }) + + it('never counts backwards', () => { + expect(nativeChatElapsedSeconds(9_000, 1_000, 5_000)).toBe(0) + }) +}) diff --git a/src/shared/native-chat-turn-status.ts b/src/shared/native-chat-turn-status.ts new file mode 100644 index 00000000000..1012f8fc604 --- /dev/null +++ b/src/shared/native-chat-turn-status.ts @@ -0,0 +1,210 @@ +// Turn-status derivation and copy for the native-chat "Thinking / Working for N / +// Worked for N" row, shared by the desktop renderer (as its i18n fallback strings) +// and the mobile app (used directly — mobile ships English only) so the two +// surfaces never drift. Everything here is pure; each platform owns its own clock. + +import type { NativeChatMessage } from './native-chat-types' + +export const NATIVE_CHAT_TURN_STATUS_COPY = { + thinking: 'Thinking', + workingFor: 'Working for {{value0}}', + workedFor: 'Worked for {{value0}}', + toggleDetails: 'Toggle turn details', + responding: 'Agent is responding' +} as const + +/** Format turn time without exposing an ever-growing raw seconds count. */ +export function formatNativeChatDuration(seconds: number): string { + const totalSeconds = Number.isFinite(seconds) ? Math.max(0, Math.floor(seconds)) : 0 + if (totalSeconds < 60) { + return `${totalSeconds}s` + } + const minutes = Math.floor(totalSeconds / 60) + const remainingSeconds = totalSeconds % 60 + if (minutes < 60) { + return `${minutes}m ${remainingSeconds}s` + } + const hours = Math.floor(minutes / 60) + return `${hours}h ${minutes % 60}m ${remainingSeconds}s` +} + +/** Which of the three copy keys a turn-status row renders, and its duration + * argument. Desktop maps this onto `translate`; mobile formats it directly. */ +export function describeNativeChatTurnStatus({ + thinking, + workedSeconds, + elapsedSeconds +}: { + thinking: boolean + workedSeconds?: number | null + elapsedSeconds: number +}): { key: 'thinking' | 'workingFor' | 'workedFor'; duration: string | null } { + if (workedSeconds != null) { + return { key: 'workedFor', duration: formatNativeChatDuration(workedSeconds) } + } + if (thinking) { + return { key: 'thinking', duration: null } + } + return { key: 'workingFor', duration: formatNativeChatDuration(elapsedSeconds) } +} + +/** Resolve the turn-status label in English. For platforms without i18n (mobile). */ +export function formatNativeChatTurnStatusLabel(input: { + thinking: boolean + workedSeconds?: number | null + elapsedSeconds: number +}): string { + const { key, duration } = describeNativeChatTurnStatus(input) + const copy = NATIVE_CHAT_TURN_STATUS_COPY[key] + return duration == null ? copy : copy.replaceAll('{{value0}}', duration) +} + +/** True once the current turn has produced anything renderable — the boundary + * between the "Thinking" label and the counting "Working for N" label. */ +export function nativeChatTurnHasResponse( + messages: readonly NativeChatMessage[], + latestUserIndex: number +): boolean { + return messages + .slice(latestUserIndex + 1) + .some( + (message) => + (message.role === 'assistant' || message.role === 'tool') && + message.blocks.some( + (block) => + block.type === 'tool-call' || + block.type === 'tool-result' || + (block.type === 'text' && block.text.trim().length > 0) + ) + ) +} + +export type NativeChatTurnTiming = { + startedAt: number + workedSeconds: number | null +} + +export type NativeChatTurnStatus = { + startedAt: number | null + thinking: boolean + workedSeconds: number | null +} + +export type NativeChatTurnTimingByTurn = Readonly> + +/** The turn-timing state machine, lifted out of the React hook so desktop and + * mobile stamp start/stop identically. Returns the same reference when nothing + * changed so callers can bail out of a state update. */ +export function reduceNativeChatTurnTiming( + current: NativeChatTurnTimingByTurn, + { + activeTurnKey, + previousActiveTurnKey, + validTurnKeys, + isWorking, + workingStartedAt, + now + }: { + activeTurnKey: string + /** The key this turn had on the previous pass. When it names a turn that has + * since left the transcript, the two are the same turn under two ids — an + * optimistic echo that the transcript replaced — so the clock carries over + * instead of restarting. Omit it to keep the plain restart behavior. */ + previousActiveTurnKey?: string + validTurnKeys: ReadonlySet + isWorking: boolean + workingStartedAt?: number | null + now: number + } +): NativeChatTurnTimingByTurn { + // The same turn under two ids: an optimistic echo the transcript has since + // replaced. Re-key its timing so neither the running clock nor an already + // settled duration is lost when the swap lands. + const replacedTiming = + previousActiveTurnKey !== undefined && + previousActiveTurnKey !== activeTurnKey && + !validTurnKeys.has(previousActiveTurnKey) && + current[activeTurnKey] === undefined + ? current[previousActiveTurnKey] + : undefined + let retained = replacedTiming ? { ...current, [activeTurnKey]: replacedTiming } : current + for (const turnKey of Object.keys(retained)) { + if (turnKey !== activeTurnKey && !validTurnKeys.has(turnKey)) { + if (retained === current) { + retained = { ...current } + } + delete (retained as Record)[turnKey] + } + } + + const timing = retained[activeTurnKey] + if (isWorking) { + // An in-flight turn keeps the start it already had; only a fresh turn (or an + // authoritative host timestamp) restamps it. + const startedAt = + workingStartedAt ?? (timing && timing.workedSeconds == null ? timing.startedAt : now) + if (timing?.startedAt === startedAt && timing.workedSeconds == null) { + return retained + } + return { ...retained, [activeTurnKey]: { startedAt, workedSeconds: null } } + } + + if (timing?.workedSeconds != null) { + return retained + } + const startedAt = timing?.startedAt ?? workingStartedAt + if (startedAt == null) { + return retained + } + return { + ...retained, + [activeTurnKey]: { + startedAt, + workedSeconds: Math.max(0, Math.floor((now - startedAt) / 1000)) + } + } +} + +/** Split the timing map into the active turn's status and the settled ones. */ +export function selectNativeChatTurnStatuses( + timingByTurn: NativeChatTurnTimingByTurn, + { + activeTurnKey, + isWorking, + workingStartedAt, + hasCurrentTurnResponse + }: { + activeTurnKey: string + isWorking: boolean + workingStartedAt?: number | null + hasCurrentTurnResponse: boolean + } +): { active: NativeChatTurnStatus | null; completedByTurn: Record } { + const completedByTurn = Object.fromEntries( + Object.entries(timingByTurn) + .filter(([, timing]) => timing.workedSeconds != null) + .map(([turnKey, timing]) => [ + turnKey, + { startedAt: timing.startedAt, thinking: false, workedSeconds: timing.workedSeconds } + ]) + ) as Record + return { + active: isWorking + ? { + startedAt: workingStartedAt ?? timingByTurn[activeTurnKey]?.startedAt ?? null, + thinking: !hasCurrentTurnResponse, + workedSeconds: null + } + : (completedByTurn[activeTurnKey] ?? null), + completedByTurn + } +} + +/** Elapsed whole seconds for a counting turn, tolerating a not-yet-stamped start. */ +export function nativeChatElapsedSeconds( + startedAt: number | null, + fallbackStartedAt: number, + now: number +): number { + return Math.max(0, Math.floor((now - (startedAt ?? fallbackStartedAt)) / 1000)) +} diff --git a/src/shared/native-chat-types.ts b/src/shared/native-chat-types.ts index 5daa16f760e..124ee55dbe1 100644 --- a/src/shared/native-chat-types.ts +++ b/src/shared/native-chat-types.ts @@ -55,11 +55,31 @@ export type NativeChatToolCallBlock = { state?: 'running' | 'completed' | 'failed' } +/** One resolved hunk from a provider's edit result, carrying true file ranges. */ +export type NativeChatEditPatchHunk = { + oldStart: number + oldLines: number + newStart: number + newLines: number + /** Signed unified rows, as the provider emitted them. */ + lines: string[] +} + +/** Hunks the provider resolved against the real file before reporting the edit. + * Claude supplies these on its edit results; Codex resolves equivalently before + * sending, so its patch already carries ranges and needs no companion. */ +export type NativeChatEditPatch = { + filePath?: string + hunks: NativeChatEditPatchHunk[] +} + /** The result returned to the agent for a prior tool call. */ export type NativeChatToolResultBlock = { type: 'tool-result' output: string isError?: boolean + /** Present only for edit tools whose result reported resolved hunks. */ + editPatch?: NativeChatEditPatch } /** A reference to an image, by local path or remote URL. Exactly the field diff --git a/src/shared/native-chat-unified-patch.ts b/src/shared/native-chat-unified-patch.ts new file mode 100644 index 00000000000..2e34c79a6b3 --- /dev/null +++ b/src/shared/native-chat-unified-patch.ts @@ -0,0 +1,255 @@ +import { FILE_SECTION_START, isFileHeaderPair } from './native-chat-diff' +import { pushEditGap, splitEditContent, type NativeChatEditLine } from './native-chat-edit-model' + +const HUNK_RANGES = /^@@+ -(\d+)(?:,(\d+))? \+(\d+)(?:,(\d+))? @@/ + +export type UnifiedPatchLines = { + lines: NativeChatEditLine[] + /** True only when every hunk carried real `@@` ranges. */ + lineNumbersKnown: boolean + /** The patch text was clipped before it became rows. */ + truncated: boolean +} + +/** Parses unified patch text, keeping the `@@` ranges as per-row line numbers. + * A hunk header whose `@@` is a bare context anchor with no ranges leaves its + * rows unnumbered rather than numbered from 1, because a wrong number reads as + * authoritative. + * + * `implicitFirstHunk` opens the body as a hunk of unknown position, for the + * patch dialect whose first chunk may carry no header at all. */ +export function editLinesFromUnifiedPatch( + text: string, + options?: { implicitFirstHunk?: boolean } +): UnifiedPatchLines | null { + const source = splitEditContent(text) + const rows = source.lines + const lines: NativeChatEditLine[] = [] + let oldNo: number | null = null + let newNo: number | null = null + let sawHunk = options?.implicitFirstHunk === true + let ranged = true + let inHunk = sawHunk + + for (let index = 0; index < rows.length; index += 1) { + const raw = rows[index] ?? '' + if (raw.startsWith('@@')) { + const match = HUNK_RANGES.exec(raw) + oldNo = match ? Number(match[1]) : null + newNo = match ? Number(match[3]) : null + // Successive hunks are separate regions of the file; concatenated with no + // break the gutter jumps and the reader sees one continuous block. + pushEditGap(lines) + sawHunk = true + inHunk = true + continue + } + // `\ No newline at end of file` sits mid-hunk, between the removed old last + // line and the added new one, so it ends nothing. + if (raw.startsWith('\\')) { + continue + } + if (!inHunk && isFileHeaderPair(rows, index)) { + index += 1 + continue + } + if (FILE_SECTION_START.test(raw)) { + inHunk = false + continue + } + if (!inHunk) { + continue + } + // Read off the rows rather than the header, so a body that opened with no + // header is reported as unlocatable just like a rangeless `@@`. + ranged &&= oldNo !== null || newNo !== null + if (raw.startsWith('+')) { + lines.push({ + kind: 'add', + text: raw.slice(1), + oldLineNumber: null, + newLineNumber: newNo + }) + newNo = newNo === null ? null : newNo + 1 + continue + } + if (raw.startsWith('-')) { + lines.push({ + kind: 'del', + text: raw.slice(1), + oldLineNumber: oldNo, + newLineNumber: null + }) + oldNo = oldNo === null ? null : oldNo + 1 + continue + } + lines.push({ + kind: 'context', + text: raw.startsWith(' ') ? raw.slice(1) : raw, + oldLineNumber: oldNo, + newLineNumber: newNo + }) + oldNo = oldNo === null ? null : oldNo + 1 + newNo = newNo === null ? null : newNo + 1 + } + + if (!sawHunk || lines.length === 0) { + return null + } + return { lines, lineNumbersKnown: ranged, truncated: source.truncated } +} + +const GIT_DIFF_HEADER = 'diff --git ' + +export type UnifiedPatchSection = { + /** Null when the patch text named no file, leaving it to the caller. */ + path: string | null + oldPath: string | null + changeKind: 'added' | 'deleted' | 'edited' | 'renamed' + body: string +} + +type Section = { + rows: string[] + oldPath: string | null + newPath: string | null + named: boolean + /** A `--- `/`+++ ` pair already named this section, so the next one is a new file. */ + hasHeaderPair: boolean + /** Only a `diff --git` header states both sides of a move as such. A bare + * pair with differing paths is just as likely two directories compared. */ + fromGitHeader: boolean +} + +/** Splits patch text into one section per file it touches. Without this a + * multi-file patch renders as a single card under the first file's name, with + * the later files' rows and gutter numbers beneath it. */ +export function unifiedPatchSections(text: string): { + sections: UnifiedPatchSection[] + truncated: boolean +} { + const source = splitEditContent(text) + const rows = source.lines + const sections: Section[] = [] + let current: Section | null = null + let inHunk = false + + const open = (): Section => { + const section: Section = { + rows: [], + oldPath: null, + newPath: null, + named: false, + hasHeaderPair: false, + fromGitHeader: false + } + sections.push(section) + return section + } + + for (let index = 0; index < rows.length; index += 1) { + const raw = rows[index] ?? '' + if (raw.startsWith(GIT_DIFF_HEADER)) { + const paths = gitHeaderPaths(raw) + current = open() + current.oldPath = paths.oldPath + current.newPath = paths.newPath + current.named = true + current.fromGitHeader = true + inHunk = false + continue + } + // A header pair is structure outside a hunk. Inside one it is also a file + // boundary, but only when a hunk header follows it immediately: a removed + // `-- x` over an added `++ y` is never followed by a column-0 `@@`, and + // that is what separates the files of a patch written without `diff --git` + // headers, where nothing else would end the previous file's hunk. + if (isFileHeaderPair(rows, index) && (!inHunk || (rows[index + 2] ?? '').startsWith('@@'))) { + // The pair names the section a `diff --git` just opened; a second pair in + // the same section is the next file of a patch written without them. + if (!current || current.hasHeaderPair) { + current = open() + } + current.oldPath = sourceHeaderPath(rows[index] ?? '') + current.newPath = sourceHeaderPath(rows[index + 1] ?? '') + current.named = true + current.hasHeaderPair = true + inHunk = false + index += 1 + continue + } + if (raw.startsWith('@@')) { + inHunk = true + } else if (FILE_SECTION_START.test(raw)) { + inHunk = false + } + current ??= open() + current.rows.push(raw) + } + + return { + sections: sections.map((section) => ({ + path: section.newPath ?? section.oldPath, + oldPath: sectionChangeKind(section) === 'renamed' ? section.oldPath : null, + changeKind: sectionChangeKind(section), + body: section.rows.join('\n') + })), + truncated: source.truncated + } +} + +function sectionChangeKind(section: Section): UnifiedPatchSection['changeKind'] { + if (!section.named) { + return 'edited' + } + if (section.newPath === null) { + return 'deleted' + } + if (section.oldPath === null) { + return 'added' + } + if (section.oldPath === section.newPath) { + return 'edited' + } + // Differing sides are a move only where the header says so. Bare pairs carry + // whatever paths the producer compared, which may be two directories. + return section.fromGitHeader ? 'renamed' : 'edited' +} + +/** `--- a/` / `+++ b/`, where the absent side is `/dev/null` and a + * trailing tab introduces the timestamp some producers append. */ +function sourceHeaderPath(line: string): string | null { + const value = (line.slice(4).split('\t')[0] ?? '').trim() + return value === '' || value === '/dev/null' ? null : value.replace(/^[ab]\//, '') +} + +function gitHeaderPaths(line: string): { oldPath: string | null; newPath: string | null } { + const rest = line.slice(GIT_DIFF_HEADER.length) + // Both halves carry the same path unless the file moved, so the second one + // starts at the last ` b/` rather than at the first space. + const split = rest.lastIndexOf(' b/') + if (split === -1) { + return { oldPath: null, newPath: null } + } + return { + oldPath: rest.slice(0, split).replace(/^a\//, ''), + newPath: rest.slice(split + 1).replace(/^b\//, '') + } +} + +/** Rows for a whole-file add or delete, which legitimately number from 1. */ +export function editLinesFromWholeFile( + content: string, + kind: 'add' | 'del' +): { lines: NativeChatEditLine[]; truncated: boolean } { + const body = splitEditContent(content) + return { + lines: body.lines.map((text, index) => ({ + kind, + text, + oldLineNumber: kind === 'del' ? index + 1 : null, + newLineNumber: kind === 'add' ? index + 1 : null + })), + truncated: body.truncated + } +} diff --git a/src/shared/process-table-snapshot-reader.ts b/src/shared/process-table-snapshot-reader.ts index 8e1655d400d..962a24c7f48 100644 --- a/src/shared/process-table-snapshot-reader.ts +++ b/src/shared/process-table-snapshot-reader.ts @@ -2,6 +2,7 @@ import { execFile as execFileCb } from 'node:child_process' import { readFile } from 'node:fs/promises' import { promisify } from 'node:util' import { + PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS, PS_ARGS, PS_MAX_BUFFER_BYTES, ProcessTableCaptureError, @@ -10,15 +11,19 @@ import { type ProcessTableRow } from './process-table-snapshot' -export { PS_ARGS, PS_MAX_BUFFER_BYTES } +export { PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS, PS_ARGS, PS_MAX_BUFFER_BYTES } const execFile = promisify(execFileCb) -/** Columns used by the evidence reader. Keep command last so its spaces survive parsing. */ -export const PS_TIMEOUT_MS = 3000 -const DEFAULT_SNAPSHOT_TTL_MS = 500 +// Why 15s: the `command=` column costs a per-pid argv read (measured 1.15s for 1,948 +// processes; 0.03s without it), and CPU contention multiplies that -- at load 27 the same +// capture measured 1.3-6.0s, so a 3s budget timed out on 6 of 20 consecutive tries and the +// whole subsystem answered "unverifiable" about a table it could read. This keeps a wedged +// `ps` bounded while staying out of reach of a host that is merely busy. +export const PS_TIMEOUT_MS = 15_000 +const DEFAULT_SNAPSHOT_TTL_MS = PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS -type Snapshot = { value: T; capturedAtMs: number } +type Snapshot = { value: T; capturedAtMs: number; completedAtMs: number } type ProcessTableSnapshotReaderDeps = { runPs: () => Promise @@ -42,11 +47,16 @@ export function createProcessTableSnapshotReader( let freshQueued: { promise: Promise; startSequence: number | null } | null = null async function runSnapshot(): Promise { + // Two stamps because they answer different questions: `capturedAtMs` is when `ps` read the + // kernel table, which is what a destructive consumer bounds staleness against, while the TTL + // keys on completion so a capture slower than the TTL still coalesces instead of forking a + // whole-machine `ps` per caller on exactly the loaded host that can least afford it. + const capturedAtMs = deps.now() const promise = deps.runPs() inFlight = promise try { const value = await promise - cached = { value, capturedAtMs: deps.now() } + cached = { value, capturedAtMs, completedAtMs: deps.now() } return value } finally { if (inFlight === promise) { @@ -56,7 +66,7 @@ export function createProcessTableSnapshotReader( } async function getSnapshot(): Promise { - if (cached && deps.now() - cached.capturedAtMs < ttlMs) { + if (cached && deps.now() - cached.completedAtMs < ttlMs) { return cached.value } if (inFlight) { @@ -198,6 +208,19 @@ function assertWholeCapture(stdout: string): string { return stdout } +/** Field 22 (`starttime`) of `/proc//stat`, read past the parenthesised comm. */ +export function parseLinuxProcStatStartTime(stat: string): string | null { + const closingParen = stat.lastIndexOf(')') + if (closingParen === -1) { + return null + } + const tail = stat + .slice(closingParen + 1) + .trim() + .split(/\s+/) + return tail[19] || null +} + /** Read Linux's stable PID start-time ticks without spawning another process. */ async function readLinuxProcessStartTimes( rows: readonly ProcessTableRow[] @@ -209,16 +232,9 @@ async function readLinuxProcessStartTimes( const starts = await Promise.all( candidates.map(async (row) => { try { - const stat = await readFile(`/proc/${row.pid}/stat`, 'utf8') - const closingParen = stat.lastIndexOf(')') - if (closingParen === -1) { - return null - } - const tail = stat - .slice(closingParen + 1) - .trim() - .split(/\s+/) - const startTime = tail[19] + const startTime = parseLinuxProcStatStartTime( + await readFile(`/proc/${row.pid}/stat`, 'utf8') + ) return startTime ? ([row.pid, startTime] as const) : null } catch { return null @@ -269,17 +285,53 @@ export async function getStrictProcessTableSnapshot(): Promise(pending: Promise): Promise { + let timer: ReturnType | undefined + try { + return await Promise.race([ + pending, + new Promise((_resolve, reject) => { + timer = setTimeout( + () => reject(new ProcessTableCaptureError('capture_over_budget')), + PROCESS_TABLE_EVIDENCE_BUDGET_MS + ) + }) + ]) + } finally { + clearTimeout(timer) + } +} + export async function getStrictProcessTableSnapshotWithAge(): Promise<{ rows: ProcessTableRow[] capturedAgeMs: number }> { - const snapshot = await processTableReader.getSnapshotWithAge() + const snapshot = await withEvidenceBudget(processTableReader.getSnapshotWithAge()) return { rows: snapshot.value.strict(), capturedAgeMs: snapshot.capturedAgeMs } } -/** How much older than its own await a TTL-cached capture may be. */ -export const PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS = DEFAULT_SNAPSHOT_TTL_MS - export function resetProcessTableSnapshotForTests(): void { processTableReader.reset() } diff --git a/src/shared/process-table-snapshot.test.ts b/src/shared/process-table-snapshot.test.ts index 80f9ed1ab20..fea3d891b6d 100644 --- a/src/shared/process-table-snapshot.test.ts +++ b/src/shared/process-table-snapshot.test.ts @@ -1,6 +1,6 @@ import { readFileSync } from 'node:fs' import { join } from 'node:path' -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const { execFileMock } = vi.hoisted(() => ({ execFileMock: vi.fn() })) @@ -10,7 +10,10 @@ import { createProcessTableSnapshotReader, getProcessTableSnapshot, getStrictProcessTableSnapshot, + getStrictProcessTableSnapshotWithAge, + PROCESS_TABLE_EVIDENCE_BUDGET_MS, PS_MAX_BUFFER_BYTES, + PS_TIMEOUT_MS, resetProcessTableSnapshotForTests } from './process-table-snapshot-reader' import { @@ -124,6 +127,26 @@ describe('process-table-snapshot reader', () => { expect(scans).toBe(1) }) + it('reports the age of the capture instant, not of the moment ps finished', async () => { + let clock = 0 + const gate = deferred() + const reader = createProcessTableSnapshotReader({ + runPs: () => gate.promise, + now: () => clock, + ttlMs: 500 + }) + + const first = reader.getSnapshot() + // A capture that took 4s of wall clock describes the machine as it was 4s ago. + clock = 4_000 + gate.resolve('scan-1') + await first + + // Age used to start at the callback, so a capture older than the destructive + // consumer's ceiling was admitted as if it had just been taken. + expect((await reader.getSnapshotWithAge()).capturedAgeMs).toBe(4_000) + }) + it('does not cache failures and retries on the next call', async () => { let scans = 0 const reader = createProcessTableSnapshotReader({ @@ -599,4 +622,118 @@ describe('process-table capture completeness', () => { await expect(getProcessTableSnapshot()).rejects.toThrow(ProcessTableCaptureError) await expect(getProcessTableSnapshot()).rejects.toThrow('empty_capture') }) + + /** + * Emulate Node's own execFile deadline: past `timeout` it SIGTERMs the child and calls + * back with `Command failed: ` and no stderr -- indistinguishable from a broken + * `ps` unless the budget itself is under test. + */ + function mockPsTakingMs(durationMs: number, stdout: string): void { + execFileMock.mockImplementation( + (command: string, args: string[], options: unknown, callback: unknown) => { + const done = callback as (err: unknown, result: { stdout: string; stderr: string }) => void + const timeout = (options as { timeout?: number })?.timeout + if (timeout !== undefined && durationMs > timeout) { + const error = Object.assign( + new Error(`Command failed: ${[command, ...args].join(' ')}\n`), + { code: null, killed: true, signal: 'SIGTERM' } + ) + done(error, { stdout: '', stderr: '' }) + return + } + done(null, { stdout, stderr: '' }) + } + ) + } + + it('reads a loaded host whose whole-machine ps runs for seconds', async () => { + // Measured on a 1,948-process host at load 27: the `command=` argv read alone costs + // 1.15s, and contention stretched the same capture to 6.0s. The old 3s budget killed + // 6 of 20 consecutive captures, so a readable table answered "unverifiable". + mockPsTakingMs(6_000, busyHostTable(200)) + + expect(PS_TIMEOUT_MS).toBeGreaterThan(6_000) + await expect(getProcessTableSnapshot()).resolves.toHaveLength(200) + }) + + it('still bounds a ps that never returns', async () => { + mockPsTakingMs(PS_TIMEOUT_MS + 1, busyHostTable(200)) + + await expect(getProcessTableSnapshot()).rejects.toThrow('Command failed: ps') + }) +}) + +/** + * The evidence-publishing read has a far shorter budget than identity proof, and the two are not + * interchangeable. Identity proof asks whether a process exists and must not read a slow capture + * as an absent one, so it waits out `PS_TIMEOUT_MS`. These consumers ask whether an observation + * describes NOW: past this budget it cannot, so the answer they need is a prompt `unverifiable`, + * which both relay call sites already produce from a rejection. + */ +describe('evidence-publishing capture budget', () => { + beforeEach(() => { + execFileMock.mockReset() + resetProcessTableSnapshotForTests() + vi.useFakeTimers() + }) + + afterEach(() => { + vi.useRealTimers() + }) + + /** A `ps` that answers after `durationMs` of wall clock, the way a loaded host does. */ + function mockPsTaking(durationMs: number, stdout = '1 0 1 1 S+ ?? Jan 1 00:00:00 2026 bash\n') { + execFileMock.mockImplementation( + (_command: string, _args: string[], _options: unknown, callback: unknown) => { + const done = callback as (err: unknown, result: { stdout: string; stderr: string }) => void + setTimeout(() => done(null, { stdout, stderr: '' }), durationMs) + } + ) + } + + it('answers from a capture that lands one tick inside the budget', async () => { + mockPsTaking(PROCESS_TABLE_EVIDENCE_BUDGET_MS - 1) + const pending = getStrictProcessTableSnapshotWithAge() + + await vi.advanceTimersByTimeAsync(PROCESS_TABLE_EVIDENCE_BUDGET_MS) + + // The age carries the capture's own duration, which is the whole reason the budget is this + // far under the 2,000ms admission ceiling rather than under PS_TIMEOUT_MS. + expect((await pending).capturedAgeMs).toBe(PROCESS_TABLE_EVIDENCE_BUDGET_MS - 1) + }) + + it('gives up on a capture one tick past the budget instead of waiting out PS_TIMEOUT_MS', async () => { + mockPsTaking(PROCESS_TABLE_EVIDENCE_BUDGET_MS + 1) + const pending = getStrictProcessTableSnapshotWithAge() + const settled = pending.then( + () => 'resolved', + (error: Error) => error.message + ) + + await vi.advanceTimersByTimeAsync(PROCESS_TABLE_EVIDENCE_BUDGET_MS) + + expect(await settled).toContain('process table unreadable') + }) + + it('leaves the abandoned capture running to fill the cache for the next reader', async () => { + // Why this matters: a whole-machine `ps` is the most expensive thing the host does, and the + // host that blows the budget is by definition the one that can least afford a second one. + mockPsTaking(PROCESS_TABLE_EVIDENCE_BUDGET_MS + 100) + const abandoned = getStrictProcessTableSnapshotWithAge().catch(() => null) + + await vi.advanceTimersByTimeAsync(PROCESS_TABLE_EVIDENCE_BUDGET_MS) + expect(await abandoned).toBeNull() + + await vi.advanceTimersByTimeAsync(100) + expect(await getStrictProcessTableSnapshotWithAge()).toMatchObject({ + capturedAgeMs: PROCESS_TABLE_EVIDENCE_BUDGET_MS + 100 + }) + expect(execFileMock).toHaveBeenCalledTimes(1) + }) + + it('keeps the identity budget far above its own', () => { + // A slow capture must still not read as an absent process; only the consumers that need the + // observation to describe NOW gave up waiting for it. + expect(PROCESS_TABLE_EVIDENCE_BUDGET_MS).toBeLessThan(PS_TIMEOUT_MS) + }) }) diff --git a/src/shared/process-table-snapshot.ts b/src/shared/process-table-snapshot.ts index 00cf1f1c380..3b3236079c4 100644 --- a/src/shared/process-table-snapshot.ts +++ b/src/shared/process-table-snapshot.ts @@ -13,19 +13,86 @@ export type ProcessTableRow = { command: string } +// Why guarded: this module is the renderer-safe half of the process-table pair, and the renderer +// runs sandboxed with contextIsolation, where a bare `process` read throws at module evaluation +// and takes the whole chunk — and the app — down with it. Only hosts ever run these argv. +const HOST_IS_DARWIN = typeof process !== 'undefined' && process.platform === 'darwin' + /** Columns used by the evidence reader. Keep command last so its spaces survive parsing. */ export const PS_ARGS = ( - process.platform === 'darwin' + HOST_IS_DARWIN ? ['-axo', 'pid=,ppid=,pgid=,tpgid=,stat=,tty=,lstart=,command='] : ['-axo', 'pid=,ppid=,pgid=,tpgid=,stat=,tty=,etimes=,command='] ) as readonly string[] +/** + * Cheap tier: the same job-control columns without `tty=` (0.29s of the 0.34s on a + * 1,900-process Mac) or `command=` (per-pid argv read, 1.15s on Linux). Enough to prove a + * pane's subtree is unchanged since the last full capture; never enough to name a process. + * No `etimes=` on Linux: it is elapsed seconds, so it changes every tick; the stable start + * marker comes from `/proc//stat` for the pane subtree only. + */ +export const CHEAP_PS_ARGS = ( + HOST_IS_DARWIN + ? ['-axo', 'pid=,ppid=,pgid=,tpgid=,stat=,lstart='] + : ['-axo', 'pid=,ppid=,pgid=,tpgid=,stat='] +) as readonly string[] + +export type CheapProcessTableRow = { + pid: number + ppid: number + pgid: number + tpgid: number + stat: string + /** Host start marker when the column set carries one (macOS `lstart`). */ + startTime?: string +} + +/** + * Parse a {@link CHEAP_PS_ARGS} capture. Lenient on purpose: a dropped row can only make a + * fingerprint DIFFER from the strict full-capture one, which escalates to the full capture -- + * the safe direction. An empty capture is unreadable, not "no processes". + */ +export function parseCheapProcessTableRows(stdout: string): CheapProcessTableRow[] { + const rows: CheapProcessTableRow[] = [] + for (const rawLine of stdout.split(/\r?\n/)) { + const match = rawLine.trim().match(/^(\d+)\s+(\d+)\s+(-?\d+)\s+(-?\d+)\s+(\S+)(?:\s+(.+?))?$/) + if (!match) { + continue + } + const pid = Number(match[1]) + if (!Number.isSafeInteger(pid) || pid <= 0) { + continue + } + rows.push({ + pid, + ppid: Number(match[2]), + pgid: Number(match[3]), + tpgid: Number(match[4]), + stat: match[5], + ...(match[6] !== undefined ? { startTime: match[6] } : {}) + }) + } + if (rows.length === 0) { + throw new ProcessTableCaptureError('empty_capture') + } + return rows +} + // Why: execFile's 1MB default leaves ~3x headroom (326KB / 1,460 processes, and // a single 5KB argv row is ordinary), so a busy host overflows it and then EVERY // capture fails — a readable process table degrading into permanent // "unverifiable". Matches the sibling reader in pty-descendant-termination.ts. export const PS_MAX_BUFFER_BYTES = 32 * 1024 * 1024 +/** How much older than its own await a TTL-cached capture may be, on top of the capture's own + * duration. Reported ages carry both, so this alone is not the staleness bound. + * + * Why here and not beside the reader that applies it: the renderer's cadence scheduler pulls a + * pane's next poll forward by at most this much, and the reader is a `node:child_process` module + * the renderer must never reach. */ +export const PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS = 500 + /** * Parse legacy or evidence-shaped `ps` output into rows. Tolerates CRLF so a * snapshot parsed on any host stays correct; `command` (last field) keeps its diff --git a/src/shared/protocol-version.ts b/src/shared/protocol-version.ts index ed235094473..76e1252640a 100644 --- a/src/shared/protocol-version.ts +++ b/src/shared/protocol-version.ts @@ -121,11 +121,14 @@ export const AGENT_SESSION_HOST_AUTHORITY_RUNTIME_CAPABILITY = 'agent-session.host-authority.v1' as const export const AGENT_SESSION_OMP_RESUME_PATH_RUNTIME_CAPABILITY = 'agent-session.omp-resume-path.v1' as const -// Why: structured sessions are journal-backed, not PTY-backed, so a client that -// cannot read them must not see them at all — it would render an agent tab it -// can neither display nor drive. The host also refuses every agentSession.* -// method from a connection that does not advertise this. +// Why: structured sessions are journal-backed, not PTY-backed, so an incapable client must not +// receive their journal or drive their lifecycle. Mobile may receive a metadata-only placeholder; +// the host still refuses agentSession.* methods and destructive tab mutations without capability. export const STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY = 'agent-session.structured.v1' as const +// Why: paired clients advertise Claude-structured support so the host can gate its agent-specific +// journal and lifecycle surfaces independently from Codex support. +export const CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY = + 'agent-session.structured.claude.v1' as const // Why: paired structured clients explicitly hold every visible session surface, allowing the host // to stop provider children after the last surface closes without tying lifetime to a transport. export const STRUCTURED_AGENT_SESSION_HOLD_RUNTIME_CAPABILITY = diff --git a/src/shared/remote-foreground-evidence-admission.ts b/src/shared/remote-foreground-evidence-admission.ts index 335d724d7b8..b9f060a346b 100644 --- a/src/shared/remote-foreground-evidence-admission.ts +++ b/src/shared/remote-foreground-evidence-admission.ts @@ -40,7 +40,16 @@ export function admitRemoteForegroundEvidence( 0, admission.receivedAtMonotonic - admission.requestStartedAtMonotonic ) - if (value.capturedAgeMs + receiveDelay > REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS) { + // The larger of the two, never their sum: `ps` runs INSIDE this round trip, so its duration is + // already in `receiveDelay`, and `capturedAgeMs` -- stamped at capture start -- is that same + // duration measured on the host's clock. Adding them charged the capture twice and halved the + // budget this ceiling actually grants a host, from ~2.0s of `ps` to ~1.0s: a 1.2s capture + // stamped 1200 and arrived at 1300, summed to 2500, and was refused as too old at 1.3s. + // + // Not the same shape as the sweep's gate, which sums deliberately and correctly: + // `evidenceAgeSinceListingMs` is stamped AFTER the listing arrives, so it measures only + // planning time and overlaps nothing. + if (Math.max(value.capturedAgeMs, receiveDelay) > REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS) { return null } if ( diff --git a/src/shared/remote-foreground-evidence.test.ts b/src/shared/remote-foreground-evidence.test.ts index 10b9371b7c0..2fc1f8f93fc 100644 --- a/src/shared/remote-foreground-evidence.test.ts +++ b/src/shared/remote-foreground-evidence.test.ts @@ -59,14 +59,82 @@ describe('remote foreground evidence contract', () => { expect( admitRemoteForegroundEvidence(live, { ...base, expectedIncarnationId: 'inc-2' }) ).toBeNull() + // `+ 1`, not the ceiling itself: the ceiling used to be crossed by the ceiling plus this + // admission's 10ms round trip, which was the capture being charged a second time. expect( admitRemoteForegroundEvidence( - { ...live, capturedAgeMs: REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS }, + { ...live, capturedAgeMs: REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS + 1 }, base ) ).toBeNull() }) + // The capture runs inside the round trip, so `capturedAgeMs` and `receiveDelay` measure + // overlapping intervals on two clocks. Summing them charged `ps` twice and halved the budget + // this ceiling grants a host; these pin the boundary on the surviving measurement. + describe('does not charge the capture twice', () => { + const admission = { + expectedPtyId: 'pty-1', + expectedIncarnationId: 'inc-1', + requestStartedAtMonotonic: 0, + receivedAtMonotonic: 0, + lastAuthorityGeneration: 'host-a', + lastObservationEpoch: 3 + } + + it('admits a capture whose duration alone is inside the ceiling', () => { + // 1,200ms on the host, 1,300ms round trip: the same 1.2s of `ps` seen twice. The sum said + // 2,500 and refused it; the observation is 1.3s old. + expect( + admitRemoteForegroundEvidence( + { ...live, capturedAgeMs: 1_200 }, + { ...admission, receivedAtMonotonic: 1_300 } + ) + ).not.toBeNull() + }) + + it.each([ + ['host-measured', REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS - 1, 0], + ['client-measured', 0, REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS - 1] + ])('admits at the %s boundary', (_label, capturedAgeMs, receiveDelay) => { + expect( + admitRemoteForegroundEvidence( + { ...live, capturedAgeMs }, + { ...admission, receivedAtMonotonic: receiveDelay } + ) + ).not.toBeNull() + }) + + it.each([ + ['host-measured', REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS + 1, 0], + ['client-measured', 0, REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS + 1] + ])('refuses one step past the %s boundary', (_label, capturedAgeMs, receiveDelay) => { + expect( + admitRemoteForegroundEvidence( + { ...live, capturedAgeMs }, + { ...admission, receivedAtMonotonic: receiveDelay } + ) + ).toBeNull() + }) + + it('still admits the prompt unverifiable a capture over its budget produces', () => { + // The reason the evidence path gives up on a slow capture rather than publishing a late + // truthful one: a REFUSED record is what bumps `consecutiveInspectionErrors` and stalls the + // completion poller, while an admitted `unverifiable` costs a poll and nothing else. + expect( + admitRemoteForegroundEvidence( + { + ...live, + verdict: 'unverifiable' as const, + reason: 'process_table_unreadable', + capturedAgeMs: 0 + }, + admission + ) + ).not.toBeNull() + }) + }) + it('rejects delayed observations from a previously accepted host generation', () => { const knownAuthorityGenerations = new Set(['host-a', 'host-b']) const admission = { diff --git a/src/shared/runtime-mobile-session-tab-contracts.ts b/src/shared/runtime-mobile-session-tab-contracts.ts index d43c43dd4e0..01a7b1ba4da 100644 --- a/src/shared/runtime-mobile-session-tab-contracts.ts +++ b/src/shared/runtime-mobile-session-tab-contracts.ts @@ -93,7 +93,7 @@ export type RuntimeMobileSessionAgentTab = { id: string title: string sessionId: string - agent: 'codex' + agent: 'claude' | 'codex' color?: string | null isPinned?: boolean isActive: boolean diff --git a/src/shared/shell-process-readiness.test.ts b/src/shared/shell-process-readiness.test.ts index 9c3abad9783..a31f8ba7737 100644 --- a/src/shared/shell-process-readiness.test.ts +++ b/src/shared/shell-process-readiness.test.ts @@ -2,7 +2,11 @@ import { mkdir, mkdtemp, rm, symlink } from 'node:fs/promises' import { tmpdir } from 'node:os' import { basename, dirname, join } from 'node:path' import { describe, expect, it } from 'vitest' -import { parseDarwinExecutablePath, resolveShellExecutablePath } from './shell-process-readiness' +import { + parseDarwinExecutablePath, + resolveInstalledShellExecutablePaths, + resolveShellExecutablePath +} from './shell-process-readiness' describe('shell process readiness', () => { it('extracts the primary text image from macOS lsof output', () => { @@ -70,4 +74,46 @@ describe('shell process readiness', () => { } } ) + + it.skipIf(process.platform === 'win32')( + 'lists every PATH installation of a shell name, deduplicated and canonical', + async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-installed-shells-')) + const first = join(root, 'first') + const second = join(root, 'second') + const missing = join(root, 'missing') + await mkdir(first) + await mkdir(second) + await symlink(process.execPath, join(first, 'shell-name')) + await symlink(process.execPath, join(second, 'shell-name')) + try { + const canonical = await resolveShellExecutablePath(process.execPath, root, '') + await expect( + resolveInstalledShellExecutablePaths('shell-name', root, `${first}:${missing}:${second}`) + ).resolves.toEqual([canonical]) + } finally { + await rm(root, { recursive: true, force: true }) + } + } + ) + + it.skipIf(process.platform === 'win32')( + 'omits a same-name executable that no PATH entry reaches', + async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-offpath-shell-')) + const onPath = join(root, 'bin') + const offPath = join(root, 'dropped') + await mkdir(onPath) + await mkdir(offPath) + await symlink(process.execPath, join(onPath, 'shell-name')) + await symlink(process.execPath, join(offPath, 'shell-name')) + try { + await expect( + resolveInstalledShellExecutablePaths('shell-name', root, onPath) + ).resolves.not.toContain(join(offPath, 'shell-name')) + } finally { + await rm(root, { recursive: true, force: true }) + } + } + ) }) diff --git a/src/shared/shell-process-readiness.ts b/src/shared/shell-process-readiness.ts index ebb94c13eb5..c9976b0d03e 100644 --- a/src/shared/shell-process-readiness.ts +++ b/src/shared/shell-process-readiness.ts @@ -56,12 +56,12 @@ export async function readShellProcessReadiness( : null } -export async function resolveShellExecutablePath( +function shellExecutableCandidates( shellPath: string, cwd: string, pathEnv: string | undefined -): Promise { - const candidates = shellPath.includes('/') +): string[] { + return shellPath.includes('/') ? [isAbsolute(shellPath) ? shellPath : resolve(cwd, shellPath)] : ( pathEnv ?? @@ -69,14 +69,42 @@ export async function resolveShellExecutablePath( ) .split(delimiter) .map((entry) => resolve(isAbsolute(entry) ? entry : resolve(cwd, entry), shellPath)) - for (const candidate of candidates) { - try { - await access(candidate, constants.X_OK) - const canonicalPath = await realpath(candidate) - if ((await stat(canonicalPath)).isFile()) { - return canonicalPath - } - } catch {} +} + +async function canonicalizeExecutable(candidate: string): Promise { + try { + await access(candidate, constants.X_OK) + const canonicalPath = await realpath(candidate) + return (await stat(canonicalPath)).isFile() ? canonicalPath : null + } catch { + return null + } +} + +export async function resolveShellExecutablePath( + shellPath: string, + cwd: string, + pathEnv: string | undefined +): Promise { + for (const candidate of shellExecutableCandidates(shellPath, cwd, pathEnv)) { + const canonicalPath = await canonicalizeExecutable(candidate) + if (canonicalPath) { + return canonicalPath + } } return null } + +/** Every canonical executable `shellName` names on `pathEnv` — the installations a + * startup profile could legitimately `exec` into, and nothing a dropped-in binary + * outside the search path can reach. `shellName` must be a bare name. */ +export async function resolveInstalledShellExecutablePaths( + shellName: string, + cwd: string, + pathEnv: string | undefined +): Promise { + const canonicalPaths = await Promise.all( + shellExecutableCandidates(shellName, cwd, pathEnv).map(canonicalizeExecutable) + ) + return [...new Set(canonicalPaths.filter((path): path is string => path !== null))] +} diff --git a/src/shared/shell-ready-marker-timing.ts b/src/shared/shell-ready-marker-timing.ts new file mode 100644 index 00000000000..7c73afb4c31 --- /dev/null +++ b/src/shared/shell-ready-marker-timing.ts @@ -0,0 +1,18 @@ +/** + * When in a shell's startup Orca's OSC 777 ready marker is published. + * + * Why this is a decision of its own: it is what separates "waiting for the marker + * is free" from "waiting for the marker costs the user real startup latency", and + * all three transports (daemon, relay, local provider) have to answer it the same + * way or a startup command is delivered twice on one of them. + */ + +/** + * True when the marker rides the shell's line editor (zsh `precmd`, bash + * `PROMPT_COMMAND`), so it arrives at the same moment the prompt can accept input. + * Every other wrapped shell emits it from startup, ahead of the reader. + */ +export function shellReadyMarkerComesFromLineEditor(shellPath: string): boolean { + const shellName = shellPath.replace(/\\/g, '/').split('/').pop()?.toLowerCase() ?? '' + return shellName === 'bash' || shellName === 'zsh' +} diff --git a/src/shared/ssh-relay-pty-ownership-proof.test.ts b/src/shared/ssh-relay-pty-ownership-proof.test.ts index b17f96808f9..92e523d3e25 100644 --- a/src/shared/ssh-relay-pty-ownership-proof.test.ts +++ b/src/shared/ssh-relay-pty-ownership-proof.test.ts @@ -174,6 +174,52 @@ describe('planRelayPtySweep', () => { expect(reasonFor(plan, 'pty-1')).toBe('host foreground observation is malformed') }) + // The kill gate's own boundary. Unlike the renderer's admission this sum is correct -- + // `evidenceAgeSinceListingMs` is stamped after the listing ARRIVES, so it measures planning time + // and overlaps the host's capture window not at all. + it.each([ + ['at the ceiling', RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, 0], + ['at the ceiling once planning is counted', RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS - 10, 10] + ])('still sweeps on an observation %s', (_label, capturedAgeMs, evidenceAgeSinceListingMs) => { + const plan = planRelayPtySweep( + [orphan({ foregroundProcessEvidence: { ...idleShell(), capturedAgeMs } })], + context({ evidenceAgeSinceListingMs }) + ) + + expect(plan.sweep).toEqual([{ ptyId: 'pty-1', incarnationId: 'inc-1' }]) + }) + + it.each([ + ['the capture alone', RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS + 1, 0], + ['the capture plus planning', RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, 1] + ])('refuses the stop one step past the ceiling on %s', (_label, capturedAgeMs, sinceListing) => { + const plan = planRelayPtySweep( + [orphan({ foregroundProcessEvidence: { ...idleShell(), capturedAgeMs } })], + context({ evidenceAgeSinceListingMs: sinceListing }) + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe( + 'host foreground observation is too old to authorize a stop' + ) + }) + + it('refuses the stop on a capture slow enough to be worth waiting out', () => { + // Measured `ps` with PS_ARGS: 2.5-9.0s on a 2,002-process laptop, 4.0-18.6s at load 46. This + // gate tolerates the fast end and refuses the rest, so waiting out a slow capture cannot buy a + // sweep -- which is why the evidence path gives up on one instead of blocking a connect for it. + for (const capturedAgeMs of [6_140, 9_010, 18_600]) { + const plan = planRelayPtySweep( + [orphan({ foregroundProcessEvidence: { ...idleShell(), capturedAgeMs } })], + context() + ) + expect(plan.sweep, `capturedAgeMs=${capturedAgeMs}`).toEqual([]) + expect(reasonFor(plan, 'pty-1'), `capturedAgeMs=${capturedAgeMs}`).toBe( + 'host foreground observation is too old to authorize a stop' + ) + } + }) + it('never sweeps on an evidence record whose other host stamps are malformed', () => { const plan = planRelayPtySweep( [ diff --git a/src/shared/structured-agent-session-coalescer.test.ts b/src/shared/structured-agent-session-coalescer.test.ts new file mode 100644 index 00000000000..770b08308af --- /dev/null +++ b/src/shared/structured-agent-session-coalescer.test.ts @@ -0,0 +1,56 @@ +import { describe, expect, it } from 'vitest' +import type { AgentSessionSubscribeEvent } from './agent-session-wire' +import { createStructuredAgentSessionEventCoalescer } from './structured-agent-session-coalescer' + +function batch( + sequence: number, + backgroundTasks?: Extract['backgroundTasks'] +): Extract { + return { + type: 'batch', + sessionId: 'session-1', + batch: { + cursor: { epoch: 'epoch-1', sequence }, + items: [], + removedItemIds: [], + submissions: [] + }, + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) + } +} + +describe('structured agent session event coalescer', () => { + it('preserves background task state when a journal batch follows it', () => { + const events: AgentSessionSubscribeEvent[] = [] + const coalescer = createStructuredAgentSessionEventCoalescer((event) => events.push(event)) + + coalescer.push( + batch(1, { + state: 'monitoring', + tasks: [{ id: 'task-1', kind: 'command', description: 'run the build' }] + }) + ) + coalescer.push(batch(2)) + coalescer.flush() + + expect(events).toHaveLength(1) + expect(events[0]).toMatchObject({ + backgroundTasks: { + state: 'monitoring', + tasks: [{ id: 'task-1', kind: 'command', description: 'run the build' }] + } + }) + }) + + it('keeps an explicit terminal state as the newest coalesced value', () => { + const events: AgentSessionSubscribeEvent[] = [] + const coalescer = createStructuredAgentSessionEventCoalescer((event) => events.push(event)) + + coalescer.push(batch(1, { state: 'monitoring' })) + coalescer.push(batch(1, null)) + coalescer.flush() + + expect(events).toHaveLength(1) + expect(events[0]).toMatchObject({ backgroundTasks: null }) + }) +}) diff --git a/src/shared/structured-agent-session-coalescer.ts b/src/shared/structured-agent-session-coalescer.ts index 51bc7fa0537..fe982d67a69 100644 --- a/src/shared/structured-agent-session-coalescer.ts +++ b/src/shared/structured-agent-session-coalescer.ts @@ -35,7 +35,15 @@ function mergeBatch( ...(right.fence !== undefined || left.fence !== undefined ? { fence: right.fence ?? left.fence } : {}), - ...(right.handoff || left.handoff ? { handoff: right.handoff ?? left.handoff } : {}) + ...(right.handoff || left.handoff ? { handoff: right.handoff ?? left.handoff } : {}), + ...(right.backgroundTasks !== undefined || left.backgroundTasks !== undefined + ? { + backgroundTasks: + right.backgroundTasks !== undefined + ? right.backgroundTasks + : (left.backgroundTasks ?? null) + } + : {}) } } diff --git a/src/shared/structured-agent-session-composer.test.ts b/src/shared/structured-agent-session-composer.test.ts new file mode 100644 index 00000000000..38fed5ddbc1 --- /dev/null +++ b/src/shared/structured-agent-session-composer.test.ts @@ -0,0 +1,31 @@ +import { describe, expect, it } from 'vitest' +import { + isStructuredAgentSessionComposerCommand, + structuredSlashCommands +} from './structured-agent-session-composer' + +describe('structuredSlashCommands', () => { + // The composer menu and the dispatcher read this one list. When they disagreed, + // a Claude session was offered Codex-only tokens that missed the command guard + // and reached the model as literal prompt text instead of erroring. + it.each(['codex', 'claude'] as const)('offers %s only commands it also accepts', (agent) => { + const offered = structuredSlashCommands(agent) + expect(offered.length).toBeGreaterThan(0) + for (const command of offered) { + expect(isStructuredAgentSessionComposerCommand(`/${command.name}`, agent)).toBe(true) + } + }) + + it('offers each agent its own catalog', () => { + const claude = structuredSlashCommands('claude').map((command) => command.name) + expect(claude).toContain('compact') + expect(claude).not.toContain('vim') + expect(structuredSlashCommands('codex').map((command) => command.name)).toContain('vim') + }) + + it('offers effort to every structured agent', () => { + for (const agent of ['codex', 'claude'] as const) { + expect(structuredSlashCommands(agent).map((command) => command.name)).toContain('effort') + } + }) +}) diff --git a/src/shared/structured-agent-session-composer.ts b/src/shared/structured-agent-session-composer.ts index 420651aba5b..18bdeab001d 100644 --- a/src/shared/structured-agent-session-composer.ts +++ b/src/shared/structured-agent-session-composer.ts @@ -35,7 +35,10 @@ function commandParts(text: string): { name: string; argument: string } | null { return match ? { name: match[1]!.toLowerCase(), argument: match[2]?.trim() ?? '' } : null } -function structuredSlashCommands(agent: AgentType): readonly SlashCommandSuggestion[] { +/** The command catalog a structured session offers and accepts. The composer menu + * and the dispatcher must read the same list, or a menu pick falls through the + * command guard and reaches the model as literal prompt text. */ +export function structuredSlashCommands(agent: AgentType): readonly SlashCommandSuggestion[] { if (agent === 'codex') { return STRUCTURED_AGENT_SESSION_SLASH_COMMANDS } diff --git a/src/shared/structured-agent-session-mutation.ts b/src/shared/structured-agent-session-mutation.ts index ccc80475c93..79c82095f29 100644 --- a/src/shared/structured-agent-session-mutation.ts +++ b/src/shared/structured-agent-session-mutation.ts @@ -26,6 +26,33 @@ export function structuredAgentSessionPayloadFingerprint(input: { return Array.from(bytes, (byte) => byte.toString(16).padStart(2, '0')).join('') } +export function structuredAgentSessionCreateFingerprint(input: { + sessionId: string + worktree: string + agent: 'claude' | 'codex' +}): string { + return structuredAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: input.sessionId, + fields: { + worktree: input.worktree, + agent: input.agent + } + }) +} + +export function showStructuredAgentSessionChoice(input: { + hostCapability: boolean + workspaceSupport: boolean + agent: string +}): boolean { + return ( + input.hostCapability && + input.workspaceSupport && + (input.agent === 'claude' || input.agent === 'codex') + ) +} + export function createStructuredAgentSessionOperationId( randomUuid: () => string, now: number = Date.now() diff --git a/src/shared/structured-agent-session-options.test.ts b/src/shared/structured-agent-session-options.test.ts index 72f817c7646..0f21efadd98 100644 --- a/src/shared/structured-agent-session-options.test.ts +++ b/src/shared/structured-agent-session-options.test.ts @@ -49,8 +49,12 @@ describe('structured agent session options', () => { models: CODEX_SESSION_OPTION_CATALOG.models, record: bridgeRecord, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) + // Same catalog, same `dispatched` vocabulary — only the transport separates them. + expect(structured.every((descriptor) => descriptor.transport === 'agent-session')).toBe(true) + expect(bridge.every((descriptor) => descriptor.transport === 'catalog')).toBe(true) expect(bridge[0]).toMatchObject({ action: { type: 'agent-picker' } }) expect(bridge.find((descriptor) => descriptor.id === 'effort')).toMatchObject({ action: { type: 'agent-picker' } diff --git a/src/shared/structured-agent-session-options.ts b/src/shared/structured-agent-session-options.ts index 94f5f45351a..d746a52025c 100644 --- a/src/shared/structured-agent-session-options.ts +++ b/src/shared/structured-agent-session-options.ts @@ -76,10 +76,14 @@ export function applyStructuredAgentSessionOptions( seed: AgentSessionOptionCatalog, result: AgentSessionOptionsResult ): StructuredAgentSessionOptionState { - applyNativeChatReportedSessionOptions(state.record, { - model: result.current.model, - ...(result.current.effort ? { effort: result.current.effort } : {}) - }) + applyNativeChatReportedSessionOptions( + state.record, + { + model: result.current.model, + ...(result.current.effort ? { effort: result.current.effort } : {}) + }, + result.current.confirmed ?? [] + ) return { ...state, catalog: structuredAgentSessionOptionCatalog(seed, result) } } diff --git a/src/shared/structured-agent-session-projection.ts b/src/shared/structured-agent-session-projection.ts index 71cffa43762..94938dc955f 100644 --- a/src/shared/structured-agent-session-projection.ts +++ b/src/shared/structured-agent-session-projection.ts @@ -6,6 +6,22 @@ function boundedText(payload: { head: string; truncated: boolean; byteLength: nu return payload.truncated ? `${payload.head}\n… (${payload.byteLength} bytes)` : payload.head } +/** The markers a clipped payload carries in its own text, anchored to the end + * so nothing that merely looks like one inside the body can match. */ +const BOUNDED_TEXT_MARKERS = [ + /\n… \(\d+ bytes\)$/, + /\n\[Orca: output truncated — \d+ bytes total, digest [0-9a-f]+\]$/ +] + +/** Recovers the clipped body from a bounded payload's text, and says whether a + * marker was there. A reader that treats the text as content renders the + * marker as a line of it — with a line number, which reads as a real position + * in the file — and reports the body as complete. */ +export function stripBoundedTextMarker(text: string): { text: string; truncated: boolean } { + const stripped = BOUNDED_TEXT_MARKERS.reduce((value, marker) => value.replace(marker, ''), text) + return { text: stripped, truncated: stripped.length !== text.length } +} + function itemBlocks(item: AgentJournalRenderItem): { role: NativeChatMessage['role'] blocks: NativeChatBlock[] diff --git a/src/shared/structured-agent-session-reducer.test.ts b/src/shared/structured-agent-session-reducer.test.ts index 222db53a564..99ced770641 100644 --- a/src/shared/structured-agent-session-reducer.test.ts +++ b/src/shared/structured-agent-session-reducer.test.ts @@ -250,4 +250,129 @@ describe('structured agent session reducer', () => { expect(state.submissions[0]?.clientMessageId).toBe('client-44') expect(state.submissions.at(-1)?.clientMessageId).toBe('client-299') }) + + it('projects additive background task state without changing transcript identity', () => { + const initial = reduceStructuredAgentSession(EMPTY_STRUCTURED_AGENT_SESSION, { + type: 'event', + event: { + type: 'snapshot', + sessionId: 'session-a', + fence: 1, + page: hydrationPage([item('message', 1)]), + backgroundTasks: null + } + }) + const monitoring = reduceStructuredAgentSession(initial, { + type: 'event', + event: { + type: 'batch', + sessionId: 'session-a', + batch: { + cursor: initial.cursor!, + items: [], + removedItemIds: [], + submissions: [] + }, + fence: 1, + backgroundTasks: { state: 'monitoring' } + } + }) + + expect(monitoring.backgroundTasks).toEqual({ state: 'monitoring' }) + expect(monitoring.items).toBe(initial.items) + }) + + it('returns the same state for duplicate background task publications', () => { + const monitoring = reduceStructuredAgentSession(EMPTY_STRUCTURED_AGENT_SESSION, { + type: 'event', + event: { + type: 'snapshot', + sessionId: 'session-a', + fence: 1, + page: hydrationPage([item('message', 1)]), + backgroundTasks: { state: 'monitoring' } + } + }) + const duplicate = reduceStructuredAgentSession(monitoring, { + type: 'event', + event: { + type: 'batch', + sessionId: 'session-a', + batch: { + cursor: monitoring.cursor!, + items: [], + removedItemIds: [], + submissions: [] + }, + fence: 1, + backgroundTasks: { state: 'monitoring' } + } + }) + + expect(duplicate).toBe(monitoring) + }) + + it('applies background task roster changes without a journal update', () => { + const monitoring = reduceStructuredAgentSession(EMPTY_STRUCTURED_AGENT_SESSION, { + type: 'event', + event: { + type: 'snapshot', + sessionId: 'session-a', + fence: 1, + page: hydrationPage([item('message', 1)]), + backgroundTasks: { + state: 'monitoring', + tasks: [{ id: 'task-1', kind: 'command', description: 'first command' }] + } + } + }) + const changed = reduceStructuredAgentSession(monitoring, { + type: 'event', + event: { + type: 'batch', + sessionId: 'session-a', + batch: { + cursor: monitoring.cursor!, + items: [], + removedItemIds: [], + submissions: [] + }, + fence: 1, + backgroundTasks: { + state: 'monitoring', + tasks: [{ id: 'task-1', kind: 'agent', description: 'review the change' }] + } + } + }) + + expect(changed).not.toBe(monitoring) + expect(changed.backgroundTasks?.tasks).toEqual([ + { id: 'task-1', kind: 'agent', description: 'review the change' } + ]) + expect(changed.items).toBe(monitoring.items) + }) + + it('clears additive background state when a replacement snapshot omits the field', () => { + const monitoring = reduceStructuredAgentSession(EMPTY_STRUCTURED_AGENT_SESSION, { + type: 'event', + event: { + type: 'snapshot', + sessionId: 'session-a', + fence: 1, + page: hydrationPage([item('message', 1)]), + backgroundTasks: { state: 'monitoring' } + } + }) + const withoutCapability = reduceStructuredAgentSession(monitoring, { + type: 'event', + event: { + type: 'snapshot', + sessionId: 'session-a', + fence: 2, + page: hydrationPage([item('message', 1)]) + } + }) + + expect(withoutCapability.backgroundTasks).toBeUndefined() + }) }) diff --git a/src/shared/structured-agent-session-reducer.ts b/src/shared/structured-agent-session-reducer.ts index 48029b5910e..0a330539548 100644 --- a/src/shared/structured-agent-session-reducer.ts +++ b/src/shared/structured-agent-session-reducer.ts @@ -4,6 +4,7 @@ import type { AgentJournalSubmission } from './agent-session-journal-types' import type { + AgentSessionBackgroundTaskState, AgentSessionHandoffStatus, AgentSessionHistoryPage, AgentSessionSubscribeEvent @@ -19,6 +20,7 @@ export type StructuredAgentSessionState = { status: 'idle' | 'loading' | 'ready' | 'error' error?: string handoff: AgentSessionHandoffStatus | null + backgroundTasks?: AgentSessionBackgroundTaskState | null } export type StructuredAgentSessionAction = @@ -42,10 +44,35 @@ export const EMPTY_STRUCTURED_AGENT_SESSION: StructuredAgentSessionState = { const MAX_RETAINED_SUBMISSIONS = 256 +function backgroundTaskStatesEqual( + left: AgentSessionBackgroundTaskState | null | undefined, + right: AgentSessionBackgroundTaskState | null | undefined +): boolean { + if (left === right) { + return true + } + if (!left || !right || left.state !== right.state) { + return false + } + if (left.tasks === right.tasks) { + return true + } + if (!left.tasks || !right.tasks || left.tasks.length !== right.tasks.length) { + return false + } + return left.tasks.every( + (task, index) => + task.id === right.tasks?.[index]?.id && + task.kind === right.tasks[index]?.kind && + task.description === right.tasks[index]?.description + ) +} + function replacePage( page: AgentSessionHistoryPage, fence: number, - handoff?: AgentSessionHandoffStatus + handoff?: AgentSessionHandoffStatus, + backgroundTasks?: AgentSessionBackgroundTaskState | null ): StructuredAgentSessionState { return { epoch: page.epoch, @@ -55,7 +82,12 @@ function replacePage( submissions: page.submissions, hasOlder: page.hasOlder, status: 'ready', - handoff: handoff ?? null + handoff: handoff ?? null, + ...(backgroundTasks !== undefined + ? { backgroundTasks } + : page.backgroundTasks !== undefined + ? { backgroundTasks: page.backgroundTasks } + : {}) } } @@ -113,12 +145,23 @@ export function reduceStructuredAgentSession( state.cursor && (!pageCursor || pageCursor.sequence <= state.cursor.sequence) ) { + const backgroundTasksChanged = + action.page.backgroundTasks !== undefined && + !backgroundTaskStatesEqual(action.page.backgroundTasks, state.backgroundTasks) if ( pageCursor?.sequence === state.cursor.sequence && - action.page.fence !== undefined && - action.page.fence !== state.fence + ((action.page.fence !== undefined && action.page.fence !== state.fence) || + backgroundTasksChanged) ) { - return { ...state, fence: action.page.fence, status: 'ready', error: undefined } + return { + ...state, + ...(action.page.fence !== undefined ? { fence: action.page.fence } : {}), + ...(action.page.backgroundTasks !== undefined + ? { backgroundTasks: action.page.backgroundTasks } + : {}), + status: 'ready', + error: undefined + } } return state } @@ -133,7 +176,12 @@ export function reduceStructuredAgentSession( : action.page.submissions, hasOlder: action.page.hasOlder, status: 'ready', - handoff: state.handoff + handoff: state.handoff, + ...(action.page.backgroundTasks !== undefined + ? { backgroundTasks: action.page.backgroundTasks } + : state.backgroundTasks !== undefined + ? { backgroundTasks: state.backgroundTasks } + : {}) } } if (action.type === 'older-page') { @@ -152,7 +200,7 @@ export function reduceStructuredAgentSession( return state } if (event.type === 'snapshot' || event.type === 'reset') { - return replacePage(event.page, event.fence, event.handoff) + return replacePage(event.page, event.fence, event.handoff, event.backgroundTasks) } if (state.epoch !== event.batch.cursor.epoch) { return state @@ -160,15 +208,37 @@ export function reduceStructuredAgentSession( if (state.cursor && event.batch.cursor.sequence < state.cursor.sequence) { return state } + const backgroundTasks = + event.backgroundTasks !== undefined ? event.backgroundTasks : state.backgroundTasks + const journalUnchanged = + event.batch.items.length === 0 && + event.batch.removedItemIds.length === 0 && + event.batch.submissions.length === 0 + if ( + event.batch.cursor.sequence === state.cursor?.sequence && + journalUnchanged && + (event.fence === undefined || event.fence === state.fence) && + (event.handoff === undefined || event.handoff === state.handoff) && + backgroundTaskStatesEqual(backgroundTasks, state.backgroundTasks) && + state.status === 'ready' && + state.error === undefined + ) { + return state + } return { ...state, cursor: event.batch.cursor, fence: event.fence ?? state.fence, - items: mergeItems(state.items, event.batch.items, event.batch.removedItemIds), - submissions: mergeSubmissions(state.submissions, event.batch.submissions), + items: journalUnchanged + ? state.items + : mergeItems(state.items, event.batch.items, event.batch.removedItemIds), + submissions: journalUnchanged + ? state.submissions + : mergeSubmissions(state.submissions, event.batch.submissions), status: 'ready', error: undefined, - handoff: event.handoff ?? state.handoff + handoff: event.handoff ?? state.handoff, + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) } } diff --git a/src/shared/terminal-partial-escape-tail.fuzz.test.ts b/src/shared/terminal-partial-escape-tail.fuzz.test.ts new file mode 100644 index 00000000000..9f3d0b0e7b9 --- /dev/null +++ b/src/shared/terminal-partial-escape-tail.fuzz.test.ts @@ -0,0 +1,154 @@ +import { describe, expect, it } from 'vitest' +import { + advancePartialEscapeTail, + extractPartialEscapeTail, + MAX_PARTIAL_ESCAPE_TAIL_LENGTH +} from './terminal-partial-escape-tail' + +// Differential fuzz for the ESC-free gate in `advancePartialEscapeTail`: the guarded fold must be +// byte-for-byte indistinguishable from the unguarded oracle (concat + full walk + cap) on every +// input, and must preserve the fold property extract(a + b) === extract(extract(a) + b). +// A 25.6M-case out-of-band sweep (exhaustive len<=5, 2000 x 16 KB random chunks, every BMP code +// unit) found 0 divergences; this is the CI-sized slice of it. + +const oracle = (pending: string, chunk: string): string => { + const tail = extractPartialEscapeTail(pending + chunk) + return tail.length > MAX_PARTIAL_ESCAPE_TAIL_LENGTH ? '' : tail +} + +// Every byte class the scanner branches on, plus code units the gate's `includes` must not confuse. +const ALPHABET = [ + '\x1b', + '\x18', + '\x1a', + '\x07', + '\\', + '[', + ']', + 'P', + 'X', + '^', + '_', + '(', + '0', + ';', + 'm', + '\n', + '\x7f', + '\x9c', + 'é', + '\u{1f600}', + '\ud83d', + '\udc00' +] + +// One representative of every state the scanner can be left in. +const PENDINGS = [ + '', + '\x1b', + '\x1b[', + '\x1b[3', + '\x1b]0;ti', + '\x1b]0;ti\x1b', + '\x1bP dcs', + '\x1bPx\x1b', + '\x1b(', + '\x1b ', + '\x1b[1;2;3' +] + +const SEQUENCES = [ + '\x1b[1;31m', + '\x1b]0;my title\x07', + '\x1b]8;;https://example.com\x1b\\', + '\x1bPq#0;2;0;0;0#0!6~\x1b\\', + '\x1b(B', + '\x1b7', + '\x1b[?1049h', + '\x1b]52;c;aGVsbG8=\x1b\\', + 'ab\x1b[2Jcd' +] + +// Yields {text, depth} because an astral symbol is two UTF-16 code units: filtering on +// `text.length` would silently drop every depth-N string containing one, so the corpus would +// not be exhaustive at depth N the way the test names claim. +function* stringsUpTo(maxDepth: number): Generator<{ depth: number; text: string }> { + yield { depth: 0, text: '' } + for (let depth = 1; depth <= maxDepth; depth++) { + const digits = Array.from({ length: depth }, () => 0) + for (;;) { + yield { depth, text: digits.map((digit) => ALPHABET[digit]).join('') } + let place = depth - 1 + while (place >= 0 && ++digits[place] === ALPHABET.length) { + digits[place--] = 0 + } + if (place < 0) { + break + } + } + } +} + +describe('advancePartialEscapeTail differential fuzz', () => { + let checked = 0 + const check = (pending: string, chunk: string): void => { + checked++ + const actual = advancePartialEscapeTail(pending, chunk) + if (actual !== oracle(pending, chunk)) { + expect.fail(`gate diverged: ${JSON.stringify({ pending, chunk, actual })}`) + } + const whole = extractPartialEscapeTail(pending + chunk) + if ( + whole.length <= MAX_PARTIAL_ESCAPE_TAIL_LENGTH && + advancePartialEscapeTail(extractPartialEscapeTail(pending), chunk) !== whole + ) { + expect.fail(`fold property broke: ${JSON.stringify({ pending, chunk })}`) + } + } + + it('matches the unguarded oracle on every chunk up to length 4', () => { + for (const { text: chunk } of stringsUpTo(3)) { + for (const pending of PENDINGS) { + check(pending, chunk) + } + } + for (const { depth, text: chunk } of stringsUpTo(4)) { + if (depth === 4) { + check('', chunk) + check('\x1b[', chunk) + } + } + }) + + it('matches at every split point of known sequences', () => { + for (const sequence of SEQUENCES) { + for (let cut = 0; cut <= sequence.length; cut++) { + const afterPrefix = advancePartialEscapeTail('', sequence.slice(0, cut)) + check('', sequence.slice(0, cut)) + for (let cut2 = cut; cut2 <= sequence.length; cut2++) { + check(afterPrefix, sequence.slice(cut, cut2)) + check( + advancePartialEscapeTail(afterPrefix, sequence.slice(cut, cut2)), + sequence.slice(cut2) + ) + } + } + } + }) + + it('matches across the tail-length cap', () => { + const max = MAX_PARTIAL_ESCAPE_TAIL_LENGTH + for (const length of [max - 1, max, max + 1, max + 100]) { + const osc = `\x1b]0;${'x'.repeat(length - 4)}` + for (const chunk of ['', 'y', '\x07', '\x1b\\', '\x1b', 'plain\n', 'x'.repeat(5000)]) { + check(osc, chunk) + check('', osc + chunk) + check('\x1b]0;', osc.slice(4) + chunk) + } + } + }) + + it('ran the whole corpus', () => { + expect(checked).toBe(593_468) + }) +}) diff --git a/src/shared/terminal-partial-escape-tail.test.ts b/src/shared/terminal-partial-escape-tail.test.ts index e0ef2095ddd..3185abc4cfc 100644 --- a/src/shared/terminal-partial-escape-tail.test.ts +++ b/src/shared/terminal-partial-escape-tail.test.ts @@ -84,3 +84,43 @@ describe('advancePartialEscapeTail', () => { expect(advancePartialEscapeTail('', huge)).toBe('') }) }) + +describe('advancePartialEscapeTail ESC-free fast path', () => { + // Every pending-tail state the scanner can be left in x every chunk shape, asserted + // indistinguishable from the unconditional fold the gate sits in front of. + const pieces = [ + '', + 'plain output\n', + '\x1b[32mgreen\x1b[0m', + '\x1b[3', + '\x1b]0;title\x07', + '\x1b]0;partial', + '\x1bP dcs payload', + '\x1b', + '\x18', + '\x1a', + '\x1b]8;;https://example.com\x1b\\', + '\x1b(', + '\x1b[1;2;3' + ] + + it('matches an unconditional fold for every pending-tail and chunk pairing', () => { + for (const pending of pieces.map((piece) => extractPartialEscapeTail(piece))) { + for (const chunk of pieces) { + // The cap belongs in the expectation: `advancePartialEscapeTail` abandons a tail over + // MAX_PARTIAL_ESCAPE_TAIL_LENGTH, so comparing it against an uncapped extract would stop + // modelling the function the moment a pairing crossed the cap. + const unguarded = extractPartialEscapeTail(pending + chunk) + expect(advancePartialEscapeTail(pending, chunk), JSON.stringify({ pending, chunk })).toBe( + unguarded.length > MAX_PARTIAL_ESCAPE_TAIL_LENGTH ? '' : unguarded + ) + } + } + }) + + it('still carries a pending tail through an ESC-free chunk', () => { + expect(advancePartialEscapeTail('\x1b]0;my-title', ' still in the OSC')).toBe( + '\x1b]0;my-title still in the OSC' + ) + }) +}) diff --git a/src/shared/terminal-partial-escape-tail.ts b/src/shared/terminal-partial-escape-tail.ts index 1ee606e60f6..b1aa44ec072 100644 --- a/src/shared/terminal-partial-escape-tail.ts +++ b/src/shared/terminal-partial-escape-tail.ts @@ -144,6 +144,14 @@ export function extractPartialEscapeTail(stream: string): string { /** Ingest-time fold: advance the tracked tail with one more chunk. Returns '' * (tracking abandoned) when the tail exceeds the cap — see the cap comment. */ export function advancePartialEscapeTail(pendingTail: string, chunk: string): string { + // Why the pre-filter: `extractPartialEscapeTail` only leaves `ground` on an ESC byte, so with + // no pending tail and no ESC in the chunk the answer is always ''. Taking it here skips both + // the full-chunk concat and the per-code-unit walk on ESC-free output (build logs, `cat`, + // piped tool output) — the same gate `TerminalOscCwdTitleScanner.scan` and + // `TerminalMouseModeMirror.scan` already apply on the very same ingest path. + if (pendingTail.length === 0 && !chunk.includes('\x1b')) { + return '' + } const tail = extractPartialEscapeTail(pendingTail + chunk) return tail.length > MAX_PARTIAL_ESCAPE_TAIL_LENGTH ? '' : tail } diff --git a/src/shared/utf8-byte-limits.test.ts b/src/shared/utf8-byte-limits.test.ts index 9556a5c39bc..f5cd2c83ed8 100644 --- a/src/shared/utf8-byte-limits.test.ts +++ b/src/shared/utf8-byte-limits.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it } from 'vitest' import { clampUtf8TextPrefix, + getUtf8ByteLength, getUtf8ChunkEndIndex, isUtf8ByteLengthWithinLimit, measureUtf8ByteLength, @@ -129,3 +130,33 @@ describe('isUtf8ByteLengthWithinLimit', () => { expect(isUtf8ByteLengthWithinLimit(text, maxBytes)).toBe(expected) }) }) + +describe('isUtf8ByteLengthWithinLimit native fast path', () => { + // The bounded check runs through TextEncoder.encodeInto; it must agree with the scan it replaced + // for every boundary shape, lone surrogates included. + const samples = [ + '', + 'a', + 'ascii only text', + 'caf\u00e9', + '\u20ac\u20ac\u20ac', + '\ud83d\ude00\ud83d\ude00', + '\ud83d', + '\ude00', + 'mixed \u00e9 \u20ac \ud83d\ude00 tail', + 'x'.repeat(64) + ] + + it('matches the scanning implementation at every limit around the boundary', () => { + for (const text of samples) { + const exactBytes = getUtf8ByteLength(text) + for (let maxBytes = 1; maxBytes <= exactBytes + 2; maxBytes += 1) { + expect({ text, maxBytes, within: isUtf8ByteLengthWithinLimit(text, maxBytes) }).toEqual({ + text, + maxBytes, + within: text.length <= maxBytes && exactBytes <= maxBytes + }) + } + } + }) +}) diff --git a/src/shared/utf8-byte-limits.ts b/src/shared/utf8-byte-limits.ts index df8f99f68ad..767d5bfb877 100644 --- a/src/shared/utf8-byte-limits.ts +++ b/src/shared/utf8-byte-limits.ts @@ -51,6 +51,14 @@ export function getUtf8ByteLength(text: string): number { return measureUtf8ByteLength(text).byteLength } +// Why a native encode: the per-code-unit JS scan walks whole terminal scrollback buffers on the +// session-write path. `encodeInto` answers "does this fit in maxBytes?" in C++ — it stops at the +// destination's end, so `read < text.length` means the text needs more than maxBytes. The scratch +// buffer is reused across calls and grows to the largest limit asked for, up to this cap. +const MAX_UTF8_SCRATCH_BYTES = 1024 * 1024 +const utf8Encoder = new TextEncoder() +let utf8Scratch = new Uint8Array(0) + export function isUtf8ByteLengthWithinLimit(text: string, maxBytes: number): boolean { if (text.length === 0) { return true @@ -58,6 +66,12 @@ export function isUtf8ByteLengthWithinLimit(text: string, maxBytes: number): boo if (text.length > maxBytes) { return false } + if (Number.isSafeInteger(maxBytes) && maxBytes <= MAX_UTF8_SCRATCH_BYTES) { + if (utf8Scratch.length < maxBytes) { + utf8Scratch = new Uint8Array(maxBytes) + } + return utf8Encoder.encodeInto(text, utf8Scratch.subarray(0, maxBytes)).read === text.length + } return !measureUtf8ByteLength(text, { stopAfterBytes: maxBytes }).exceededLimit } diff --git a/src/shared/workspace-session-terminal-buffers.ts b/src/shared/workspace-session-terminal-buffers.ts index 96bb76cafe2..ef8ac6a273b 100644 --- a/src/shared/workspace-session-terminal-buffers.ts +++ b/src/shared/workspace-session-terminal-buffers.ts @@ -3,7 +3,7 @@ import type { WorkspaceSessionState } from './workspace-session-state-types' import { FLOATING_TERMINAL_WORKTREE_ID } from './constants' import { getRepoIdFromWorktreeId } from './worktree/id' import { TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT } from './terminal-scrollback-limits' -import { clampUtf8TextTail, measureUtf8ByteLength } from './utf8-byte-limits' +import { clampUtf8TextTail, isUtf8ByteLengthWithinLimit } from './utf8-byte-limits' import { parseExecutionHostId } from './execution-host' export type RepoConnection = Pick @@ -50,12 +50,7 @@ export function shouldPreserveTerminalScrollbackBuffers( } export function capTerminalScrollbackSessionBuffer(buffer: string): string { - if ( - buffer.length <= TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT && - !measureUtf8ByteLength(buffer, { - stopAfterBytes: TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT - }).exceededLimit - ) { + if (isUtf8ByteLengthWithinLimit(buffer, TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT)) { return buffer } return clampUtf8TextTail(buffer, TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT).text diff --git a/src/shared/worktree/types.ts b/src/shared/worktree/types.ts index 3e5c05a65bc..e368716dc01 100644 --- a/src/shared/worktree/types.ts +++ b/src/shared/worktree/types.ts @@ -221,4 +221,6 @@ export type DetectedWorktreeListResult = { authoritative: boolean source: DetectedWorktreeListSource worktrees: DetectedWorktree[] + /** Why a non-authoritative listing could not be scanned; additive, older hosts omit it. */ + unavailableReason?: string } diff --git a/tests/e2e/browser-loading-surface-oracle.ts b/tests/e2e/browser-loading-surface-oracle.ts new file mode 100644 index 00000000000..5fab1c78c94 --- /dev/null +++ b/tests/e2e/browser-loading-surface-oracle.ts @@ -0,0 +1,300 @@ +import { createServer } from 'node:http' +import { writeFile } from 'node:fs/promises' +import type { AddressInfo } from 'node:net' +import { expect, type Page } from '@stablyai/playwright-test' +import { PNG } from 'pngjs' + +// Hold the response, not a timer: every screenshot precedes the first document commit. +export async function observeBrowserLoadingSurface( + page: Page, + outputPath: (name: string) => string, + crashGuest?: (id: number) => Promise +) { + const pendingResponses: (() => void)[] = [] + const release = (): void => { + pendingResponses.splice(0).forEach((send) => send()) + } + let requestCount = 0 + let flushPrefix: (() => void) | undefined + let retryReady = false + const server = createServer((request, response) => { + if (request.url === '/fail' && !retryReady) { + response.destroy() + return + } + if (request.url !== '/held' && request.url !== '/fail') { + response.writeHead(204).end() + return + } + requestCount += 1 + let prefixSent = false + flushPrefix = () => { + prefixSent = true + response.writeHead(200, { 'Content-Type': 'text/html' }) + response.write( + `Surface oracle` + ) + } + pendingResponses.push(() => + response.end( + `${prefixSent ? '' : 'Surface oracle'}

Usable webpage

` + ) + ) + }) + await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve)) + const url = `http://127.0.0.1:${(server.address() as AddressInfo).port}/held` + const observations: Record[] = [] + // Freezing attachment for a screenshot must also freeze the missing-guest watchdog. + await page.clock.pauseAt(new Date()) + const attachGate = await page.evaluateHandle((heldUrl) => { + const original = Element.prototype.setAttribute + const pending: (() => void)[] = [] + Element.prototype.setAttribute = function (name, value) { + if (this.tagName === 'WEBVIEW' && name === 'src' && value === heldUrl) { + pending.push(() => original.call(this, name, value)) + return + } + original.call(this, name, value) + } + return () => { + Element.prototype.setAttribute = original + pending.splice(0).forEach((release) => release()) + } + }, url) + try { + await page.evaluate(async () => { + await window.__store!.getState().updateSettingsOrThrow({ theme: 'dark' }) + }) + await expect(page.locator('html')).toHaveClass(/dark/) + const tab = await page.evaluate((url) => { + const s = window.__store!.getState() + return s.createBrowserTab(s.activeWorktreeId!, url, { + activate: true, + title: 'Surface oracle' + }) + }, url) + let guest = page.locator(`[data-browser-overlay-tab-id="${tab.id}"] webview`) + await expect(guest).toHaveCount(1) + const capture = async (phase: string, expected: 'theme' | 'white', loading = false) => { + const state = await guest.evaluate((element) => { + const webview = element as Electron.WebviewTag + const rect = webview.closest('[data-browser-page-container]')!.getBoundingClientRect() + let loading = false + let url = '' + let attached = false + try { + loading = webview.isLoading() + url = webview.getURL() + attached = webview.getWebContentsId() > 0 + } catch { + /* Guest creation is held by the oracle. */ + } + return { + attached, + background: getComputedStyle(webview).backgroundColor, + theme: getComputedStyle(webview.closest('[data-browser-page-container]')!) + .backgroundColor, + visibility: getComputedStyle(webview).visibility, + display: getComputedStyle(webview).display, + loading, + url, + rect: { x: rect.x, y: rect.y, width: rect.width, height: rect.height } + } + }) + const target = + expected === 'white' ? [255, 255, 255] : state.theme.match(/\d+/g)!.slice(0, 3).map(Number) + let samples: number[][] = [] + const sampleSurface = async (): Promise => { + const screenshot = await page.screenshot({ path: outputPath(`${phase}.png`), scale: 'css' }) + const png = PNG.sync.read(screenshot) + samples = [0.08, 0.92].flatMap((x) => + [0.08, 0.85, 0.92].map((y) => { + const offset = + (Math.floor(state.rect.y + state.rect.height * y) * png.width + + Math.floor(state.rect.x + state.rect.width * x)) * + 4 + return [...png.data.subarray(offset, offset + 3)] + }) + ) + return samples.every((rgb) => rgb.every((c, i) => Math.abs(c - target[i]) <= 2)) + } + let pixelPass = await sampleSurface() + if (expected === 'white' && !pixelPass) { + // Loading can stop before the compositor presents the recovered guest's first frame. + await expect.poll(async () => (pixelPass = await sampleSurface())).toBe(true) + } + const statePass = + expected === 'theme' + ? state.background === state.theme || + state.visibility === 'hidden' || + state.display === 'none' + : state.url.startsWith('http://127.0.0.1:') + expect(state.loading).toBe(loading) + observations.push({ + phase, + ...state, + samples, + pixelPass, + statePass, + pass: pixelPass && statePass + }) + } + const dismissDrawHint = page.getByRole('button', { name: 'Got it', exact: true }) + if (await dismissDrawHint.isVisible()) { + await dismissDrawHint.click() + await expect(dismissDrawHint).not.toBeVisible() + } + await page.keyboard.press('Escape') + await page.mouse.move(0, 0) + await capture('pre-attach', 'theme') + await attachGate.evaluate((release) => release()) + await page.clock.resume() + await expect.poll(() => requestCount).toBe(1) + await capture('dark-held', 'theme', true) + await page.evaluate(async () => { + await window.__store!.getState().updateSettingsOrThrow({ theme: 'light' }) + }) + await expect(page.locator('html')).toHaveClass(/light/) + await capture('light-held', 'theme', true) + await page.evaluate(async () => { + await window.__store!.getState().updateSettingsOrThrow({ theme: 'dark' }) + }) + await expect(page.locator('html')).toHaveClass(/dark/) + await capture('dark-again-held', 'theme', true) + await page.emulateMedia({ colorScheme: 'light' }) + await page.evaluate(async () => { + await window.__store!.getState().updateSettingsOrThrow({ theme: 'system' }) + }) + await expect(page.locator('html')).toHaveClass(/light/) + await capture('system-light-held', 'theme', true) + await page.emulateMedia({ colorScheme: 'dark' }) + await expect(page.locator('html')).toHaveClass(/dark/) + await capture('system-dark-held', 'theme', true) + await page.emulateMedia({ colorScheme: null }) + await page.evaluate(async () => { + await window.__store!.getState().updateSettingsOrThrow({ theme: 'dark' }) + }) + flushPrefix!() + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).getTitle())) + .toBe('Surface oracle') + await capture('committed-empty', 'theme', true) + release!() + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).isLoading())) + .toBe(false) + expect( + await guest.evaluate((e) => + (e as Electron.WebviewTag).executeJavaScript( + '({ text: document.querySelector("h1").textContent, style: document.body.getAttribute("style") })' + ) + ) + ).toEqual({ text: 'Usable webpage', style: null }) + await capture('unstyled-painted', 'white') + await page.evaluate(async () => { + await window.__store!.getState().updateSettingsOrThrow({ theme: 'light' }) + }) + await expect(page.locator('html')).toHaveClass(/light/) + await capture('unstyled-light', 'white') + await page.evaluate(async () => { + await window.__store!.getState().updateSettingsOrThrow({ theme: 'dark' }) + }) + await expect(page.locator('html')).toHaveClass(/dark/) + const pane = page.locator(`[data-browser-overlay-tab-id="${tab.id}"]`) + await pane.getByRole('button', { name: 'Reload', exact: true }).click() + await expect.poll(() => requestCount).toBe(2) + await capture('reload-retained', 'white', true) + release!() + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).isLoading())) + .toBe(false) + const guestId = await guest.evaluate((e) => (e as Electron.WebviewTag).getWebContentsId()) + const worktreeId = await page.evaluate(() => window.__store!.getState().activeWorktreeId!) + await page.evaluate(() => window.__store!.getState().setActiveWorktree(null)) + await expect(pane).not.toBeVisible() + await page.evaluate( + ({ worktreeId, tabId }) => { + const s = window.__store!.getState() + s.setActiveWorktree(worktreeId) + s.setActiveBrowserTab(tabId) + }, + { worktreeId, tabId: tab.id } + ) + await expect(pane).toBeVisible() + expect(await guest.evaluate((e) => (e as Electron.WebviewTag).getWebContentsId())).toBe(guestId) + await capture('unpark-retained', 'white') + if (crashGuest) { + await crashGuest(guestId) + await expect.poll(() => requestCount).toBe(3) + await capture('recovery-held', 'theme', true) + release!() + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).isLoading())) + .toBe(false) + await capture('recovery-painted', 'white') + } + const blank = await page.evaluate(() => { + const s = window.__store!.getState() + return s.createBrowserTab(s.activeWorktreeId!, 'about:blank', { activate: true }) + }) + guest = page.locator(`[data-browser-overlay-tab-id="${blank.id}"] webview`) + await expect + .poll(() => + guest.evaluate((e) => { + try { + return (e as Electron.WebviewTag).getURL() + } catch { + return '' + } + }) + ) + .toMatch(/about:blank|data:text\/html,/) + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).isLoading())) + .toBe(false) + await page.keyboard.press('Escape') + await capture('new-tab', 'theme') + // Navigate through the real address input, retaining the blank document while the server waits. + const address = page.locator(`[data-browser-overlay-tab-id="${blank.id}"] input`).first() + await address.fill(url) + await address.press('Enter') + await expect.poll(() => requestCount).toBe(crashGuest ? 4 : 3) + await capture('new-tab-first-navigation-held', 'theme', true) + release!() + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).isLoading())) + .toBe(false) + await capture('new-tab-painted', 'white') + await address.fill(url.replace('/held', '/fail')) + await address.press('Enter') + const retry = page + .locator(`[data-browser-overlay-tab-id="${blank.id}"]`) + .getByRole('button', { name: 'Retry', exact: true }) + .filter({ hasText: 'Retry' }) + await expect(retry).toBeVisible() + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).isLoading())) + .toBe(false) + await capture('network-error', 'theme') + retryReady = true + const countBeforeRetry = requestCount + await retry.click() + await expect.poll(() => requestCount).toBeGreaterThan(countBeforeRetry) + await capture('network-retry-held', 'theme', true) + release!() + await expect(retry).not.toBeVisible() + await expect + .poll(() => guest.evaluate((e) => (e as Electron.WebviewTag).isLoading())) + .toBe(false) + await capture('network-retry-painted', 'white') + await writeFile(outputPath('observations.json'), JSON.stringify(observations, null, 2)) + return observations + } finally { + await attachGate.evaluate((release) => release()).catch(() => {}) + await page.clock.resume().catch(() => {}) + await attachGate.dispose() + release?.() + server.closeAllConnections() + await new Promise((resolve) => server.close(() => resolve())) + } +} diff --git a/tests/e2e/browser-loading-surface.spec.ts b/tests/e2e/browser-loading-surface.spec.ts new file mode 100644 index 00000000000..ac1f21eac57 --- /dev/null +++ b/tests/e2e/browser-loading-surface.spec.ts @@ -0,0 +1,21 @@ +import { expect, test } from './helpers/orca-app' +import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' +import { crashGuestRenderer } from './browser-guest-runtime-oracle' +import { observeBrowserLoadingSurface } from './browser-loading-surface-oracle' + +test('browser host follows the theme before content and preserves the webpage canvas', async ({ + orcaPage, + electronApp +}, testInfo) => { + await waitForSessionReady(orcaPage) + await ensureTerminalVisible(orcaPage) + await waitForActiveWorktree(orcaPage) + const observations = await observeBrowserLoadingSurface( + orcaPage, + (name) => testInfo.outputPath(name), + async (id) => { + await crashGuestRenderer(electronApp, id) + } + ) + expect(observations.filter((entry) => !entry.pass)).toEqual([]) +}) diff --git a/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts b/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts index d79c99b679d..a2a64897fcb 100644 --- a/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts +++ b/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts @@ -2,9 +2,14 @@ // way the terminal wire harness is: current code against a real published release. // // Three skews matter here, and none can be checked from one build alone — an old -// client must not be shown a session it cannot render, a new client must find an -// old host's missing surface cleanly, and a client's cursor must survive the host -// process that minted it. +// client must not receive a journal-backed RPC surface it cannot read, a new client +// must find an old host's missing surface cleanly, and a client's cursor must survive +// the host process that minted it. +// +// The session-tabs projection may keep a metadata-only row for an incapable mobile client so the +// chat is not simply absent on the phone. Every `agentSession.*` method and destructive close stays +// refused, which is what the tests below pin; the row-level behaviour is pinned in +// src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts. import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' @@ -77,6 +82,11 @@ const STRUCTURED_CALLS: { hostMethod: 'setOption', result: { ok: true, replayed: false } }, + { + method: 'agentSession.requestHandoff', + hostMethod: 'requestHandoff', + result: { status: { owner: 'native' } } + }, { method: 'agentSession.handoffStatus', hostMethod: 'handoffStatus', @@ -197,6 +207,14 @@ function paramsFor(method: string): unknown { const fields = { itemId: 'item-1', expectedRevision: 1, optionId: 'allow' } return { envelope: envelope({ method, fields, fence }), ...fields } } + case 'agentSession.requestHandoff': { + const fields = { + direction: 'to-tui' as const, + mode: 'now' as const, + action: 'start' as const + } + return { envelope: envelope({ method, fields, fence }), ...fields } + } case 'agentSession.setOption': { const fields = { key: 'model', value: 'gpt-5' } return { envelope: envelope({ method, fields, fence }), ...fields } @@ -286,6 +304,10 @@ async function callBuild( function structuredHostStub(): Record> { return { attach: vi.fn(async () => ({ ok: true, replayed: false, value: { sessionId: SESSION } })), + // Attach-shaped entries take a client-supplied location, so the host is asked whether it + // supports creating there. A real host always answers; leaving it unstubbed made every + // `ensure` refuse for the harness's own reason rather than the location's. + supportsCreate: vi.fn(() => true), send: vi.fn(async () => ({ ok: true, replayed: false })), cancel: vi.fn(async () => ({ ok: true, replayed: false })), close: vi.fn(async () => undefined), @@ -723,6 +745,10 @@ describe('cross-version structured agent sessions', () => { /** Phase 2 owns provider processes; the adapter is the only stub here. */ function adapter(): StructuredAgentSessionAdapter { return { + // Every real adapter answers this; without it adapterSupportsCreate falls through to + // `supportsLocation`, which this fake also lacks, so the client-supplied-location gate + // refused for the fake's silence rather than for the location. + supportsCreate: () => true, acquire: async ({ fence }) => ({ process: { hostId: 'local', diff --git a/tests/e2e/new-workspace-create-more.spec.ts b/tests/e2e/new-workspace-create-more.spec.ts new file mode 100644 index 00000000000..b166d61dfe5 --- /dev/null +++ b/tests/e2e/new-workspace-create-more.spec.ts @@ -0,0 +1,106 @@ +import { writeFileSync } from 'node:fs' +import { execFileSync } from 'node:child_process' +import { test, expect } from './helpers/orca-app' +import { waitForActiveWorktree, waitForSessionReady } from './helpers/store' + +test.use({ orcaAppExtraEnv: { ORCA_BACKGROUND_LAUNCH: '1' } }) + +test('Create more clears the GitHub PR source before the next worktree', async ({ + electronApp, + orcaPage, + testRepoPath +}, testInfo) => { + await waitForSessionReady(orcaPage) + await waitForActiveWorktree(orcaPage) + const sha = execFileSync('git', ['rev-parse', 'HEAD'], { + cwd: testRepoPath, + encoding: 'utf8' + }).trim() + await electronApp.evaluate(({ ipcMain }, baseBranch) => { + ipcMain.removeHandler('worktrees:resolvePrBase') + ipcMain.handle('worktrees:resolvePrBase', () => ({ baseBranch })) + }, sha) + await orcaPage.evaluate(() => { + const store = window.__store! + const state = store.getState() + store.setState({ settings: { ...state.settings!, defaultTuiAgent: 'blank' } }) + }) + await orcaPage.getByRole('button', { name: 'New workspace', exact: true }).click() + await orcaPage.evaluate(() => { + const store = window.__store! + const repoId = store.getState().repos[0].id + const item = { + id: 'pr-4242', + provider: 'github' as const, + type: 'pr' as const, + number: 4242, + title: 'Fix workspace task reset', + state: 'open' as const, + url: 'https://github.com/acme/app/pull/4242', + labels: [], + updatedAt: '2026-09-01T00:00:00Z', + author: 'e2e', + repoId + } + store.setState({ + getCachedWorkItems: () => [item], + fetchWorkItems: async () => [item], + fetchWorkItemsAcrossRepos: async () => ({ + items: [item], + failedCount: 0, + githubUnavailable: false + }) + }) + }) + const dialog = orcaPage.getByRole('dialog', { name: /Create (Workspace|Worktree)/i }) + const input = dialog.locator('[data-workspace-name-input="true"]') + await input.click() + await orcaPage + .getByRole('option', { name: '#4242 Fix workspace task reset', exact: true }) + .click() + const pill = dialog.locator('[data-workspace-source-pill="true"]') + await expect(pill).toContainText('Fix workspace task reset') + await dialog.getByRole('switch', { name: 'Create more' }).click() + await dialog.getByRole('button', { name: /^Create/ }).click() + await expect(dialog).toBeVisible() + await expect(input).toHaveValue('') + await expect + .poll(() => + orcaPage.evaluate(() => + window + .__store!.getState() + .allWorktrees() + .some((worktree) => worktree.linkedPR === 4242) + ) + ) + .toBe(true) + const cdp = await orcaPage.context().newCDPSession(orcaPage) + const screenshot = await cdp.send('Page.captureScreenshot') + const proofPath = testInfo.outputPath('create-more-result.png') + writeFileSync(proofPath, Buffer.from(screenshot.data, 'base64')) + await testInfo.attach('create-more-result.png', { + path: proofPath, + contentType: 'image/png' + }) + await cdp.detach() + await expect(pill).toHaveCount(0) + await expect(dialog.getByRole('switch', { name: 'Create more' })).toHaveAttribute( + 'aria-checked', + 'true' + ) + await input.fill('next-independent-worktree') + await dialog.getByRole('button', { name: /^Create/ }).click() + await expect + .poll(() => + orcaPage.evaluate(() => { + const worktree = window + .__store!.getState() + .allWorktrees() + .find((entry) => entry.displayName === 'next-independent-worktree') + return worktree ? { linkedPR: worktree.linkedPR, linkedIssue: worktree.linkedIssue } : null + }) + ) + .toEqual({ linkedPR: null, linkedIssue: null }) + await expect(input).toHaveValue('') + await expect(pill).toHaveCount(0) +}) diff --git a/tests/tools/google-signin-ua-probe.cjs b/tests/tools/google-signin-ua-probe.cjs index 7adf06bb7a0..70c834b3d9d 100644 --- a/tests/tools/google-signin-ua-probe.cjs +++ b/tests/tools/google-signin-ua-probe.cjs @@ -10,8 +10,8 @@ const MODES = new Set([ 'electron-fixed', 'firefox-auth', 'firefox-fixed', - // Replicates the SHIPPED app exactly (setupClientHintsOverride + - // applyGoogleAuthUserAgent): Firefox UA is written to the WebContents on auth + // Replicates the app as it shipped before the UA rewrite was removed + // (cleaned Chrome-shaped session UA + the Google auth Firefox switch): Firefox UA is written to the WebContents on auth // navs and to the request header only for auth-host URLs; every other request // keeps whatever UA the WebContents carries. Logs incoming vs outgoing // identity for ALL requests to expose cross-host mismatches during the flow. @@ -185,7 +185,7 @@ app.whenReady().then(async () => { if (mode === 'app-fixed' && currentUa === identities.firefox) { removeClientHints(headers) } else { - // Real setupClientHintsOverride builds Chrome hints once from the + // The retired client-hints rewrite built Chrome hints once from the // session's cleaned UA (a closure), never from the per-request UA. applyChromeClientHints(headers, identities.cleaned) }